{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:V7CVRBQNIRS37RONRSSAA5L4P4","short_pith_number":"pith:V7CVRBQN","schema_version":"1.0","canonical_sha256":"afc558860d4465bfc5cd8ca400757c7f05f878bde94d993a8a6cc20c4dc07a49","source":{"kind":"arxiv","id":"2505.22019","version":2},"attestation_state":"computed","paper":{"title":"VRAG-RL: Empower Vision-Perception-Based RAG for Visually Rich Information Understanding via Iterative Reasoning with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Feng Zhao, Lin Chen, Pengjun Xie, Qiuchen Wang, Ruixue Ding, Shihang Wang, Yu Zeng, Zehui Chen","submitted_at":"2025-05-28T06:30:51Z","abstract_excerpt":"Effectively retrieving, reasoning and understanding visually rich information remains a challenge for RAG methods. Traditional text-based methods cannot handle visual-related information. On the other hand, current vision-based RAG approaches are often limited by fixed pipelines and frequently struggle to reason effectively due to the insufficient activation of the fundamental capabilities of models. As RL has been proven to be beneficial for model reasoning, we introduce VRAG-RL, a novel RL framework tailored for complex reasoning across visually rich information. With this framework, VLMs in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22019","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T06:30:51Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"f86e42d8b4fc181bd80c7762476b10efe05d7e579cf7e68d294644b613f3e6e7","abstract_canon_sha256":"0f5716110f5df4598c1c40abf6d77a72c280f53a81e0b5ed14ba3941cea5556e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:00.585650Z","signature_b64":"ujKxABwLGzFCQ2bAbwk7ZYoac8rvoH4mbKFHGxQbY8NXwZglIDZBI1CcEUKF6WszG1mlhFcxxvYpad6mFtHcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afc558860d4465bfc5cd8ca400757c7f05f878bde94d993a8a6cc20c4dc07a49","last_reissued_at":"2026-07-05T11:15:00.585113Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:00.585113Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VRAG-RL: Empower Vision-Perception-Based RAG for Visually Rich Information Understanding via Iterative Reasoning with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Feng Zhao, Lin Chen, Pengjun Xie, Qiuchen Wang, Ruixue Ding, Shihang Wang, Yu Zeng, Zehui Chen","submitted_at":"2025-05-28T06:30:51Z","abstract_excerpt":"Effectively retrieving, reasoning and understanding visually rich information remains a challenge for RAG methods. Traditional text-based methods cannot handle visual-related information. On the other hand, current vision-based RAG approaches are often limited by fixed pipelines and frequently struggle to reason effectively due to the insufficient activation of the fundamental capabilities of models. As RL has been proven to be beneficial for model reasoning, we introduce VRAG-RL, a novel RL framework tailored for complex reasoning across visually rich information. With this framework, VLMs in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22019","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22019/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22019","created_at":"2026-07-05T11:15:00.585170+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22019v2","created_at":"2026-07-05T11:15:00.585170+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22019","created_at":"2026-07-05T11:15:00.585170+00:00"},{"alias_kind":"pith_short_12","alias_value":"V7CVRBQNIRS3","created_at":"2026-07-05T11:15:00.585170+00:00"},{"alias_kind":"pith_short_16","alias_value":"V7CVRBQNIRS37RON","created_at":"2026-07-05T11:15:00.585170+00:00"},{"alias_kind":"pith_short_8","alias_value":"V7CVRBQN","created_at":"2026-07-05T11:15:00.585170+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":215,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31504","citing_title":"SimpleSearch-VL: A Simple Recipe for Multimodal Agentic Deep Search","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15710","citing_title":"SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2511.05271","citing_title":"DeepEyesV2: Toward Agentic Multimodal Model","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04017","citing_title":"GeoBrowse: A Geolocation Benchmark for Agentic Tool Use with Expert-Annotated Reasoning Traces","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03903","citing_title":"CC-OCR V2: Benchmarking Large Multimodal Models for Literacy in Real-world Document Processing","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07419","citing_title":"ReAlign: Optimizing the Visual Document Retriever with Reasoning-Guided Fine-Grained Alignment","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09508","citing_title":"VISOR: Agentic Visual Retrieval-Augmented Generation via Iterative Search and Over-horizon Reasoning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14029","citing_title":"POINTS-Seeker: An Open Recipe for Multimodal Search Agents with Visual Memory Management","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4","json":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4.json","graph_json":"https://pith.science/api/pith-number/V7CVRBQNIRS37RONRSSAA5L4P4/graph.json","events_json":"https://pith.science/api/pith-number/V7CVRBQNIRS37RONRSSAA5L4P4/events.json","paper":"https://pith.science/paper/V7CVRBQN"},"agent_actions":{"view_html":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4","download_json":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4.json","view_paper":"https://pith.science/paper/V7CVRBQN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22019&json=true","fetch_graph":"https://pith.science/api/pith-number/V7CVRBQNIRS37RONRSSAA5L4P4/graph.json","fetch_events":"https://pith.science/api/pith-number/V7CVRBQNIRS37RONRSSAA5L4P4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4/action/storage_attestation","attest_author":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4/action/author_attestation","sign_citation":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4/action/citation_signature","submit_replication":"https://pith.science/pith/V7CVRBQNIRS37RONRSSAA5L4P4/action/replication_record"}},"created_at":"2026-07-05T11:15:00.585170+00:00","updated_at":"2026-07-05T11:15:00.585170+00:00"}