{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:INNWAIVVEA4BBPY5KE5TMSIE55","short_pith_number":"pith:INNWAIVV","schema_version":"1.0","canonical_sha256":"435b6022b5203810bf1d513b364904ef44ae092bfe42f784e7f50daed114f545","source":{"kind":"arxiv","id":"2508.06328","version":1},"attestation_state":"computed","paper":{"title":"M2IO-R1: An Efficient RL-Enhanced Reasoning Framework for Multimodal Retrieval Augmented Multimodal Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Binghui Li, Chong Chen, Geng Chen, Qinhan Yu, Wentao Zhang, Zhiyou Xiao","submitted_at":"2025-08-08T14:00:19Z","abstract_excerpt":"Current research on Multimodal Retrieval-Augmented Generation (MRAG) enables diverse multimodal inputs but remains limited to single-modality outputs, restricting expressive capacity and practical utility. In contrast, real-world applications often demand both multimodal inputs and multimodal outputs for effective communication and grounded reasoning. Motivated by the recent success of Reinforcement Learning (RL) in complex reasoning tasks for Large Language Models (LLMs), we adopt RL as a principled and effective paradigm to address the multi-step, outcome-driven challenges inherent in multim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.06328","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2025-08-08T14:00:19Z","cross_cats_sorted":[],"title_canon_sha256":"ca85b00c5dccaac823a7e2e8002bb6f2dc06de4c3cadf776046b506bf6e93fa4","abstract_canon_sha256":"d3d5b2f5f8614c2dba78f29e8d65f58d27b53efadf1bbb4d055f30797824710e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:54.301704Z","signature_b64":"s2zutrSSUVSi4eqkLsx4RHDFH4pUWnzMO9T2QOfmO/2S37GpyO05oyoiJKGVAPyUcWnIJVG4TtZR9Y36n02CCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"435b6022b5203810bf1d513b364904ef44ae092bfe42f784e7f50daed114f545","last_reissued_at":"2026-07-05T11:50:54.301130Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:54.301130Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"M2IO-R1: An Efficient RL-Enhanced Reasoning Framework for Multimodal Retrieval Augmented Multimodal Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Binghui Li, Chong Chen, Geng Chen, Qinhan Yu, Wentao Zhang, Zhiyou Xiao","submitted_at":"2025-08-08T14:00:19Z","abstract_excerpt":"Current research on Multimodal Retrieval-Augmented Generation (MRAG) enables diverse multimodal inputs but remains limited to single-modality outputs, restricting expressive capacity and practical utility. In contrast, real-world applications often demand both multimodal inputs and multimodal outputs for effective communication and grounded reasoning. Motivated by the recent success of Reinforcement Learning (RL) in complex reasoning tasks for Large Language Models (LLMs), we adopt RL as a principled and effective paradigm to address the multi-step, outcome-driven challenges inherent in multim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.06328","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.06328/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.06328","created_at":"2026-07-05T11:50:54.301191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.06328v1","created_at":"2026-07-05T11:50:54.301191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.06328","created_at":"2026-07-05T11:50:54.301191+00:00"},{"alias_kind":"pith_short_12","alias_value":"INNWAIVVEA4B","created_at":"2026-07-05T11:50:54.301191+00:00"},{"alias_kind":"pith_short_16","alias_value":"INNWAIVVEA4BBPY5","created_at":"2026-07-05T11:50:54.301191+00:00"},{"alias_kind":"pith_short_8","alias_value":"INNWAIVV","created_at":"2026-07-05T11:50:54.301191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.28767","citing_title":"Gen-Searcher: Reinforcing Agentic Search for Image Generation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15253","citing_title":"Scaling Beyond Context: A Survey of Multimodal Retrieval-Augmented Generation for Document Understanding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28767","citing_title":"Gen-Searcher: Reinforcing Agentic Search for Image Generation","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55","json":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55.json","graph_json":"https://pith.science/api/pith-number/INNWAIVVEA4BBPY5KE5TMSIE55/graph.json","events_json":"https://pith.science/api/pith-number/INNWAIVVEA4BBPY5KE5TMSIE55/events.json","paper":"https://pith.science/paper/INNWAIVV"},"agent_actions":{"view_html":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55","download_json":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55.json","view_paper":"https://pith.science/paper/INNWAIVV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.06328&json=true","fetch_graph":"https://pith.science/api/pith-number/INNWAIVVEA4BBPY5KE5TMSIE55/graph.json","fetch_events":"https://pith.science/api/pith-number/INNWAIVVEA4BBPY5KE5TMSIE55/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55/action/timestamp_anchor","attest_storage":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55/action/storage_attestation","attest_author":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55/action/author_attestation","sign_citation":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55/action/citation_signature","submit_replication":"https://pith.science/pith/INNWAIVVEA4BBPY5KE5TMSIE55/action/replication_record"}},"created_at":"2026-07-05T11:50:54.301191+00:00","updated_at":"2026-07-05T11:50:54.301191+00:00"}