{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:2QUBORBG3YDSG3PNIUPUKJI3L7","short_pith_number":"pith:2QUBORBG","schema_version":"1.0","canonical_sha256":"d428174426de07236ded451f45251b5feb013b3f42f297c7ec37094e719ad4c8","source":{"kind":"arxiv","id":"2603.05256","version":2},"attestation_state":"computed","paper":{"title":"Wiki-R1: Incentivizing Multimodal Reasoning for Knowledge-based VQA via Data and Sampling Curriculum","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Longtian Qiu, Shan Ning, Xuming He","submitted_at":"2026-03-05T15:08:06Z","abstract_excerpt":"Knowledge-Based Visual Question Answering (KB-VQA) requires models to answer questions about an image by integrating external knowledge, posing significant challenges due to noisy retrieval and the structured, encyclopedic nature of the knowledge base. These characteristics create a distributional gap from pretrained multimodal large language models (MLLMs), making effective reasoning and domain adaptation difficult in the post-training stage. In this work, we propose \\textit{Wiki-R1}, a data-generation-based curriculum reinforcement learning framework that systematically incentivizes reasonin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.05256","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-03-05T15:08:06Z","cross_cats_sorted":[],"title_canon_sha256":"7b2e53911fdbe7a2ee253afacc04cce896b6d3438d594dd42c76d293d3a963d9","abstract_canon_sha256":"7d9d1b9288005432efc050d4bf223bf998f134ef75e373bc2a45bb58ecaffc8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-03T01:17:54.639743Z","signature_b64":"IZku8vD6Krzj7sn6ratYgjZnbqNatVvk7wLhqNA3bDUk5hONZ9swibB4F2zOb6D/4sEt0lIevDGzGbs+F11dCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d428174426de07236ded451f45251b5feb013b3f42f297c7ec37094e719ad4c8","last_reissued_at":"2026-07-03T01:17:54.639205Z","signature_status":"signed_v1","first_computed_at":"2026-07-03T01:17:54.639205Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Wiki-R1: Incentivizing Multimodal Reasoning for Knowledge-based VQA via Data and Sampling Curriculum","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Longtian Qiu, Shan Ning, Xuming He","submitted_at":"2026-03-05T15:08:06Z","abstract_excerpt":"Knowledge-Based Visual Question Answering (KB-VQA) requires models to answer questions about an image by integrating external knowledge, posing significant challenges due to noisy retrieval and the structured, encyclopedic nature of the knowledge base. These characteristics create a distributional gap from pretrained multimodal large language models (MLLMs), making effective reasoning and domain adaptation difficult in the post-training stage. In this work, we propose \\textit{Wiki-R1}, a data-generation-based curriculum reinforcement learning framework that systematically incentivizes reasonin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.05256","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.05256/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.05256","created_at":"2026-07-03T01:17:54.639270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.05256v2","created_at":"2026-07-03T01:17:54.639270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.05256","created_at":"2026-07-03T01:17:54.639270+00:00"},{"alias_kind":"pith_short_12","alias_value":"2QUBORBG3YDS","created_at":"2026-07-03T01:17:54.639270+00:00"},{"alias_kind":"pith_short_16","alias_value":"2QUBORBG3YDSG3PN","created_at":"2026-07-03T01:17:54.639270+00:00"},{"alias_kind":"pith_short_8","alias_value":"2QUBORBG","created_at":"2026-07-03T01:17:54.639270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2603.09921","citing_title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7","json":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7.json","graph_json":"https://pith.science/api/pith-number/2QUBORBG3YDSG3PNIUPUKJI3L7/graph.json","events_json":"https://pith.science/api/pith-number/2QUBORBG3YDSG3PNIUPUKJI3L7/events.json","paper":"https://pith.science/paper/2QUBORBG"},"agent_actions":{"view_html":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7","download_json":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7.json","view_paper":"https://pith.science/paper/2QUBORBG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.05256&json=true","fetch_graph":"https://pith.science/api/pith-number/2QUBORBG3YDSG3PNIUPUKJI3L7/graph.json","fetch_events":"https://pith.science/api/pith-number/2QUBORBG3YDSG3PNIUPUKJI3L7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7/action/storage_attestation","attest_author":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7/action/author_attestation","sign_citation":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7/action/citation_signature","submit_replication":"https://pith.science/pith/2QUBORBG3YDSG3PNIUPUKJI3L7/action/replication_record"}},"created_at":"2026-07-03T01:17:54.639270+00:00","updated_at":"2026-07-03T01:17:54.639270+00:00"}