{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HQQLQIWHL4VBNN3THBY57YHKMK","short_pith_number":"pith:HQQLQIWH","schema_version":"1.0","canonical_sha256":"3c20b822c75f2a16b7733871dfe0ea62946e00c1d401bdb7c1b5167a16ed8515","source":{"kind":"arxiv","id":"2506.07785","version":1},"attestation_state":"computed","paper":{"title":"Re-ranking Reasoning Context with Tree Search Makes Large Vision-Language Models Stronger","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chenghao Zhang, Jieping Ye, Kun Ding, Lubin Fan, Qi Yang, Shiming Xiang","submitted_at":"2025-06-09T14:00:57Z","abstract_excerpt":"Recent advancements in Large Vision Language Models (LVLMs) have significantly improved performance in Visual Question Answering (VQA) tasks through multimodal Retrieval-Augmented Generation (RAG). However, existing methods still face challenges, such as the scarcity of knowledge with reasoning examples and erratic responses from retrieved knowledge. To address these issues, in this study, we propose a multimodal RAG framework, termed RCTS, which enhances LVLMs by constructing a Reasoning Context-enriched knowledge base and a Tree Search re-ranking method. Specifically, we introduce a self-con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07785","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-09T14:00:57Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"724dadc9997173e08d9fb4ad631b0c1c4e19c53490acd878c9ba127fefe60faf","abstract_canon_sha256":"e6deb153351261b38a8924fc37f2327587713db32466bf17d3a5b9077ac646d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:33.384167Z","signature_b64":"4dcsUgYa6ZyPRLFOHOd0kb6Oa5JolXFRIGkJ4Qw3V+/tas1vOCEg1z/NQxEacuAsmFDehgOCKZGpMbaFKOXoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c20b822c75f2a16b7733871dfe0ea62946e00c1d401bdb7c1b5167a16ed8515","last_reissued_at":"2026-07-05T11:18:33.383656Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:33.383656Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Re-ranking Reasoning Context with Tree Search Makes Large Vision-Language Models Stronger","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chenghao Zhang, Jieping Ye, Kun Ding, Lubin Fan, Qi Yang, Shiming Xiang","submitted_at":"2025-06-09T14:00:57Z","abstract_excerpt":"Recent advancements in Large Vision Language Models (LVLMs) have significantly improved performance in Visual Question Answering (VQA) tasks through multimodal Retrieval-Augmented Generation (RAG). However, existing methods still face challenges, such as the scarcity of knowledge with reasoning examples and erratic responses from retrieved knowledge. To address these issues, in this study, we propose a multimodal RAG framework, termed RCTS, which enhances LVLMs by constructing a Reasoning Context-enriched knowledge base and a Tree Search re-ranking method. Specifically, we introduce a self-con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07785","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07785/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07785","created_at":"2026-07-05T11:18:33.383719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07785v1","created_at":"2026-07-05T11:18:33.383719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07785","created_at":"2026-07-05T11:18:33.383719+00:00"},{"alias_kind":"pith_short_12","alias_value":"HQQLQIWHL4VB","created_at":"2026-07-05T11:18:33.383719+00:00"},{"alias_kind":"pith_short_16","alias_value":"HQQLQIWHL4VBNN3T","created_at":"2026-07-05T11:18:33.383719+00:00"},{"alias_kind":"pith_short_8","alias_value":"HQQLQIWH","created_at":"2026-07-05T11:18:33.383719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK","json":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK.json","graph_json":"https://pith.science/api/pith-number/HQQLQIWHL4VBNN3THBY57YHKMK/graph.json","events_json":"https://pith.science/api/pith-number/HQQLQIWHL4VBNN3THBY57YHKMK/events.json","paper":"https://pith.science/paper/HQQLQIWH"},"agent_actions":{"view_html":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK","download_json":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK.json","view_paper":"https://pith.science/paper/HQQLQIWH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07785&json=true","fetch_graph":"https://pith.science/api/pith-number/HQQLQIWHL4VBNN3THBY57YHKMK/graph.json","fetch_events":"https://pith.science/api/pith-number/HQQLQIWHL4VBNN3THBY57YHKMK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK/action/storage_attestation","attest_author":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK/action/author_attestation","sign_citation":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK/action/citation_signature","submit_replication":"https://pith.science/pith/HQQLQIWHL4VBNN3THBY57YHKMK/action/replication_record"}},"created_at":"2026-07-05T11:18:33.383719+00:00","updated_at":"2026-07-05T11:18:33.383719+00:00"}