{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QH57IUSWMYBWFY44OTHYZ7EB3M","short_pith_number":"pith:QH57IUSW","schema_version":"1.0","canonical_sha256":"81fbf45256660362e39c74cf8cfc81db1bd9a0ca83d5ef07134d335f24aec870","source":{"kind":"arxiv","id":"2507.05714","version":3},"attestation_state":"computed","paper":{"title":"HIRAG: Hierarchical-Thought Instruction-Tuning Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dan Yang, Duolin Sun, Jian Wang, Jie Feng, Peng Wei, Yihan Jiao, Yue Shen, Zhehao Tan","submitted_at":"2025-07-08T06:53:28Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has become a fundamental paradigm for addressing the challenges faced by large language models in handling real-time information and domain-specific problems. Traditional RAG systems primarily rely on the in-context learning (ICL) capabilities of the large language model itself. Still, in-depth research on the specific capabilities needed by the RAG generation model is lacking, leading to challenges with inconsistent document quality and retrieval system imperfections. Even the limited studies that fine-tune RAG generative models often \\textit{lack a granul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.05714","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-08T06:53:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ce06e0a517f97e2d11d8131677b50b150c4d14e1b097f5f3ceeea2dc57cb61e5","abstract_canon_sha256":"5b6a2c114e402802f255bf214cf79cb0e7e9ca1f179d897688e45b92c1255b98"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:08:05.203725Z","signature_b64":"EbW1ECB6ghnqiI6+3k3sYsYEWLkt4f5YnvJlsL5+EgV1rZP4DdaHDgCvbcVmdBPwHaGvT6WoLBtTxJI4VXQaAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81fbf45256660362e39c74cf8cfc81db1bd9a0ca83d5ef07134d335f24aec870","last_reissued_at":"2026-07-05T12:08:05.203128Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:08:05.203128Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HIRAG: Hierarchical-Thought Instruction-Tuning Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dan Yang, Duolin Sun, Jian Wang, Jie Feng, Peng Wei, Yihan Jiao, Yue Shen, Zhehao Tan","submitted_at":"2025-07-08T06:53:28Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has become a fundamental paradigm for addressing the challenges faced by large language models in handling real-time information and domain-specific problems. Traditional RAG systems primarily rely on the in-context learning (ICL) capabilities of the large language model itself. Still, in-depth research on the specific capabilities needed by the RAG generation model is lacking, leading to challenges with inconsistent document quality and retrieval system imperfections. Even the limited studies that fine-tune RAG generative models often \\textit{lack a granul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.05714","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.05714/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.05714","created_at":"2026-07-05T12:08:05.203214+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.05714v3","created_at":"2026-07-05T12:08:05.203214+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.05714","created_at":"2026-07-05T12:08:05.203214+00:00"},{"alias_kind":"pith_short_12","alias_value":"QH57IUSWMYBW","created_at":"2026-07-05T12:08:05.203214+00:00"},{"alias_kind":"pith_short_16","alias_value":"QH57IUSWMYBWFY44","created_at":"2026-07-05T12:08:05.203214+00:00"},{"alias_kind":"pith_short_8","alias_value":"QH57IUSW","created_at":"2026-07-05T12:08:05.203214+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08979","citing_title":"EviProp: Seeded Relevance Diffusion on Chunk-Page Graphs for Long Multimodal Document Retrieval","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28447","citing_title":"SemFlowRAG: Directed Semantic Flow from Abstraction to Evidence for Complex Reasoning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11653","citing_title":"GroupRank: A Groupwise Paradigm for Effective and Efficient Passage Reranking with LLMs","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M","json":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M.json","graph_json":"https://pith.science/api/pith-number/QH57IUSWMYBWFY44OTHYZ7EB3M/graph.json","events_json":"https://pith.science/api/pith-number/QH57IUSWMYBWFY44OTHYZ7EB3M/events.json","paper":"https://pith.science/paper/QH57IUSW"},"agent_actions":{"view_html":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M","download_json":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M.json","view_paper":"https://pith.science/paper/QH57IUSW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.05714&json=true","fetch_graph":"https://pith.science/api/pith-number/QH57IUSWMYBWFY44OTHYZ7EB3M/graph.json","fetch_events":"https://pith.science/api/pith-number/QH57IUSWMYBWFY44OTHYZ7EB3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M/action/storage_attestation","attest_author":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M/action/author_attestation","sign_citation":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M/action/citation_signature","submit_replication":"https://pith.science/pith/QH57IUSWMYBWFY44OTHYZ7EB3M/action/replication_record"}},"created_at":"2026-07-05T12:08:05.203214+00:00","updated_at":"2026-07-05T12:08:05.203214+00:00"}