{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CNY2Q3ZHR5UG76ZT34WDXGYTIW","short_pith_number":"pith:CNY2Q3ZH","schema_version":"1.0","canonical_sha256":"1371a86f278f686ffb33df2c3b9b1345b7d5d6fe8d6b822687c37b3ed9b81606","source":{"kind":"arxiv","id":"2505.04846","version":1},"attestation_state":"computed","paper":{"title":"HiPerRAG: High-Performance Retrieval Augmented Generation for Scientific Insights","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CE","cs.CL","cs.DC","cs.LG"],"primary_cat":"cs.IR","authors_text":"Alexander Brace, Anima Anandkumar, Arvind Ramanathan, Aswathy Ajith, Azton Wells, Bharat Kale, Brian Hsu, Carlo Siebenschuh, Francis J. Alexander, Heng Ma, Huihuo Zheng, Ian Foster, J. Gregory Pauloski, Kyle Hippe, Michael E. Papka, Nicholas Chia, Ozan Gokdemir, Priyanka V. Setty, Rick Stevens, Sam Foreman, Thomas Brettin, Thomas Gibbs, Varuni Sastry, Venkatram Vishwanath","submitted_at":"2025-05-07T22:50:23Z","abstract_excerpt":"The volume of scientific literature is growing exponentially, leading to underutilized discoveries, duplicated efforts, and limited cross-disciplinary collaboration. Retrieval Augmented Generation (RAG) offers a way to assist scientists by improving the factuality of Large Language Models (LLMs) in processing this influx of information. However, scaling RAG to handle millions of articles introduces significant challenges, including the high computational costs associated with parsing documents and embedding scientific knowledge, as well as the algorithmic complexity of aligning these represent"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.04846","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.IR","submitted_at":"2025-05-07T22:50:23Z","cross_cats_sorted":["cs.CE","cs.CL","cs.DC","cs.LG"],"title_canon_sha256":"83e8d6599de89723b456f2eadda63c82d32c00b826bf440bce7b2e59af262d55","abstract_canon_sha256":"cebdc6d6a88114a81f03526000f3099e6cbba888c2652c23d305f15de0074a1a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:14.633582Z","signature_b64":"EGKzMTE+WMX9FP7MIr7M4i2cY/ZXZMNsbGcrovp1rTfT5yxCcy9Lu6WVTBLP6BdO544l1H46wvxEpS90m2TjAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1371a86f278f686ffb33df2c3b9b1345b7d5d6fe8d6b822687c37b3ed9b81606","last_reissued_at":"2026-07-05T11:00:14.633082Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:14.633082Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HiPerRAG: High-Performance Retrieval Augmented Generation for Scientific Insights","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CE","cs.CL","cs.DC","cs.LG"],"primary_cat":"cs.IR","authors_text":"Alexander Brace, Anima Anandkumar, Arvind Ramanathan, Aswathy Ajith, Azton Wells, Bharat Kale, Brian Hsu, Carlo Siebenschuh, Francis J. Alexander, Heng Ma, Huihuo Zheng, Ian Foster, J. Gregory Pauloski, Kyle Hippe, Michael E. Papka, Nicholas Chia, Ozan Gokdemir, Priyanka V. Setty, Rick Stevens, Sam Foreman, Thomas Brettin, Thomas Gibbs, Varuni Sastry, Venkatram Vishwanath","submitted_at":"2025-05-07T22:50:23Z","abstract_excerpt":"The volume of scientific literature is growing exponentially, leading to underutilized discoveries, duplicated efforts, and limited cross-disciplinary collaboration. Retrieval Augmented Generation (RAG) offers a way to assist scientists by improving the factuality of Large Language Models (LLMs) in processing this influx of information. However, scaling RAG to handle millions of articles introduces significant challenges, including the high computational costs associated with parsing documents and embedding scientific knowledge, as well as the algorithmic complexity of aligning these represent"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.04846","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.04846/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.04846","created_at":"2026-07-05T11:00:14.633137+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.04846v1","created_at":"2026-07-05T11:00:14.633137+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.04846","created_at":"2026-07-05T11:00:14.633137+00:00"},{"alias_kind":"pith_short_12","alias_value":"CNY2Q3ZHR5UG","created_at":"2026-07-05T11:00:14.633137+00:00"},{"alias_kind":"pith_short_16","alias_value":"CNY2Q3ZHR5UG76ZT","created_at":"2026-07-05T11:00:14.633137+00:00"},{"alias_kind":"pith_short_8","alias_value":"CNY2Q3ZH","created_at":"2026-07-05T11:00:14.633137+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16347","citing_title":"HPC-LLM: Practical Domain Adaptation and Retrieval-Augmented Generation for HPC Support","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW","json":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW.json","graph_json":"https://pith.science/api/pith-number/CNY2Q3ZHR5UG76ZT34WDXGYTIW/graph.json","events_json":"https://pith.science/api/pith-number/CNY2Q3ZHR5UG76ZT34WDXGYTIW/events.json","paper":"https://pith.science/paper/CNY2Q3ZH"},"agent_actions":{"view_html":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW","download_json":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW.json","view_paper":"https://pith.science/paper/CNY2Q3ZH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.04846&json=true","fetch_graph":"https://pith.science/api/pith-number/CNY2Q3ZHR5UG76ZT34WDXGYTIW/graph.json","fetch_events":"https://pith.science/api/pith-number/CNY2Q3ZHR5UG76ZT34WDXGYTIW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW/action/storage_attestation","attest_author":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW/action/author_attestation","sign_citation":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW/action/citation_signature","submit_replication":"https://pith.science/pith/CNY2Q3ZHR5UG76ZT34WDXGYTIW/action/replication_record"}},"created_at":"2026-07-05T11:00:14.633137+00:00","updated_at":"2026-07-05T11:00:14.633137+00:00"}