{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LUGEUHDR4JDJE7OFSEXKGDWO45","short_pith_number":"pith:LUGEUHDR","schema_version":"1.0","canonical_sha256":"5d0c4a1c71e246927dc5912ea30ecee7612f7a1a3bc532774dbaa752d5c14769","source":{"kind":"arxiv","id":"2412.06206","version":2},"attestation_state":"computed","paper":{"title":"SiReRAG: Indexing Similar and Related Information for Multihop Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander Fabbri, Caiming Xiong, Chien-Sheng Wu, Gabriel Bernadett-Shapiro, Nan Zhang, Prafulla Kumar Choubey, Prasenjit Mitra, Rui Zhang","submitted_at":"2024-12-09T04:56:43Z","abstract_excerpt":"Indexing is an important step towards strong performance in retrieval-augmented generation (RAG) systems. However, existing methods organize data based on either semantic similarity (similarity) or related information (relatedness), but do not cover both perspectives comprehensively. Our analysis reveals that modeling only one perspective results in insufficient knowledge synthesis, leading to suboptimal performance on complex tasks requiring multihop reasoning. In this paper, we propose SiReRAG, a novel RAG indexing approach that explicitly considers both similar and related information. On t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.06206","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-09T04:56:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"591a1119a491232455bbdaa010fd98c2af7135a508c43d48c263370f1ffe8328","abstract_canon_sha256":"4185394a7207205937ba7c7746a11a5f97d0132c3acb27ab7f89634e6266e542"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:38.669250Z","signature_b64":"hcf/6nkc13ZqLyOHoJZTjqwumlbJ6J+jpBSfep0JlrAJbwjilloZ4guwhbQZZMFp2pJNvqBVrPL9mW93uOkLAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d0c4a1c71e246927dc5912ea30ecee7612f7a1a3bc532774dbaa752d5c14769","last_reissued_at":"2026-07-05T10:45:38.668711Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:38.668711Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SiReRAG: Indexing Similar and Related Information for Multihop Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander Fabbri, Caiming Xiong, Chien-Sheng Wu, Gabriel Bernadett-Shapiro, Nan Zhang, Prafulla Kumar Choubey, Prasenjit Mitra, Rui Zhang","submitted_at":"2024-12-09T04:56:43Z","abstract_excerpt":"Indexing is an important step towards strong performance in retrieval-augmented generation (RAG) systems. However, existing methods organize data based on either semantic similarity (similarity) or related information (relatedness), but do not cover both perspectives comprehensively. Our analysis reveals that modeling only one perspective results in insufficient knowledge synthesis, leading to suboptimal performance on complex tasks requiring multihop reasoning. In this paper, we propose SiReRAG, a novel RAG indexing approach that explicitly considers both similar and related information. On t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.06206","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.06206/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.06206","created_at":"2026-07-05T10:45:38.668774+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.06206v2","created_at":"2026-07-05T10:45:38.668774+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.06206","created_at":"2026-07-05T10:45:38.668774+00:00"},{"alias_kind":"pith_short_12","alias_value":"LUGEUHDR4JDJ","created_at":"2026-07-05T10:45:38.668774+00:00"},{"alias_kind":"pith_short_16","alias_value":"LUGEUHDR4JDJE7OF","created_at":"2026-07-05T10:45:38.668774+00:00"},{"alias_kind":"pith_short_8","alias_value":"LUGEUHDR","created_at":"2026-07-05T10:45:38.668774+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28365","citing_title":"CAMI: Cost-Aware Agent-Guided Multi-Indexing for Semantic Retrieval","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09891","citing_title":"ArchRAG: Attributed Community-based Hierarchical Retrieval-Augmented Generation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17934","citing_title":"AtlasKV: Augmenting LLMs with Billion-Scale Knowledge Graphs in 20GB VRAM","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45","json":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45.json","graph_json":"https://pith.science/api/pith-number/LUGEUHDR4JDJE7OFSEXKGDWO45/graph.json","events_json":"https://pith.science/api/pith-number/LUGEUHDR4JDJE7OFSEXKGDWO45/events.json","paper":"https://pith.science/paper/LUGEUHDR"},"agent_actions":{"view_html":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45","download_json":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45.json","view_paper":"https://pith.science/paper/LUGEUHDR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.06206&json=true","fetch_graph":"https://pith.science/api/pith-number/LUGEUHDR4JDJE7OFSEXKGDWO45/graph.json","fetch_events":"https://pith.science/api/pith-number/LUGEUHDR4JDJE7OFSEXKGDWO45/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45/action/storage_attestation","attest_author":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45/action/author_attestation","sign_citation":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45/action/citation_signature","submit_replication":"https://pith.science/pith/LUGEUHDR4JDJE7OFSEXKGDWO45/action/replication_record"}},"created_at":"2026-07-05T10:45:38.668774+00:00","updated_at":"2026-07-05T10:45:38.668774+00:00"}