{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PYBWS5OD55ZGH4GE6WAQ4BVFFI","short_pith_number":"pith:PYBWS5OD","schema_version":"1.0","canonical_sha256":"7e036975c3ef7263f0c4f5810e06a52a19a2bb5f9d114ce35317860baf14ec38","source":{"kind":"arxiv","id":"2411.06037","version":3},"attestation_state":"computed","paper":{"title":"Sufficient Context: A New Lens on Retrieval Augmented Generation Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ankur Taly, Chun-Sung Ferng, Cyrus Rashtchian, Da-Cheng Juan, Hailey Joren, Jianyi Zhang","submitted_at":"2024-11-09T02:13:14Z","abstract_excerpt":"Augmenting LLMs with context leads to improved performance across many applications. Despite much research on Retrieval Augmented Generation (RAG) systems, an open question is whether errors arise because LLMs fail to utilize the context from retrieval or the context itself is insufficient to answer the query. To shed light on this, we develop a new notion of sufficient context, along with a method to classify instances that have enough information to answer the query. We then use sufficient context to analyze several models and datasets. By stratifying errors based on context sufficiency, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.06037","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-09T02:13:14Z","cross_cats_sorted":[],"title_canon_sha256":"691072e550d94b9c9953fe98c2611375e826048c1a687522b407c24e9ce13114","abstract_canon_sha256":"86e4a97656bbb5b9aa5a38a4e5b011fb66d5ebbe9c2464b59f424ef77eb0ab2a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:38.297897Z","signature_b64":"dZHE3EG6cA27D5C1jBHi8cyR+F+6US1NnYrQ6W7KgAQxm2DMc/JvRfgsvZwanUlXVEqz2uBFfwoHx4gZFc3hDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e036975c3ef7263f0c4f5810e06a52a19a2bb5f9d114ce35317860baf14ec38","last_reissued_at":"2026-07-05T10:52:38.297418Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:38.297418Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sufficient Context: A New Lens on Retrieval Augmented Generation Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ankur Taly, Chun-Sung Ferng, Cyrus Rashtchian, Da-Cheng Juan, Hailey Joren, Jianyi Zhang","submitted_at":"2024-11-09T02:13:14Z","abstract_excerpt":"Augmenting LLMs with context leads to improved performance across many applications. Despite much research on Retrieval Augmented Generation (RAG) systems, an open question is whether errors arise because LLMs fail to utilize the context from retrieval or the context itself is insufficient to answer the query. To shed light on this, we develop a new notion of sufficient context, along with a method to classify instances that have enough information to answer the query. We then use sufficient context to analyze several models and datasets. By stratifying errors based on context sufficiency, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.06037","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.06037/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.06037","created_at":"2026-07-05T10:52:38.297477+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.06037v3","created_at":"2026-07-05T10:52:38.297477+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.06037","created_at":"2026-07-05T10:52:38.297477+00:00"},{"alias_kind":"pith_short_12","alias_value":"PYBWS5OD55ZG","created_at":"2026-07-05T10:52:38.297477+00:00"},{"alias_kind":"pith_short_16","alias_value":"PYBWS5OD55ZGH4GE","created_at":"2026-07-05T10:52:38.297477+00:00"},{"alias_kind":"pith_short_8","alias_value":"PYBWS5OD","created_at":"2026-07-05T10:52:38.297477+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23716","citing_title":"Legal Reasoning Is Not Lawyering: Rethinking Legal Benchmarks for Pro Se Access to Justice","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04109","citing_title":"Discourse-Role Labels as Presentation-Time Variables for Context Use in Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29875","citing_title":"UnWeaving the knots of GraphRAG -- turns out VectorRAG is almost enough","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13052","citing_title":"RAG-Enhanced Large Language Models for Dynamic Content Expiration Prediction in Web Search","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17227","citing_title":"Cloud-native and Distributed Systems for Efficient and Scalable Large Language Models -- A Research Agenda","ref_index":115,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI","json":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI.json","graph_json":"https://pith.science/api/pith-number/PYBWS5OD55ZGH4GE6WAQ4BVFFI/graph.json","events_json":"https://pith.science/api/pith-number/PYBWS5OD55ZGH4GE6WAQ4BVFFI/events.json","paper":"https://pith.science/paper/PYBWS5OD"},"agent_actions":{"view_html":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI","download_json":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI.json","view_paper":"https://pith.science/paper/PYBWS5OD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.06037&json=true","fetch_graph":"https://pith.science/api/pith-number/PYBWS5OD55ZGH4GE6WAQ4BVFFI/graph.json","fetch_events":"https://pith.science/api/pith-number/PYBWS5OD55ZGH4GE6WAQ4BVFFI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI/action/storage_attestation","attest_author":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI/action/author_attestation","sign_citation":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI/action/citation_signature","submit_replication":"https://pith.science/pith/PYBWS5OD55ZGH4GE6WAQ4BVFFI/action/replication_record"}},"created_at":"2026-07-05T10:52:38.297477+00:00","updated_at":"2026-07-05T10:52:38.297477+00:00"}