{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K63XQGH2GEHMWW3BESSX6RCHN6","short_pith_number":"pith:K63XQGH2","schema_version":"1.0","canonical_sha256":"57b77818fa310ecb5b6124a57f44476f827775d04a63ccd29180add2abe87620","source":{"kind":"arxiv","id":"2409.09916","version":1},"attestation_state":"computed","paper":{"title":"SFR-RAG: Towards Contextually Faithful LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Austin Xu, Caiming Xong, Hailin Chen, Senthil Purushwalkam, Shafiq Joty, Shrey Pandit, Silvio Savarese, Xuan-Phi Nguyen, Yifei Ming, Zixuan Ke","submitted_at":"2024-09-16T01:08:18Z","abstract_excerpt":"Retrieval Augmented Generation (RAG), a paradigm that integrates external contextual information with large language models (LLMs) to enhance factual accuracy and relevance, has emerged as a pivotal area in generative AI. The LLMs used in RAG applications are required to faithfully and completely comprehend the provided context and users' questions, avoid hallucination, handle unanswerable, counterfactual or otherwise low-quality and irrelevant contexts, perform complex multi-hop reasoning and produce reliable citations. In this paper, we introduce SFR-RAG, a small LLM that is instruction-tune"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.09916","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-16T01:08:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f490a2fd35b20ef8503b90791de4778bfa0a547891f8ea2d691e34830ee37794","abstract_canon_sha256":"47077ae82f2e22937e67caedde72b50fe395d88956722ba4774b4fda35e7cd12"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:16.391351Z","signature_b64":"URz2XuLhlW3qVPAYhnlud69vdR/erKACtFgwjIA2JNtTvGRbhnbUsbk1tYuUSxGCPvGUoNNQCpy6H5/wXh1FBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57b77818fa310ecb5b6124a57f44476f827775d04a63ccd29180add2abe87620","last_reissued_at":"2026-07-05T09:07:16.390492Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:16.390492Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SFR-RAG: Towards Contextually Faithful LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Austin Xu, Caiming Xong, Hailin Chen, Senthil Purushwalkam, Shafiq Joty, Shrey Pandit, Silvio Savarese, Xuan-Phi Nguyen, Yifei Ming, Zixuan Ke","submitted_at":"2024-09-16T01:08:18Z","abstract_excerpt":"Retrieval Augmented Generation (RAG), a paradigm that integrates external contextual information with large language models (LLMs) to enhance factual accuracy and relevance, has emerged as a pivotal area in generative AI. The LLMs used in RAG applications are required to faithfully and completely comprehend the provided context and users' questions, avoid hallucination, handle unanswerable, counterfactual or otherwise low-quality and irrelevant contexts, perform complex multi-hop reasoning and produce reliable citations. In this paper, we introduce SFR-RAG, a small LLM that is instruction-tune"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.09916","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.09916/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.09916","created_at":"2026-07-05T09:07:16.390894+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.09916v1","created_at":"2026-07-05T09:07:16.390894+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.09916","created_at":"2026-07-05T09:07:16.390894+00:00"},{"alias_kind":"pith_short_12","alias_value":"K63XQGH2GEHM","created_at":"2026-07-05T09:07:16.390894+00:00"},{"alias_kind":"pith_short_16","alias_value":"K63XQGH2GEHMWW3B","created_at":"2026-07-05T09:07:16.390894+00:00"},{"alias_kind":"pith_short_8","alias_value":"K63XQGH2","created_at":"2026-07-05T09:07:16.390894+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.12538","citing_title":"Agentic Reasoning for Large Language Models","ref_index":270,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02605","citing_title":"Do Audio-Visual Large Language Models Really See and Hear?","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16686","citing_title":"No-Worse Context-Aware Decoding: Preventing Neutral Regression in Context-Conditioned Generation","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6","json":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6.json","graph_json":"https://pith.science/api/pith-number/K63XQGH2GEHMWW3BESSX6RCHN6/graph.json","events_json":"https://pith.science/api/pith-number/K63XQGH2GEHMWW3BESSX6RCHN6/events.json","paper":"https://pith.science/paper/K63XQGH2"},"agent_actions":{"view_html":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6","download_json":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6.json","view_paper":"https://pith.science/paper/K63XQGH2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.09916&json=true","fetch_graph":"https://pith.science/api/pith-number/K63XQGH2GEHMWW3BESSX6RCHN6/graph.json","fetch_events":"https://pith.science/api/pith-number/K63XQGH2GEHMWW3BESSX6RCHN6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6/action/storage_attestation","attest_author":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6/action/author_attestation","sign_citation":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6/action/citation_signature","submit_replication":"https://pith.science/pith/K63XQGH2GEHMWW3BESSX6RCHN6/action/replication_record"}},"created_at":"2026-07-05T09:07:16.390894+00:00","updated_at":"2026-07-05T09:07:16.390894+00:00"}