{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PPBZ32DB6OQCSDSCO4GW7EZOXU","short_pith_number":"pith:PPBZ32DB","schema_version":"1.0","canonical_sha256":"7bc39de861f3a0290e42770d6f932ebd1fd24c68adebe0abce7d18326e988419","source":{"kind":"arxiv","id":"2410.14675","version":2},"attestation_state":"computed","paper":{"title":"To Trust or Not to Trust? Enhancing Large Language Models' Situated Faithfulness to External Contexts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bhuwan Dhingra, Hongyi Cai, Sanxing Chen, Yukun Huang","submitted_at":"2024-10-18T17:59:47Z","abstract_excerpt":"Large Language Models (LLMs) are often augmented with external contexts, such as those used in retrieval-augmented generation (RAG). However, these contexts can be inaccurate or intentionally misleading, leading to conflicts with the model's internal knowledge. We argue that robust LLMs should demonstrate situated faithfulness, dynamically calibrating their trust in external information based on their confidence in the internal knowledge and the external context to resolve knowledge conflicts. To benchmark this capability, we evaluate LLMs across several QA datasets, including a newly created "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14675","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-18T17:59:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4500df0de50d31a553aaf347a3cc26914f954d0387992fadc308adbf40cb3e4f","abstract_canon_sha256":"1469e71e3443f1d0ed13a06cb6cb78e78e8412843dda006821763b764933aa62"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:32:32.357650Z","signature_b64":"iB2eTMC0bwzC0Z0s42SAW231/UQTsTyi5UdUmFMYDWsI8cQcJ66epvdh1O/RL87iVVIIn1s0UleOTiUxFigaDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7bc39de861f3a0290e42770d6f932ebd1fd24c68adebe0abce7d18326e988419","last_reissued_at":"2026-07-05T10:32:32.357156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:32:32.357156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"To Trust or Not to Trust? Enhancing Large Language Models' Situated Faithfulness to External Contexts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bhuwan Dhingra, Hongyi Cai, Sanxing Chen, Yukun Huang","submitted_at":"2024-10-18T17:59:47Z","abstract_excerpt":"Large Language Models (LLMs) are often augmented with external contexts, such as those used in retrieval-augmented generation (RAG). However, these contexts can be inaccurate or intentionally misleading, leading to conflicts with the model's internal knowledge. We argue that robust LLMs should demonstrate situated faithfulness, dynamically calibrating their trust in external information based on their confidence in the internal knowledge and the external context to resolve knowledge conflicts. To benchmark this capability, we evaluate LLMs across several QA datasets, including a newly created "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14675","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14675/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14675","created_at":"2026-07-05T10:32:32.357218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14675v2","created_at":"2026-07-05T10:32:32.357218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14675","created_at":"2026-07-05T10:32:32.357218+00:00"},{"alias_kind":"pith_short_12","alias_value":"PPBZ32DB6OQC","created_at":"2026-07-05T10:32:32.357218+00:00"},{"alias_kind":"pith_short_16","alias_value":"PPBZ32DB6OQCSDSC","created_at":"2026-07-05T10:32:32.357218+00:00"},{"alias_kind":"pith_short_8","alias_value":"PPBZ32DB","created_at":"2026-07-05T10:32:32.357218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.13785","citing_title":"LLM-Driven Data Generation and a Novel Soft Metric for Evaluating Text-to-SQL in Aviation MRO","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU","json":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU.json","graph_json":"https://pith.science/api/pith-number/PPBZ32DB6OQCSDSCO4GW7EZOXU/graph.json","events_json":"https://pith.science/api/pith-number/PPBZ32DB6OQCSDSCO4GW7EZOXU/events.json","paper":"https://pith.science/paper/PPBZ32DB"},"agent_actions":{"view_html":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU","download_json":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU.json","view_paper":"https://pith.science/paper/PPBZ32DB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14675&json=true","fetch_graph":"https://pith.science/api/pith-number/PPBZ32DB6OQCSDSCO4GW7EZOXU/graph.json","fetch_events":"https://pith.science/api/pith-number/PPBZ32DB6OQCSDSCO4GW7EZOXU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU/action/storage_attestation","attest_author":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU/action/author_attestation","sign_citation":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU/action/citation_signature","submit_replication":"https://pith.science/pith/PPBZ32DB6OQCSDSCO4GW7EZOXU/action/replication_record"}},"created_at":"2026-07-05T10:32:32.357218+00:00","updated_at":"2026-07-05T10:32:32.357218+00:00"}