{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:63IAZTXW5PIKN6PXKD24QWOTVX","short_pith_number":"pith:63IAZTXW","schema_version":"1.0","canonical_sha256":"f6d00ccef6ebd0a6f9f750f5c859d3adfb61241693bd9c5e76a6c39f61e31d64","source":{"kind":"arxiv","id":"2410.14155","version":2},"attestation_state":"computed","paper":{"title":"Towards Faithful Natural Language Explanations: A Study Using Activation Patching in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Erik Cambria, Ranjan Satapathy, Wei Jie Yeo","submitted_at":"2024-10-18T03:45:42Z","abstract_excerpt":"Large Language Models (LLMs) are capable of generating persuasive Natural Language Explanations (NLEs) to justify their answers. However, the faithfulness of these explanations should not be readily trusted at face value. Recent studies have proposed various methods to measure the faithfulness of NLEs, typically by inserting perturbations at the explanation or feature level. We argue that these approaches are neither comprehensive nor correctly designed according to the established definition of faithfulness. Moreover, we highlight the risks of grounding faithfulness findings on out-of-distrib"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14155","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-18T03:45:42Z","cross_cats_sorted":[],"title_canon_sha256":"af8f17f186f7d0aa52e458adaa57b199a96f1511632906e8801a527f62acf1e8","abstract_canon_sha256":"2d87c8bd8f222353fde1b5e602db548a85bd91257a78fe1058311b4c41dbbed7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:34.274007Z","signature_b64":"NGZka3tuxNdvnDba9AgJg5+0gDkoZQATdqlh5uRF9bEBZGnN1dBiIHPkg8jaNM3WXQYsSK9Tr8QsXFf38ZR5CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f6d00ccef6ebd0a6f9f750f5c859d3adfb61241693bd9c5e76a6c39f61e31d64","last_reissued_at":"2026-07-05T09:29:34.273524Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:34.273524Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Faithful Natural Language Explanations: A Study Using Activation Patching in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Erik Cambria, Ranjan Satapathy, Wei Jie Yeo","submitted_at":"2024-10-18T03:45:42Z","abstract_excerpt":"Large Language Models (LLMs) are capable of generating persuasive Natural Language Explanations (NLEs) to justify their answers. However, the faithfulness of these explanations should not be readily trusted at face value. Recent studies have proposed various methods to measure the faithfulness of NLEs, typically by inserting perturbations at the explanation or feature level. We argue that these approaches are neither comprehensive nor correctly designed according to the established definition of faithfulness. Moreover, we highlight the risks of grounding faithfulness findings on out-of-distrib"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14155","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14155/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14155","created_at":"2026-07-05T09:29:34.273589+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14155v2","created_at":"2026-07-05T09:29:34.273589+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14155","created_at":"2026-07-05T09:29:34.273589+00:00"},{"alias_kind":"pith_short_12","alias_value":"63IAZTXW5PIK","created_at":"2026-07-05T09:29:34.273589+00:00"},{"alias_kind":"pith_short_16","alias_value":"63IAZTXW5PIKN6PX","created_at":"2026-07-05T09:29:34.273589+00:00"},{"alias_kind":"pith_short_8","alias_value":"63IAZTXW","created_at":"2026-07-05T09:29:34.273589+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28615","citing_title":"What LLMs explain is not what they believe: Evaluating explanation sufficiency under models' own input beliefs","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX","json":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX.json","graph_json":"https://pith.science/api/pith-number/63IAZTXW5PIKN6PXKD24QWOTVX/graph.json","events_json":"https://pith.science/api/pith-number/63IAZTXW5PIKN6PXKD24QWOTVX/events.json","paper":"https://pith.science/paper/63IAZTXW"},"agent_actions":{"view_html":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX","download_json":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX.json","view_paper":"https://pith.science/paper/63IAZTXW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14155&json=true","fetch_graph":"https://pith.science/api/pith-number/63IAZTXW5PIKN6PXKD24QWOTVX/graph.json","fetch_events":"https://pith.science/api/pith-number/63IAZTXW5PIKN6PXKD24QWOTVX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX/action/storage_attestation","attest_author":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX/action/author_attestation","sign_citation":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX/action/citation_signature","submit_replication":"https://pith.science/pith/63IAZTXW5PIKN6PXKD24QWOTVX/action/replication_record"}},"created_at":"2026-07-05T09:29:34.273589+00:00","updated_at":"2026-07-05T09:29:34.273589+00:00"}