{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P6GXSDZYS7BPPOLPZ62GITVTI3","short_pith_number":"pith:P6GXSDZY","schema_version":"1.0","canonical_sha256":"7f8d790f3897c2f7b96fcfb4644eb346f046b525458eec7e8b2c79541451ac50","source":{"kind":"arxiv","id":"2402.04614","version":3},"attestation_state":"computed","paper":{"title":"Faithfulness vs. Plausibility: On the (Un)Reliability of Explanations from Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chirag Agarwal, Himabindu Lakkaraju, Sree Harsha Tanneru","submitted_at":"2024-02-07T06:32:50Z","abstract_excerpt":"Large Language Models (LLMs) are deployed as powerful tools for several natural language processing (NLP) applications. Recent works show that modern LLMs can generate self-explanations (SEs), which elicit their intermediate reasoning steps for explaining their behavior. Self-explanations have seen widespread adoption owing to their conversational and plausible nature. However, there is little to no understanding of their faithfulness. In this work, we discuss the dichotomy between faithfulness and plausibility in SEs generated by LLMs. We argue that while LLMs are adept at generating plausibl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.04614","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-07T06:32:50Z","cross_cats_sorted":[],"title_canon_sha256":"8946785d12f4040562bbceb2a16438c6181b286e18b1b5e299a65b3fe34f7611","abstract_canon_sha256":"3ecca3c53a10a09e45302787d58c23cac9f25d1a917d9b075c86e719eadbd668"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:03.079795Z","signature_b64":"HeplZqw911mTdSo4g/Xt0wKPLebDh/LUxcjPT+9QNxolF2dZp5kkaiCDXwAUC/vnQL0SMS6ckDssDucDA0oCBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f8d790f3897c2f7b96fcfb4644eb346f046b525458eec7e8b2c79541451ac50","last_reissued_at":"2026-07-05T07:56:03.079433Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:03.079433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Faithfulness vs. Plausibility: On the (Un)Reliability of Explanations from Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chirag Agarwal, Himabindu Lakkaraju, Sree Harsha Tanneru","submitted_at":"2024-02-07T06:32:50Z","abstract_excerpt":"Large Language Models (LLMs) are deployed as powerful tools for several natural language processing (NLP) applications. Recent works show that modern LLMs can generate self-explanations (SEs), which elicit their intermediate reasoning steps for explaining their behavior. Self-explanations have seen widespread adoption owing to their conversational and plausible nature. However, there is little to no understanding of their faithfulness. In this work, we discuss the dichotomy between faithfulness and plausibility in SEs generated by LLMs. We argue that while LLMs are adept at generating plausibl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04614","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04614/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.04614","created_at":"2026-07-05T07:56:03.079490+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.04614v3","created_at":"2026-07-05T07:56:03.079490+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04614","created_at":"2026-07-05T07:56:03.079490+00:00"},{"alias_kind":"pith_short_12","alias_value":"P6GXSDZYS7BP","created_at":"2026-07-05T07:56:03.079490+00:00"},{"alias_kind":"pith_short_16","alias_value":"P6GXSDZYS7BPPOLP","created_at":"2026-07-05T07:56:03.079490+00:00"},{"alias_kind":"pith_short_8","alias_value":"P6GXSDZY","created_at":"2026-07-05T07:56:03.079490+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27274","citing_title":"BetXplain: An Explanation-Annotated Dataset for Detecting Manipulative Betting Advertisements on Social Media","ref_index":286,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21121","citing_title":"Answer Engineering: Local Trajectory Editing for Protocol-Constrained Decision Making in Large Language Models","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31285","citing_title":"Spatial Reasoning via Modality Switching Between Language and Symbolic Representation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11445","citing_title":"Forecasting Future Behavior as a Learning Task","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09470","citing_title":"A Finetuned SpeechLLM for Joint Multi-Granular L2 Assessment and Natural-Language Rationales","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28615","citing_title":"What LLMs explain is not what they believe: Evaluating explanation sufficiency under models' own input beliefs","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31616","citing_title":"Scientific Explanations in Health Sciences: Causality, Trust, and Epistemic Adequacy","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31285","citing_title":"Spatial Reasoning via Modality Switching Between Language and Symbolic Representation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23772","citing_title":"PageGuide: Browser extension to assist users in navigating a webpage and locating information","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24960","citing_title":"Investigating the Interplay between Contextual and Parametric Chain-of-Thought Faithfulness under Optimization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25052","citing_title":"Faithfulness Metrics Don't Measure Faithfulness: A Meta-Evaluation with Ground Truth","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25603","citing_title":"Detecting Unfaithful Chain-of-Thought via Circuit-Guided Internal-External Discrepancy","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2410.03296","citing_title":"A Systematic Comparison between Extractive Self-Explanations and Human Rationales in Text Classification","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2412.16720","citing_title":"OpenAI o1 System Card","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20665","citing_title":"The Shape of Reasoning: Topological Analysis of Reasoning Traces in Large Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2602.04003","citing_title":"When AI Persuades: Adversarial Explanation Attacks on Human Trust in AI-Assisted Decision Making","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11161","citing_title":"Interpretability Can Be Actionable","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23772","citing_title":"PageGuide: Browser extension to assist users in navigating a webpage and locating information","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06422","citing_title":"When to Call an Apple Red: Humans Follow Introspective Rules, VLMs Don't","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16158","citing_title":"AtManRL: Towards Faithful Reasoning via Differentiable Attention Saliency","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06640","citing_title":"Concept-Based Abductive and Contrastive Explanations for Behaviors of Vision Models","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3","json":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3.json","graph_json":"https://pith.science/api/pith-number/P6GXSDZYS7BPPOLPZ62GITVTI3/graph.json","events_json":"https://pith.science/api/pith-number/P6GXSDZYS7BPPOLPZ62GITVTI3/events.json","paper":"https://pith.science/paper/P6GXSDZY"},"agent_actions":{"view_html":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3","download_json":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3.json","view_paper":"https://pith.science/paper/P6GXSDZY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.04614&json=true","fetch_graph":"https://pith.science/api/pith-number/P6GXSDZYS7BPPOLPZ62GITVTI3/graph.json","fetch_events":"https://pith.science/api/pith-number/P6GXSDZYS7BPPOLPZ62GITVTI3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3/action/storage_attestation","attest_author":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3/action/author_attestation","sign_citation":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3/action/citation_signature","submit_replication":"https://pith.science/pith/P6GXSDZYS7BPPOLPZ62GITVTI3/action/replication_record"}},"created_at":"2026-07-05T07:56:03.079490+00:00","updated_at":"2026-07-05T07:56:03.079490+00:00"}