{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:C4J7SV7HTIDCOFN4JRCTA3QJPB","short_pith_number":"pith:C4J7SV7H","schema_version":"1.0","canonical_sha256":"1713f957e79a062715bc4c45306e09786055755c3890194d20384d3b10a61312","source":{"kind":"arxiv","id":"2405.05386","version":2},"attestation_state":"computed","paper":{"title":"Interpretability Needs a New Paradigm","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andreas Madsen, Himabindu Lakkaraju, Sarath Chandar, Siva Reddy","submitted_at":"2024-05-08T19:31:06Z","abstract_excerpt":"Interpretability is the study of explaining models in understandable terms to humans. At present, interpretability is divided into two paradigms: the intrinsic paradigm, which believes that only models designed to be explained can be explained, and the post-hoc paradigm, which believes that black-box models can be explained. At the core of this debate is how each paradigm ensures its explanations are faithful, i.e., true to the model's behavior. This is important, as false but convincing explanations lead to unsupported confidence in artificial intelligence (AI), which can be dangerous. This p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.05386","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-08T19:31:06Z","cross_cats_sorted":["cs.CL","cs.CV","stat.ML"],"title_canon_sha256":"b6e84c6b0726b5570f2f66c238074da6d25257fca035a74d564d286aec5acf17","abstract_canon_sha256":"bf5c263d442c5075493b96851e3b79506f15c742ecf1606710f5d8cdf132c805"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:34:35.677065Z","signature_b64":"etBmqZEhpHP07zf9uZ2KAIq9qnozT6ig30B9kblVhu6R/0mBp8Sh9moQLMsQZQbhsB86Z5qZaGc5cQpDxfpxDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1713f957e79a062715bc4c45306e09786055755c3890194d20384d3b10a61312","last_reissued_at":"2026-07-05T09:34:35.676598Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:34:35.676598Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpretability Needs a New Paradigm","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andreas Madsen, Himabindu Lakkaraju, Sarath Chandar, Siva Reddy","submitted_at":"2024-05-08T19:31:06Z","abstract_excerpt":"Interpretability is the study of explaining models in understandable terms to humans. At present, interpretability is divided into two paradigms: the intrinsic paradigm, which believes that only models designed to be explained can be explained, and the post-hoc paradigm, which believes that black-box models can be explained. At the core of this debate is how each paradigm ensures its explanations are faithful, i.e., true to the model's behavior. This is important, as false but convincing explanations lead to unsupported confidence in artificial intelligence (AI), which can be dangerous. This p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.05386","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.05386/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.05386","created_at":"2026-07-05T09:34:35.676651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.05386v2","created_at":"2026-07-05T09:34:35.676651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.05386","created_at":"2026-07-05T09:34:35.676651+00:00"},{"alias_kind":"pith_short_12","alias_value":"C4J7SV7HTIDC","created_at":"2026-07-05T09:34:35.676651+00:00"},{"alias_kind":"pith_short_16","alias_value":"C4J7SV7HTIDCOFN4","created_at":"2026-07-05T09:34:35.676651+00:00"},{"alias_kind":"pith_short_8","alias_value":"C4J7SV7H","created_at":"2026-07-05T09:34:35.676651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29346","citing_title":"Reliability, Faithfulness, and the Limits of Post-hoc Explanations of Opaque Scientific Models","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11161","citing_title":"Interpretability Can Be Actionable","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08934","citing_title":"From Mechanistic to Compositional Interpretability","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB","json":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB.json","graph_json":"https://pith.science/api/pith-number/C4J7SV7HTIDCOFN4JRCTA3QJPB/graph.json","events_json":"https://pith.science/api/pith-number/C4J7SV7HTIDCOFN4JRCTA3QJPB/events.json","paper":"https://pith.science/paper/C4J7SV7H"},"agent_actions":{"view_html":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB","download_json":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB.json","view_paper":"https://pith.science/paper/C4J7SV7H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.05386&json=true","fetch_graph":"https://pith.science/api/pith-number/C4J7SV7HTIDCOFN4JRCTA3QJPB/graph.json","fetch_events":"https://pith.science/api/pith-number/C4J7SV7HTIDCOFN4JRCTA3QJPB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB/action/storage_attestation","attest_author":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB/action/author_attestation","sign_citation":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB/action/citation_signature","submit_replication":"https://pith.science/pith/C4J7SV7HTIDCOFN4JRCTA3QJPB/action/replication_record"}},"created_at":"2026-07-05T09:34:35.676651+00:00","updated_at":"2026-07-05T09:34:35.676651+00:00"}