{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WTIZN47OYXVSETBUNMMQPIRUGJ","short_pith_number":"pith:WTIZN47O","schema_version":"1.0","canonical_sha256":"b4d196f3eec5eb224c346b1907a23432609587cf915bb5033486e647174a56da","source":{"kind":"arxiv","id":"2312.03656","version":2},"attestation_state":"computed","paper":{"title":"Interpretability Illusions in the Generalization of Simplified Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Andrew Lampinen, Asma Ghandeharioun, Dan Friedman, Danqi Chen, Lucas Dixon","submitted_at":"2023-12-06T18:25:53Z","abstract_excerpt":"A common method to study deep learning systems is to use simplified model representations--for example, using singular value decomposition to visualize the model's hidden states in a lower dimensional space. This approach assumes that the results of these simplifications are faithful to the original model. Here, we illustrate an important caveat to this assumption: even if the simplified representations can accurately approximate the full model on the training set, they may fail to accurately capture the model's behavior out of distribution. We illustrate this by training Transformer models on"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.03656","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-06T18:25:53Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"6cea2eb129ba32e80d7287afcd642d5727c5e781330e0512be00346cbad3b2d1","abstract_canon_sha256":"902def203e994200a6fb04d034184f50ed71564dd09a07b5216d6ba89a079ee6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:44.381545Z","signature_b64":"CQVs401AbH0W37cYvl0msJWA8+ORxW3ZtdbreeP4GWI/KRxH1EcvRa0Yj89s2YuQTo5ECiu8L4QoGKZoBA40Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b4d196f3eec5eb224c346b1907a23432609587cf915bb5033486e647174a56da","last_reissued_at":"2026-07-05T08:27:44.381012Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:44.381012Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpretability Illusions in the Generalization of Simplified Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Andrew Lampinen, Asma Ghandeharioun, Dan Friedman, Danqi Chen, Lucas Dixon","submitted_at":"2023-12-06T18:25:53Z","abstract_excerpt":"A common method to study deep learning systems is to use simplified model representations--for example, using singular value decomposition to visualize the model's hidden states in a lower dimensional space. This approach assumes that the results of these simplifications are faithful to the original model. Here, we illustrate an important caveat to this assumption: even if the simplified representations can accurately approximate the full model on the training set, they may fail to accurately capture the model's behavior out of distribution. We illustrate this by training Transformer models on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03656","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.03656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.03656","created_at":"2026-07-05T08:27:44.381071+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.03656v2","created_at":"2026-07-05T08:27:44.381071+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03656","created_at":"2026-07-05T08:27:44.381071+00:00"},{"alias_kind":"pith_short_12","alias_value":"WTIZN47OYXVS","created_at":"2026-07-05T08:27:44.381071+00:00"},{"alias_kind":"pith_short_16","alias_value":"WTIZN47OYXVSETBU","created_at":"2026-07-05T08:27:44.381071+00:00"},{"alias_kind":"pith_short_8","alias_value":"WTIZN47O","created_at":"2026-07-05T08:27:44.381071+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08292","citing_title":"Necessary, Decodable and Reversible, Yet Not Transferable: A Stress Test for Attention-Head Role Claims","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15054","citing_title":"Size Doesn't Matter: Cosine-Scored Sparse Autoencoders","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ","json":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ.json","graph_json":"https://pith.science/api/pith-number/WTIZN47OYXVSETBUNMMQPIRUGJ/graph.json","events_json":"https://pith.science/api/pith-number/WTIZN47OYXVSETBUNMMQPIRUGJ/events.json","paper":"https://pith.science/paper/WTIZN47O"},"agent_actions":{"view_html":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ","download_json":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ.json","view_paper":"https://pith.science/paper/WTIZN47O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.03656&json=true","fetch_graph":"https://pith.science/api/pith-number/WTIZN47OYXVSETBUNMMQPIRUGJ/graph.json","fetch_events":"https://pith.science/api/pith-number/WTIZN47OYXVSETBUNMMQPIRUGJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ/action/storage_attestation","attest_author":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ/action/author_attestation","sign_citation":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ/action/citation_signature","submit_replication":"https://pith.science/pith/WTIZN47OYXVSETBUNMMQPIRUGJ/action/replication_record"}},"created_at":"2026-07-05T08:27:44.381071+00:00","updated_at":"2026-07-05T08:27:44.381071+00:00"}