{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DVVDNV45K65P2N3LH5CAGGEQXT","short_pith_number":"pith:DVVDNV45","schema_version":"1.0","canonical_sha256":"1d6a36d79d57bafd376b3f44031890bcdaf94a6c0e03384ca827dd5da2821399","source":{"kind":"arxiv","id":"2303.09435","version":2},"attestation_state":"computed","paper":{"title":"Jump to Conclusions: Short-Cutting Transformers With Linear Transformations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Yom Din, Leshem Choshen, Mor Geva, Taelin Karidi","submitted_at":"2023-03-16T16:10:16Z","abstract_excerpt":"Transformer-based language models create hidden representations of their inputs at every layer, but only use final-layer representations for prediction. This obscures the internal decision-making process of the model and the utility of its intermediate representations. One way to elucidate this is to cast the hidden representations as final representations, bypassing the transformer computation in-between. In this work, we suggest a simple method for such casting, using linear transformations. This approximation far exceeds the prevailing practice of inspecting hidden representations from all "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.09435","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-16T16:10:16Z","cross_cats_sorted":[],"title_canon_sha256":"2e2734759cd1e4ee45c90265256fced1d8f37e5f2b421d9003943717723f989e","abstract_canon_sha256":"e31ee7260c2167eef1ea30997075d8855632317bfdf3ec3c28d16a7d799ec7a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:02.469392Z","signature_b64":"UzVUYv/8NUYW0fSxKy3G+3eIoWVWUqa3EOYBx0Qdlhw9DFsEItz4aIujbqgOoT1fIdyKZBE6Tn/zOn9iFv2vDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d6a36d79d57bafd376b3f44031890bcdaf94a6c0e03384ca827dd5da2821399","last_reissued_at":"2026-07-05T08:34:02.468887Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:02.468887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Jump to Conclusions: Short-Cutting Transformers With Linear Transformations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Yom Din, Leshem Choshen, Mor Geva, Taelin Karidi","submitted_at":"2023-03-16T16:10:16Z","abstract_excerpt":"Transformer-based language models create hidden representations of their inputs at every layer, but only use final-layer representations for prediction. This obscures the internal decision-making process of the model and the utility of its intermediate representations. One way to elucidate this is to cast the hidden representations as final representations, bypassing the transformer computation in-between. In this work, we suggest a simple method for such casting, using linear transformations. This approximation far exceeds the prevailing practice of inspecting hidden representations from all "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.09435","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.09435/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.09435","created_at":"2026-07-05T08:34:02.468947+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.09435v2","created_at":"2026-07-05T08:34:02.468947+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.09435","created_at":"2026-07-05T08:34:02.468947+00:00"},{"alias_kind":"pith_short_12","alias_value":"DVVDNV45K65P","created_at":"2026-07-05T08:34:02.468947+00:00"},{"alias_kind":"pith_short_16","alias_value":"DVVDNV45K65P2N3L","created_at":"2026-07-05T08:34:02.468947+00:00"},{"alias_kind":"pith_short_8","alias_value":"DVVDNV45","created_at":"2026-07-05T08:34:02.468947+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.03258","citing_title":"The Right Answer, the Wrong Direction: Why Transformers Fail at Counting and How to Fix It","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08112","citing_title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03258","citing_title":"The Right Answer, the Wrong Direction: Why Transformers Fail at Counting and How to Fix It","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT","json":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT.json","graph_json":"https://pith.science/api/pith-number/DVVDNV45K65P2N3LH5CAGGEQXT/graph.json","events_json":"https://pith.science/api/pith-number/DVVDNV45K65P2N3LH5CAGGEQXT/events.json","paper":"https://pith.science/paper/DVVDNV45"},"agent_actions":{"view_html":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT","download_json":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT.json","view_paper":"https://pith.science/paper/DVVDNV45","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.09435&json=true","fetch_graph":"https://pith.science/api/pith-number/DVVDNV45K65P2N3LH5CAGGEQXT/graph.json","fetch_events":"https://pith.science/api/pith-number/DVVDNV45K65P2N3LH5CAGGEQXT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT/action/storage_attestation","attest_author":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT/action/author_attestation","sign_citation":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT/action/citation_signature","submit_replication":"https://pith.science/pith/DVVDNV45K65P2N3LH5CAGGEQXT/action/replication_record"}},"created_at":"2026-07-05T08:34:02.468947+00:00","updated_at":"2026-07-05T08:34:02.468947+00:00"}