{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NJADG56MDVW32E3UZYZC6YDDAH","short_pith_number":"pith:NJADG56M","schema_version":"1.0","canonical_sha256":"6a403377cc1d6dbd1374ce322f606301fc0a9145879879870c1eb186c3b91a68","source":{"kind":"arxiv","id":"2402.12865","version":1},"attestation_state":"computed","paper":{"title":"Backward Lens: Projecting Language Model Gradients into the Vocabulary Space","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Lior Wolf, Mor Geva, Shahar Katz, Yonatan Belinkov","submitted_at":"2024-02-20T09:57:08Z","abstract_excerpt":"Understanding how Transformer-based Language Models (LMs) learn and recall information is a key goal of the deep learning community. Recent interpretability methods project weights and hidden states obtained from the forward pass to the models' vocabularies, helping to uncover how information flows within LMs. In this work, we extend this methodology to LMs' backward pass and gradients. We first prove that a gradient matrix can be cast as a low-rank linear combination of its forward and backward passes' inputs. We then develop methods to project these gradients into vocabulary items and explor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12865","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-20T09:57:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"cd78026bf2dd50a038f5ef0ac987c84d242ea155c63e45281da11d9fed064a77","abstract_canon_sha256":"458c04ff85e49786ba47c613586214d63c6d173c0b95841510b4e2fae46df56a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:47:17.503489Z","signature_b64":"CFretA44oDjuudMsbGjk9YivJcOUUf5SsdiCWfvU751fwM4VTWk/MlTgZvQYYZnNk/wUlYq4v8HngskSsTrEDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a403377cc1d6dbd1374ce322f606301fc0a9145879879870c1eb186c3b91a68","last_reissued_at":"2026-07-05T07:47:17.503005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:47:17.503005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Backward Lens: Projecting Language Model Gradients into the Vocabulary Space","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Lior Wolf, Mor Geva, Shahar Katz, Yonatan Belinkov","submitted_at":"2024-02-20T09:57:08Z","abstract_excerpt":"Understanding how Transformer-based Language Models (LMs) learn and recall information is a key goal of the deep learning community. Recent interpretability methods project weights and hidden states obtained from the forward pass to the models' vocabularies, helping to uncover how information flows within LMs. In this work, we extend this methodology to LMs' backward pass and gradients. We first prove that a gradient matrix can be cast as a low-rank linear combination of its forward and backward passes' inputs. We then develop methods to project these gradients into vocabulary items and explor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12865","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12865/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12865","created_at":"2026-07-05T07:47:17.503064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12865v1","created_at":"2026-07-05T07:47:17.503064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12865","created_at":"2026-07-05T07:47:17.503064+00:00"},{"alias_kind":"pith_short_12","alias_value":"NJADG56MDVW3","created_at":"2026-07-05T07:47:17.503064+00:00"},{"alias_kind":"pith_short_16","alias_value":"NJADG56MDVW32E3U","created_at":"2026-07-05T07:47:17.503064+00:00"},{"alias_kind":"pith_short_8","alias_value":"NJADG56M","created_at":"2026-07-05T07:47:17.503064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21345","citing_title":"Factual Retrieval in LLMs Is a Redundant, Distributed and Non-Contiguous Process","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22005","citing_title":"Check Your LLM's Secret Dictionary! Five Lines of Code Reveal What Your LLM Learned (Including What It Shouldn't Have)","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20340","citing_title":"Latent Trajectory Dynamics in Large Language Models: A Manifold Evolution Framework with Empirical Validation","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH","json":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH.json","graph_json":"https://pith.science/api/pith-number/NJADG56MDVW32E3UZYZC6YDDAH/graph.json","events_json":"https://pith.science/api/pith-number/NJADG56MDVW32E3UZYZC6YDDAH/events.json","paper":"https://pith.science/paper/NJADG56M"},"agent_actions":{"view_html":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH","download_json":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH.json","view_paper":"https://pith.science/paper/NJADG56M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12865&json=true","fetch_graph":"https://pith.science/api/pith-number/NJADG56MDVW32E3UZYZC6YDDAH/graph.json","fetch_events":"https://pith.science/api/pith-number/NJADG56MDVW32E3UZYZC6YDDAH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH/action/storage_attestation","attest_author":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH/action/author_attestation","sign_citation":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH/action/citation_signature","submit_replication":"https://pith.science/pith/NJADG56MDVW32E3UZYZC6YDDAH/action/replication_record"}},"created_at":"2026-07-05T07:47:17.503064+00:00","updated_at":"2026-07-05T07:47:17.503064+00:00"}