{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:ESM27PTSNBKUY3ZUUETNPFCIOF","short_pith_number":"pith:ESM27PTS","schema_version":"1.0","canonical_sha256":"2499afbe7268554c6f34a126d79448715c371cab024e683978cec2ac7ceb4ff8","source":{"kind":"arxiv","id":"1909.01380","version":1},"attestation_state":"computed","paper":{"title":"The Bottom-up Evolution of Representations in the Transformer: A Study with Machine Translation and Language Modeling Objectives","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Elena Voita, Ivan Titov, Rico Sennrich","submitted_at":"2019-09-03T18:06:03Z","abstract_excerpt":"We seek to understand how the representations of individual tokens and the structure of the learned feature space evolve between layers in deep neural networks under different learning objectives. We focus on the Transformers for our analysis as they have been shown effective on various tasks, including machine translation (MT), standard left-to-right language models (LM) and masked language modeling (MLM). Previous work used black-box probing tasks to show that the representations learned by the Transformer differ significantly depending on the objective. In this work, we use canonical correl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.01380","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-09-03T18:06:03Z","cross_cats_sorted":[],"title_canon_sha256":"0cb0edb38fa9b597070b5411994c1b8488efd18af3ffb2bef3ed9168e21db8eb","abstract_canon_sha256":"47289c18fa96b090fcdde6baed0fe4db47a9a14fcdc168004caf657bcc63a237"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:02:15.607075Z","signature_b64":"t+O9Lgl+U9lpZkdD08aDeVBiK+gBN7HgeQhqJC359OkqnPU94juVmj++6e2WToWrlSiYoD6ltJMyHkN1kfKgAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2499afbe7268554c6f34a126d79448715c371cab024e683978cec2ac7ceb4ff8","last_reissued_at":"2026-07-05T00:02:15.606516Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:02:15.606516Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Bottom-up Evolution of Representations in the Transformer: A Study with Machine Translation and Language Modeling Objectives","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Elena Voita, Ivan Titov, Rico Sennrich","submitted_at":"2019-09-03T18:06:03Z","abstract_excerpt":"We seek to understand how the representations of individual tokens and the structure of the learned feature space evolve between layers in deep neural networks under different learning objectives. We focus on the Transformers for our analysis as they have been shown effective on various tasks, including machine translation (MT), standard left-to-right language models (LM) and masked language modeling (MLM). Previous work used black-box probing tasks to show that the representations learned by the Transformer differ significantly depending on the objective. In this work, we use canonical correl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.01380","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.01380/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.01380","created_at":"2026-07-05T00:02:15.606583+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.01380v1","created_at":"2026-07-05T00:02:15.606583+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.01380","created_at":"2026-07-05T00:02:15.606583+00:00"},{"alias_kind":"pith_short_12","alias_value":"ESM27PTSNBKU","created_at":"2026-07-05T00:02:15.606583+00:00"},{"alias_kind":"pith_short_16","alias_value":"ESM27PTSNBKUY3ZU","created_at":"2026-07-05T00:02:15.606583+00:00"},{"alias_kind":"pith_short_8","alias_value":"ESM27PTS","created_at":"2026-07-05T00:02:15.606583+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.05387","citing_title":"The Generalization Ridge: Information Flow in Natural Language Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"1910.10683","citing_title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05741","citing_title":"HyperLens: Quantifying Cognitive Effort in LLMs with Fine-grained Confidence Trajectory","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07569","citing_title":"Learning is Forgetting: LLM Training As Lossy Compression","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF","json":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF.json","graph_json":"https://pith.science/api/pith-number/ESM27PTSNBKUY3ZUUETNPFCIOF/graph.json","events_json":"https://pith.science/api/pith-number/ESM27PTSNBKUY3ZUUETNPFCIOF/events.json","paper":"https://pith.science/paper/ESM27PTS"},"agent_actions":{"view_html":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF","download_json":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF.json","view_paper":"https://pith.science/paper/ESM27PTS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.01380&json=true","fetch_graph":"https://pith.science/api/pith-number/ESM27PTSNBKUY3ZUUETNPFCIOF/graph.json","fetch_events":"https://pith.science/api/pith-number/ESM27PTSNBKUY3ZUUETNPFCIOF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF/action/storage_attestation","attest_author":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF/action/author_attestation","sign_citation":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF/action/citation_signature","submit_replication":"https://pith.science/pith/ESM27PTSNBKUY3ZUUETNPFCIOF/action/replication_record"}},"created_at":"2026-07-05T00:02:15.606583+00:00","updated_at":"2026-07-05T00:02:15.606583+00:00"}