{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:34ZZCEDEQHX62Z4BAKTEDX2RQ6","short_pith_number":"pith:34ZZCEDE","schema_version":"1.0","canonical_sha256":"df3391106481efed678102a641df518785a8cdce7bb63cd9caf78df7400de581","source":{"kind":"arxiv","id":"1907.12009","version":1},"attestation_state":"computed","paper":{"title":"Representation Degeneration Problem in Training Natural Language Generation Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Di He, Jun Gao, Liwei Wang, Tao Qin, Tie-Yan Liu, Xu Tan","submitted_at":"2019-07-28T03:57:41Z","abstract_excerpt":"We study an interesting problem in training neural network-based models for natural language generation tasks, which we call the \\emph{representation degeneration problem}. We observe that when training a model for natural language generation tasks through likelihood maximization with the weight tying trick, especially with big training datasets, most of the learnt word embeddings tend to degenerate and be distributed into a narrow cone, which largely limits the representation power of word embeddings. We analyze the conditions and causes of this problem and propose a novel regularization meth"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1907.12009","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2019-07-28T03:57:41Z","cross_cats_sorted":[],"title_canon_sha256":"137f5b4f2f60207fb9ebc7bae0d19c2c6aedd770877a3bb43dd51711d83e3f1a","abstract_canon_sha256":"eaeeb577a24af1faac0edaf38482275465cfe4d2795c0992c0f6f516aef38b1b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:50:04.589294Z","signature_b64":"aNJNENc4wDv313AgtafsxCbHKlk45r/S0yVMNyhh+Pd2ja1mdSjfWnWiqEYdEAl5wePpS/I0Dc7aOmtbOi7kDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df3391106481efed678102a641df518785a8cdce7bb63cd9caf78df7400de581","last_reissued_at":"2026-07-04T23:50:04.588819Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:50:04.588819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Representation Degeneration Problem in Training Natural Language Generation Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Di He, Jun Gao, Liwei Wang, Tao Qin, Tie-Yan Liu, Xu Tan","submitted_at":"2019-07-28T03:57:41Z","abstract_excerpt":"We study an interesting problem in training neural network-based models for natural language generation tasks, which we call the \\emph{representation degeneration problem}. We observe that when training a model for natural language generation tasks through likelihood maximization with the weight tying trick, especially with big training datasets, most of the learnt word embeddings tend to degenerate and be distributed into a narrow cone, which largely limits the representation power of word embeddings. We analyze the conditions and causes of this problem and propose a novel regularization meth"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1907.12009","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1907.12009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1907.12009","created_at":"2026-07-04T23:50:04.588881+00:00"},{"alias_kind":"arxiv_version","alias_value":"1907.12009v1","created_at":"2026-07-04T23:50:04.588881+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1907.12009","created_at":"2026-07-04T23:50:04.588881+00:00"},{"alias_kind":"pith_short_12","alias_value":"34ZZCEDEQHX6","created_at":"2026-07-04T23:50:04.588881+00:00"},{"alias_kind":"pith_short_16","alias_value":"34ZZCEDEQHX62Z4B","created_at":"2026-07-04T23:50:04.588881+00:00"},{"alias_kind":"pith_short_8","alias_value":"34ZZCEDE","created_at":"2026-07-04T23:50:04.588881+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22019","citing_title":"Channel Location Constrains the Auditability of Subliminal Learning","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20138","citing_title":"Learning to Prompt: Improving Student Engagement with Adaptive LLM-based High-School Tutoring","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24956","citing_title":"NITP: Next Implicit Token Prediction for LLM Pre-training","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01844","citing_title":"Decoupled Residual Quantization for Robust Semantic IDs in Recommendation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28063","citing_title":"How to deal with machine learning bias in economic history","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24956","citing_title":"NITP: Next Implicit Token Prediction for LLM Pre-training","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2412.13663","citing_title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12714","citing_title":"Layer-wise Representation Dynamics: An Empirical Investigation Across Embedders and Base LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2208.07339","citing_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12292","citing_title":"STRABLE: Benchmarking Tabular Machine Learning with Strings","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10790","citing_title":"Elucidating Representation Degradation Problem in Diffusion Model Training","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05741","citing_title":"HyperLens: Quantifying Cognitive Effort in LLMs with Fine-grained Confidence Trajectory","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11344","citing_title":"Geometry-Aware Localized Watermarking for Copyright Protection in Embedding-as-a-Service","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08764","citing_title":"Revisiting Anisotropy in Language Transformers: The Geometry of Learning Dynamics","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06826","citing_title":"How Does Attention Help? Insights from Random Matrices on Signal Recovery from Sequence Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15009","citing_title":"Towards Faster Language Model Inference Using Mixture-of-Experts Flow Matching","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18804","citing_title":"Geometric Decoupling: Diagnosing the Structural Instability of Latent","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6","json":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6.json","graph_json":"https://pith.science/api/pith-number/34ZZCEDEQHX62Z4BAKTEDX2RQ6/graph.json","events_json":"https://pith.science/api/pith-number/34ZZCEDEQHX62Z4BAKTEDX2RQ6/events.json","paper":"https://pith.science/paper/34ZZCEDE"},"agent_actions":{"view_html":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6","download_json":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6.json","view_paper":"https://pith.science/paper/34ZZCEDE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1907.12009&json=true","fetch_graph":"https://pith.science/api/pith-number/34ZZCEDEQHX62Z4BAKTEDX2RQ6/graph.json","fetch_events":"https://pith.science/api/pith-number/34ZZCEDEQHX62Z4BAKTEDX2RQ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6/action/storage_attestation","attest_author":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6/action/author_attestation","sign_citation":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6/action/citation_signature","submit_replication":"https://pith.science/pith/34ZZCEDEQHX62Z4BAKTEDX2RQ6/action/replication_record"}},"created_at":"2026-07-04T23:50:04.588881+00:00","updated_at":"2026-07-04T23:50:04.588881+00:00"}