{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5DQJGSYX6ZCJ2K4WQNOJBQNHB3","short_pith_number":"pith:5DQJGSYX","schema_version":"1.0","canonical_sha256":"e8e0934b17f6449d2b96835c90c1a70ec5c2c79bc1815b7d41dbd3670fffdff0","source":{"kind":"arxiv","id":"2109.09519","version":2},"attestation_state":"computed","paper":{"title":"PLATO-XL: Exploring the Large-scale Pre-training of Dialogue Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fan Wang, HaiFeng Wang, Hua Lu, Huang He, Hua Wu, Siqi Bao, Wenquan Wu, Xinchao Xu, Xin Tian, Xinxian Huang, Yingzhan Lin, Zhen Guo, Zheng-Yu Niu, Zhihua Wu","submitted_at":"2021-09-20T13:10:23Z","abstract_excerpt":"To explore the limit of dialogue generation pre-training, we present the models of PLATO-XL with up to 11 billion parameters, trained on both Chinese and English social media conversations. To train such large models, we adopt the architecture of unified transformer with high computation and parameter efficiency. In addition, we carry out multi-party aware pre-training to better distinguish the characteristic information in social media conversations. With such designs, PLATO-XL successfully achieves superior performances as compared to other approaches in both Chinese and English chitchat. We"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.09519","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-09-20T13:10:23Z","cross_cats_sorted":[],"title_canon_sha256":"f27aebfce16b9d94638e58645a02393e5484207bb1e1eae8c1abe9d6ff5207c7","abstract_canon_sha256":"74b9e00e644c687e1276631d50ebc524696d39bab64db593bdef483b14f7bc82"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:08:14.206114Z","signature_b64":"8yyv457FRfspayrqhQKO7HI7Rh0S0QMN0zRQvw093sO/he0WyZmyg0spUqlxNfteVMSfJ5ic3LBFnS7O8PRDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e8e0934b17f6449d2b96835c90c1a70ec5c2c79bc1815b7d41dbd3670fffdff0","last_reissued_at":"2026-07-05T05:08:14.205659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:08:14.205659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PLATO-XL: Exploring the Large-scale Pre-training of Dialogue Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fan Wang, HaiFeng Wang, Hua Lu, Huang He, Hua Wu, Siqi Bao, Wenquan Wu, Xinchao Xu, Xin Tian, Xinxian Huang, Yingzhan Lin, Zhen Guo, Zheng-Yu Niu, Zhihua Wu","submitted_at":"2021-09-20T13:10:23Z","abstract_excerpt":"To explore the limit of dialogue generation pre-training, we present the models of PLATO-XL with up to 11 billion parameters, trained on both Chinese and English social media conversations. To train such large models, we adopt the architecture of unified transformer with high computation and parameter efficiency. In addition, we carry out multi-party aware pre-training to better distinguish the characteristic information in social media conversations. With such designs, PLATO-XL successfully achieves superior performances as compared to other approaches in both Chinese and English chitchat. We"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.09519","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.09519/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.09519","created_at":"2026-07-05T05:08:14.205720+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.09519v2","created_at":"2026-07-05T05:08:14.205720+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.09519","created_at":"2026-07-05T05:08:14.205720+00:00"},{"alias_kind":"pith_short_12","alias_value":"5DQJGSYX6ZCJ","created_at":"2026-07-05T05:08:14.205720+00:00"},{"alias_kind":"pith_short_16","alias_value":"5DQJGSYX6ZCJ2K4W","created_at":"2026-07-05T05:08:14.205720+00:00"},{"alias_kind":"pith_short_8","alias_value":"5DQJGSYX","created_at":"2026-07-05T05:08:14.205720+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2205.01068","citing_title":"OPT: Open Pre-trained Transformer Language Models","ref_index":297,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3","json":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3.json","graph_json":"https://pith.science/api/pith-number/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/graph.json","events_json":"https://pith.science/api/pith-number/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/events.json","paper":"https://pith.science/paper/5DQJGSYX"},"agent_actions":{"view_html":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3","download_json":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3.json","view_paper":"https://pith.science/paper/5DQJGSYX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.09519&json=true","fetch_graph":"https://pith.science/api/pith-number/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/graph.json","fetch_events":"https://pith.science/api/pith-number/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/action/storage_attestation","attest_author":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/action/author_attestation","sign_citation":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/action/citation_signature","submit_replication":"https://pith.science/pith/5DQJGSYX6ZCJ2K4WQNOJBQNHB3/action/replication_record"}},"created_at":"2026-07-05T05:08:14.205720+00:00","updated_at":"2026-07-05T05:08:14.205720+00:00"}