{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5P62PVJ5RNFFEOBZA6GXBIF4LG","short_pith_number":"pith:5P62PVJ5","schema_version":"1.0","canonical_sha256":"ebfda7d53d8b4a523839078d70a0bc5984335c7ef6d4cf13fec9be80fb9c61c5","source":{"kind":"arxiv","id":"2308.12219","version":3},"attestation_state":"computed","paper":{"title":"Diffusion Language Models Can Perform Many Tasks with Scaling and Instruction-Finetuning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiasheng Ye, Lihua Qian, Quanquan Gu, Yu Bao, Zaixiang Zheng","submitted_at":"2023-08-23T16:01:12Z","abstract_excerpt":"The recent surge of generative AI has been fueled by the generative power of diffusion probabilistic models and the scalable capabilities of large language models. Despite their potential, it remains elusive whether diffusion language models can solve general language tasks comparable to their autoregressive counterparts. This paper demonstrates that scaling diffusion models w.r.t. data, sizes, and tasks can effectively make them strong language learners. We build competent diffusion language models at scale by first acquiring knowledge from massive data via masked language modeling pretrainin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.12219","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-23T16:01:12Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8730dac2510c8eae1a4a8f9a2bf3e91065daf6c3b9d1f38ae143141e68cd5c61","abstract_canon_sha256":"37c0dbc6cbdea8a1385b5a3570b54427d24f8c0a85f4cc8363e7d9349bd68dc2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:32.592967Z","signature_b64":"EKqmZNq8apQIM+jix7ms7V+JYL/oN/LXj/ngPkqfLj3oJBvrMIaFmQD5yWBbHYU2FHA2WESWL88d4aALSmJLCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ebfda7d53d8b4a523839078d70a0bc5984335c7ef6d4cf13fec9be80fb9c61c5","last_reissued_at":"2026-07-05T10:18:32.592518Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:32.592518Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diffusion Language Models Can Perform Many Tasks with Scaling and Instruction-Finetuning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiasheng Ye, Lihua Qian, Quanquan Gu, Yu Bao, Zaixiang Zheng","submitted_at":"2023-08-23T16:01:12Z","abstract_excerpt":"The recent surge of generative AI has been fueled by the generative power of diffusion probabilistic models and the scalable capabilities of large language models. Despite their potential, it remains elusive whether diffusion language models can solve general language tasks comparable to their autoregressive counterparts. This paper demonstrates that scaling diffusion models w.r.t. data, sizes, and tasks can effectively make them strong language learners. We build competent diffusion language models at scale by first acquiring knowledge from massive data via masked language modeling pretrainin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.12219","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.12219/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.12219","created_at":"2026-07-05T10:18:32.592574+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.12219v3","created_at":"2026-07-05T10:18:32.592574+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.12219","created_at":"2026-07-05T10:18:32.592574+00:00"},{"alias_kind":"pith_short_12","alias_value":"5P62PVJ5RNFF","created_at":"2026-07-05T10:18:32.592574+00:00"},{"alias_kind":"pith_short_16","alias_value":"5P62PVJ5RNFFEOBZ","created_at":"2026-07-05T10:18:32.592574+00:00"},{"alias_kind":"pith_short_8","alias_value":"5P62PVJ5","created_at":"2026-07-05T10:18:32.592574+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22976","citing_title":"Understanding Parallel Samplers in Masked Diffusion via Random Walks on Graphs","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08554","citing_title":"A Theoretical Analysis of Memory and Overfitting Phenomena in Stochastic Interpolation Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00182","citing_title":"Towards A Generative Protein Evolution Machine with DPLM-Evo","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17119","citing_title":"Diffusion and Flow Matching Models for Tabular Data: A Survey","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17891","citing_title":"Scaling Diffusion Language Models via Adaptation from Autoregressive Models","ref_index":198,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19726","citing_title":"Efficient Long-Context Modeling in Diffusion Language Models via Block Approximate Sparse Attention","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16933","citing_title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2505.22618","citing_title":"Fast-dLLM: Training-free Acceleration of Diffusion LLM by Enabling KV Cache and Parallel Decoding","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00182","citing_title":"Towards A Generative Protein Evolution Machine with DPLM-Evo","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10218","citing_title":"Relative Score Policy Optimization for Diffusion Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23235","citing_title":"Measuring Temporal Linguistic Emergence in Diffusion Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15487","citing_title":"Dream 7B: Diffusion Large Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00182","citing_title":"Towards A Generative Protein Evolution Machine with DPLM-Evo","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09992","citing_title":"Large Language Diffusion Models","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04291","citing_title":"Leveraging Pretrained Language Models as Energy Functions for Glauber Dynamics Text Diffusion","ref_index":112,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG","json":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG.json","graph_json":"https://pith.science/api/pith-number/5P62PVJ5RNFFEOBZA6GXBIF4LG/graph.json","events_json":"https://pith.science/api/pith-number/5P62PVJ5RNFFEOBZA6GXBIF4LG/events.json","paper":"https://pith.science/paper/5P62PVJ5"},"agent_actions":{"view_html":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG","download_json":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG.json","view_paper":"https://pith.science/paper/5P62PVJ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.12219&json=true","fetch_graph":"https://pith.science/api/pith-number/5P62PVJ5RNFFEOBZA6GXBIF4LG/graph.json","fetch_events":"https://pith.science/api/pith-number/5P62PVJ5RNFFEOBZA6GXBIF4LG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG/action/storage_attestation","attest_author":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG/action/author_attestation","sign_citation":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG/action/citation_signature","submit_replication":"https://pith.science/pith/5P62PVJ5RNFFEOBZA6GXBIF4LG/action/replication_record"}},"created_at":"2026-07-05T10:18:32.592574+00:00","updated_at":"2026-07-05T10:18:32.592574+00:00"}