{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F7WTJQEMEGX6OYYVRQPJQZJ6LW","short_pith_number":"pith:F7WTJQEM","schema_version":"1.0","canonical_sha256":"2fed34c08c21afe763158c1e98653e5db8759c8d32dc0311eb3a465edf0bfe10","source":{"kind":"arxiv","id":"2407.21046","version":1},"attestation_state":"computed","paper":{"title":"Promises and Pitfalls of Generative Masked Language Modeling: Theoretical Framework and Practical Guidelines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aashay Mehta, Alexandre Kirchmeyer, Andrej Risteski, Boris Dadachev, Kishore Papineni, Sanjiv Kumar, Yilong Qin, Yuchen Li","submitted_at":"2024-07-22T18:00:00Z","abstract_excerpt":"Autoregressive language models are the currently dominant paradigm for text generation, but they have some fundamental limitations that cannot be remedied by scale-for example inherently sequential and unidirectional generation. While alternate classes of models have been explored, we have limited mathematical understanding of their fundamental power and limitations. In this paper we focus on Generative Masked Language Models (GMLMs), a non-autoregressive paradigm in which we train a model to fit conditional probabilities of the data distribution via masking, which are subsequently used as inp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.21046","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-22T18:00:00Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"857cd5f4dcaacde707afd23e7d44286af041f7991e1c0e2dd4211cd4ddad9a6e","abstract_canon_sha256":"e6075b5dadacc0fa19efe825ca01b214f226e08477d7495c04b3e63e7d13da13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:32.831279Z","signature_b64":"EG8si2UlF32JnO/opX9wyk1llnv/VWJI5jKZxcH2bdoDUQ2IIra1G9TPpaGhDWTcvTUxC+YzR1VfRDCSg2WWBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fed34c08c21afe763158c1e98653e5db8759c8d32dc0311eb3a465edf0bfe10","last_reissued_at":"2026-07-05T08:50:32.830738Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:32.830738Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Promises and Pitfalls of Generative Masked Language Modeling: Theoretical Framework and Practical Guidelines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aashay Mehta, Alexandre Kirchmeyer, Andrej Risteski, Boris Dadachev, Kishore Papineni, Sanjiv Kumar, Yilong Qin, Yuchen Li","submitted_at":"2024-07-22T18:00:00Z","abstract_excerpt":"Autoregressive language models are the currently dominant paradigm for text generation, but they have some fundamental limitations that cannot be remedied by scale-for example inherently sequential and unidirectional generation. While alternate classes of models have been explored, we have limited mathematical understanding of their fundamental power and limitations. In this paper we focus on Generative Masked Language Models (GMLMs), a non-autoregressive paradigm in which we train a model to fit conditional probabilities of the data distribution via masking, which are subsequently used as inp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.21046","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.21046/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.21046","created_at":"2026-07-05T08:50:32.830796+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.21046v1","created_at":"2026-07-05T08:50:32.830796+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.21046","created_at":"2026-07-05T08:50:32.830796+00:00"},{"alias_kind":"pith_short_12","alias_value":"F7WTJQEMEGX6","created_at":"2026-07-05T08:50:32.830796+00:00"},{"alias_kind":"pith_short_16","alias_value":"F7WTJQEMEGX6OYYV","created_at":"2026-07-05T08:50:32.830796+00:00"},{"alias_kind":"pith_short_8","alias_value":"F7WTJQEM","created_at":"2026-07-05T08:50:32.830796+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14567","citing_title":"Scaling Laws from Sequential Feature Recovery: A Solvable Hierarchical Model","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW","json":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW.json","graph_json":"https://pith.science/api/pith-number/F7WTJQEMEGX6OYYVRQPJQZJ6LW/graph.json","events_json":"https://pith.science/api/pith-number/F7WTJQEMEGX6OYYVRQPJQZJ6LW/events.json","paper":"https://pith.science/paper/F7WTJQEM"},"agent_actions":{"view_html":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW","download_json":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW.json","view_paper":"https://pith.science/paper/F7WTJQEM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.21046&json=true","fetch_graph":"https://pith.science/api/pith-number/F7WTJQEMEGX6OYYVRQPJQZJ6LW/graph.json","fetch_events":"https://pith.science/api/pith-number/F7WTJQEMEGX6OYYVRQPJQZJ6LW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW/action/storage_attestation","attest_author":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW/action/author_attestation","sign_citation":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW/action/citation_signature","submit_replication":"https://pith.science/pith/F7WTJQEMEGX6OYYVRQPJQZJ6LW/action/replication_record"}},"created_at":"2026-07-05T08:50:32.830796+00:00","updated_at":"2026-07-05T08:50:32.830796+00:00"}