{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LGAZQ6WXZ3O32LRJWVGHIIP72K","short_pith_number":"pith:LGAZQ6WX","schema_version":"1.0","canonical_sha256":"5981987ad7ceddbd2e29b54c7421ffd29caf780ee49704d38adad7999fcf1518","source":{"kind":"arxiv","id":"2310.14400","version":1},"attestation_state":"computed","paper":{"title":"A Pytorch Reproduction of Masked Generative Image Transformer","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Mickael Chen, Victor Besnier","submitted_at":"2023-10-22T20:21:11Z","abstract_excerpt":"In this technical report, we present a reproduction of MaskGIT: Masked Generative Image Transformer, using PyTorch. The approach involves leveraging a masked bidirectional transformer architecture, enabling image generation with only few steps (8~16 steps) for 512 x 512 resolution images, i.e., ~64x faster than an auto-regressive approach. Through rigorous experimentation and optimization, we achieved results that closely align with the findings presented in the original paper. We match the reported FID of 7.32 with our replication and obtain 7.59 with similar hyperparameters on ImageNet at re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.14400","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-22T20:21:11Z","cross_cats_sorted":[],"title_canon_sha256":"df04aa8e629c610a334b8a36782d370eebf4f2ba6792dff223f66abf966967be","abstract_canon_sha256":"cacee8e069782d7b3046ecbaad677243d80db0969abb662e8bdaf64410d4ba5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:43.914154Z","signature_b64":"QaXx61OaoIIOPczDtXKwyitw+ntF0Aud0R9zuxWEEeybGrbDa0v9GugmGzDfIB/k5E5JXyZbUZdKtCNHQMoJAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5981987ad7ceddbd2e29b54c7421ffd29caf780ee49704d38adad7999fcf1518","last_reissued_at":"2026-07-05T07:03:43.913673Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:43.913673Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Pytorch Reproduction of Masked Generative Image Transformer","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Mickael Chen, Victor Besnier","submitted_at":"2023-10-22T20:21:11Z","abstract_excerpt":"In this technical report, we present a reproduction of MaskGIT: Masked Generative Image Transformer, using PyTorch. The approach involves leveraging a masked bidirectional transformer architecture, enabling image generation with only few steps (8~16 steps) for 512 x 512 resolution images, i.e., ~64x faster than an auto-regressive approach. Through rigorous experimentation and optimization, we achieved results that closely align with the findings presented in the original paper. We match the reported FID of 7.32 with our replication and obtain 7.59 with similar hyperparameters on ImageNet at re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.14400","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.14400/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.14400","created_at":"2026-07-05T07:03:43.913736+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.14400v1","created_at":"2026-07-05T07:03:43.913736+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.14400","created_at":"2026-07-05T07:03:43.913736+00:00"},{"alias_kind":"pith_short_12","alias_value":"LGAZQ6WXZ3O3","created_at":"2026-07-05T07:03:43.913736+00:00"},{"alias_kind":"pith_short_16","alias_value":"LGAZQ6WXZ3O32LRJ","created_at":"2026-07-05T07:03:43.913736+00:00"},{"alias_kind":"pith_short_8","alias_value":"LGAZQ6WX","created_at":"2026-07-05T07:03:43.913736+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18250","citing_title":"Future Dynamic 3D Reconstruction: A 3D World Model with Disentangled Ego-Motion","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00773","citing_title":"Accelerating Discrete Diffusion Models with Parallel-In-Time Sampling","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21484","citing_title":"One-Step Distillation of Discrete Diffusion Image Generators via Fixed-Point Iteration","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04525","citing_title":"Demystifying MaskGIT Sampler and Beyond: Adaptive Order Selection in Masked Diffusion","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11707","citing_title":"Representations Before Pixels: Semantics-Guided Hierarchical Video Prediction","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K","json":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K.json","graph_json":"https://pith.science/api/pith-number/LGAZQ6WXZ3O32LRJWVGHIIP72K/graph.json","events_json":"https://pith.science/api/pith-number/LGAZQ6WXZ3O32LRJWVGHIIP72K/events.json","paper":"https://pith.science/paper/LGAZQ6WX"},"agent_actions":{"view_html":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K","download_json":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K.json","view_paper":"https://pith.science/paper/LGAZQ6WX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.14400&json=true","fetch_graph":"https://pith.science/api/pith-number/LGAZQ6WXZ3O32LRJWVGHIIP72K/graph.json","fetch_events":"https://pith.science/api/pith-number/LGAZQ6WXZ3O32LRJWVGHIIP72K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K/action/storage_attestation","attest_author":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K/action/author_attestation","sign_citation":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K/action/citation_signature","submit_replication":"https://pith.science/pith/LGAZQ6WXZ3O32LRJWVGHIIP72K/action/replication_record"}},"created_at":"2026-07-05T07:03:43.913736+00:00","updated_at":"2026-07-05T07:03:43.913736+00:00"}