{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GI57ZYXZCUNITZNKYMIX2SIJKJ","short_pith_number":"pith:GI57ZYXZ","schema_version":"1.0","canonical_sha256":"323bfce2f9151a89e5aac3117d4909527d3603bfa0236d8525f4d4a7aafd4016","source":{"kind":"arxiv","id":"2507.23268","version":2},"attestation_state":"computed","paper":{"title":"PixNerd: Pixel Neural Field Diffusion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenhui Zhu, Limin Wang, Shuai Wang, Weilin Huang, Ziteng Gao","submitted_at":"2025-07-31T06:07:20Z","abstract_excerpt":"The current success of diffusion transformers heavily depends on the compressed latent space shaped by the pre-trained variational autoencoder(VAE). However, this two-stage training paradigm inevitably introduces accumulated errors and decoding artifacts. To address the aforementioned problems, researchers return to pixel space at the cost of complicated cascade pipelines and increased token complexity. In contrast to their efforts, we propose to model the patch-wise decoding with neural field and present a single-scale, single-stage, efficient, end-to-end solution, coined as pixel neural fiel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.23268","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-31T06:07:20Z","cross_cats_sorted":[],"title_canon_sha256":"f24ddc0f4b2c99a09e5c7c00d9853bd8f40f1153bc67ec6917fc86818dd2ccb5","abstract_canon_sha256":"22bfcb1727f29ac4722f177eb417cbcb71aa4d6642d5a64b53a6fbe054215c8c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:47:27.680291Z","signature_b64":"zlKb4bSoI/sj4Zabvl/vwqYDiDyMjpWmeT4612ULNpCtq1Dx3hCgiQeG1m+soI1V5W3IKHf3IlUFK3+Xc1sxAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"323bfce2f9151a89e5aac3117d4909527d3603bfa0236d8525f4d4a7aafd4016","last_reissued_at":"2026-07-05T11:47:27.679723Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:47:27.679723Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PixNerd: Pixel Neural Field Diffusion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenhui Zhu, Limin Wang, Shuai Wang, Weilin Huang, Ziteng Gao","submitted_at":"2025-07-31T06:07:20Z","abstract_excerpt":"The current success of diffusion transformers heavily depends on the compressed latent space shaped by the pre-trained variational autoencoder(VAE). However, this two-stage training paradigm inevitably introduces accumulated errors and decoding artifacts. To address the aforementioned problems, researchers return to pixel space at the cost of complicated cascade pipelines and increased token complexity. In contrast to their efforts, we propose to model the patch-wise decoding with neural field and present a single-scale, single-stage, efficient, end-to-end solution, coined as pixel neural fiel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.23268","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.23268/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.23268","created_at":"2026-07-05T11:47:27.679790+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.23268v2","created_at":"2026-07-05T11:47:27.679790+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.23268","created_at":"2026-07-05T11:47:27.679790+00:00"},{"alias_kind":"pith_short_12","alias_value":"GI57ZYXZCUNI","created_at":"2026-07-05T11:47:27.679790+00:00"},{"alias_kind":"pith_short_16","alias_value":"GI57ZYXZCUNITZNK","created_at":"2026-07-05T11:47:27.679790+00:00"},{"alias_kind":"pith_short_8","alias_value":"GI57ZYXZ","created_at":"2026-07-05T11:47:27.679790+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26016","citing_title":"MIMFlow: Integrating Masked Image Modeling with Normalizing Flows for End-to-End Image Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24888","citing_title":"DiffusionBench: On Holistic Evaluation of Diffusion Transformers","ref_index":171,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01803","citing_title":"PixGS: Pixel-Space Diffusion for Direct 3D Gaussian Splat Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09048","citing_title":"BareWave: Waveform-Native Flow-Matching Text-to-Speech","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03455","citing_title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31604","citing_title":"Representation Forcing for Bottleneck-Free Unified Multimodal Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15741","citing_title":"HyperDiT: Hyper-Connected Transformers for High-Fidelity Pixel-Space Diffusion","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18267","citing_title":"SRC-Flow: Compact Semantic Representations Enable Normalizing Flows for Image Generation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27760","citing_title":"PixelU: A U-Shaped Transformer for Efficient End-to-End Pixel Diffusion","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27978","citing_title":"Parallel Rollout Approximation for Pixel-Space Autoregressive Image Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21981","citing_title":"RiT: Vanilla Diffusion Transformers Suffice in Representation Space","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15741","citing_title":"HyperDiT: Hyper-Connected Transformers for High-Fidelity Pixel-Space Diffusion","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17759","citing_title":"FrequencyBooster: Full-Frequency Modeling for High-Fidelity Pixel Diffusion","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18267","citing_title":"SRC-Flow: Compact Semantic Representations Enable Normalizing Flows for Image Generation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20147","citing_title":"PixVerve: Advancing Native UHR Image Generation to 100MP with a Large-Scale High-Quality Dataset","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19365","citing_title":"DeCo: Frequency-Decoupled Pixel Diffusion for End-to-End Image Generation","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20645","citing_title":"PixelDiT: Pixel Diffusion Transformers for Image Generation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02493","citing_title":"PixelGen: Improving Pixel Diffusion with Perceptual Supervision","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12013","citing_title":"L2P: Unlocking Latent Potential for Pixel Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2511.13720","citing_title":"Back to Basics: Let Denoising Generative Models Denoise","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06421","citing_title":"FREPix: Frequency-Heterogeneous Flow Matching for Pixel-Space Image Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20041","citing_title":"Normalizing Flows with Iterative Denoising","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12525","citing_title":"CoD-Lite: Real-Time Diffusion-Based Generative Image Compression","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11521","citing_title":"Continuous Adversarial Flow Models","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16558","citing_title":"Cross-Modal Generation: From Commodity WiFi to High-Fidelity mmWave and RFID Sensing","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ","json":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ.json","graph_json":"https://pith.science/api/pith-number/GI57ZYXZCUNITZNKYMIX2SIJKJ/graph.json","events_json":"https://pith.science/api/pith-number/GI57ZYXZCUNITZNKYMIX2SIJKJ/events.json","paper":"https://pith.science/paper/GI57ZYXZ"},"agent_actions":{"view_html":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ","download_json":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ.json","view_paper":"https://pith.science/paper/GI57ZYXZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.23268&json=true","fetch_graph":"https://pith.science/api/pith-number/GI57ZYXZCUNITZNKYMIX2SIJKJ/graph.json","fetch_events":"https://pith.science/api/pith-number/GI57ZYXZCUNITZNKYMIX2SIJKJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ/action/storage_attestation","attest_author":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ/action/author_attestation","sign_citation":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ/action/citation_signature","submit_replication":"https://pith.science/pith/GI57ZYXZCUNITZNKYMIX2SIJKJ/action/replication_record"}},"created_at":"2026-07-05T11:47:27.679790+00:00","updated_at":"2026-07-05T11:47:27.679790+00:00"}