{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FL2AKGAKC7CGMMOB3PO4C2PVZ4","short_pith_number":"pith:FL2AKGAK","schema_version":"1.0","canonical_sha256":"2af405180a17c46631c1dbddc169f5cf23493df6c4ff2cba55a0c1996d0a7a6a","source":{"kind":"arxiv","id":"2507.09308","version":1},"attestation_state":"computed","paper":{"title":"AlphaVAE: Unified End-to-End RGBA Image Reconstruction and Generation with Alpha-Aware Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chun Yuan, Hao Yu, Jiabo Zhan, Zile Wang","submitted_at":"2025-07-12T14:53:42Z","abstract_excerpt":"Recent advances in latent diffusion models have achieved remarkable results in high-fidelity RGB image synthesis by leveraging pretrained VAEs to compress and reconstruct pixel data at low computational cost. However, the generation of transparent or layered content (RGBA image) remains largely unexplored, due to the lack of large-scale benchmarks. In this work, we propose ALPHA, the first comprehensive RGBA benchmark that adapts standard RGB metrics to four-channel images via alpha blending over canonical backgrounds. We further introduce ALPHAVAE, a unified end-to-end RGBA VAE that extends a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.09308","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-12T14:53:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b999f1b0c220b770060c866325ffd13cb34cbb7605baffd7c6c6d6eebe3d340d","abstract_canon_sha256":"a9f0657488c145ebe6d9cfde994e194ab6435db8c6de7f500c21e3369fbc1a29"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:07.023132Z","signature_b64":"a3Q620go9a6KOAjDDzRfFeNiMB99Kz2kSczPRg6D4olWxypNathyadu1eiKBjlaQqJaWIPWU26zRaVMhBvTpCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2af405180a17c46631c1dbddc169f5cf23493df6c4ff2cba55a0c1996d0a7a6a","last_reissued_at":"2026-07-05T11:36:07.022637Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:07.022637Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AlphaVAE: Unified End-to-End RGBA Image Reconstruction and Generation with Alpha-Aware Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chun Yuan, Hao Yu, Jiabo Zhan, Zile Wang","submitted_at":"2025-07-12T14:53:42Z","abstract_excerpt":"Recent advances in latent diffusion models have achieved remarkable results in high-fidelity RGB image synthesis by leveraging pretrained VAEs to compress and reconstruct pixel data at low computational cost. However, the generation of transparent or layered content (RGBA image) remains largely unexplored, due to the lack of large-scale benchmarks. In this work, we propose ALPHA, the first comprehensive RGBA benchmark that adapts standard RGB metrics to four-channel images via alpha blending over canonical backgrounds. We further introduce ALPHAVAE, a unified end-to-end RGBA VAE that extends a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.09308","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.09308/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.09308","created_at":"2026-07-05T11:36:07.022700+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.09308v1","created_at":"2026-07-05T11:36:07.022700+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.09308","created_at":"2026-07-05T11:36:07.022700+00:00"},{"alias_kind":"pith_short_12","alias_value":"FL2AKGAKC7CG","created_at":"2026-07-05T11:36:07.022700+00:00"},{"alias_kind":"pith_short_16","alias_value":"FL2AKGAKC7CGMMOB","created_at":"2026-07-05T11:36:07.022700+00:00"},{"alias_kind":"pith_short_8","alias_value":"FL2AKGAK","created_at":"2026-07-05T11:36:07.022700+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.20211","citing_title":"OmniAlpha: Aligning Transparency-Aware Generation via Multi-Task Unified Reinforcement Learning","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11818","citing_title":"RevealLayer: Disentangling Hidden and Visible Layers via Occlusion-Aware Image Decomposition","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10319","citing_title":"LimeCross: Context-Conditioned Layered Image Editing with Structural Consistency","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19858","citing_title":"Wan-Image: Pushing the Boundaries of Generative Visual Intelligence","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4","json":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4.json","graph_json":"https://pith.science/api/pith-number/FL2AKGAKC7CGMMOB3PO4C2PVZ4/graph.json","events_json":"https://pith.science/api/pith-number/FL2AKGAKC7CGMMOB3PO4C2PVZ4/events.json","paper":"https://pith.science/paper/FL2AKGAK"},"agent_actions":{"view_html":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4","download_json":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4.json","view_paper":"https://pith.science/paper/FL2AKGAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.09308&json=true","fetch_graph":"https://pith.science/api/pith-number/FL2AKGAKC7CGMMOB3PO4C2PVZ4/graph.json","fetch_events":"https://pith.science/api/pith-number/FL2AKGAKC7CGMMOB3PO4C2PVZ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4/action/storage_attestation","attest_author":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4/action/author_attestation","sign_citation":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4/action/citation_signature","submit_replication":"https://pith.science/pith/FL2AKGAKC7CGMMOB3PO4C2PVZ4/action/replication_record"}},"created_at":"2026-07-05T11:36:07.022700+00:00","updated_at":"2026-07-05T11:36:07.022700+00:00"}