{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ME5VXQXQX23KY4223DLMAJ6D6X","short_pith_number":"pith:ME5VXQXQ","schema_version":"1.0","canonical_sha256":"613b5bc2f0beb6ac735ad8d6c027c3f5e62d67128abfe0fcafe1c5ae6157f64c","source":{"kind":"arxiv","id":"2505.24015","version":1},"attestation_state":"computed","paper":{"title":"Semantics-Guided Generative Image Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.IV","authors_text":"Cheng-Lin Wu, Hyomin Choi, Ivan V. Baji\\'c","submitted_at":"2025-05-29T21:33:33Z","abstract_excerpt":"Advancements in text-to-image generative AI with large multimodal models are spreading into the field of image compression, creating high-quality representation of images at extremely low bit rates. This work introduces novel components to the existing multimodal image semantic compression (MISC) approach, enhancing the quality of the generated images in terms of PSNR and perceptual metrics. The new components include semantic segmentation guidance for the generative decoder, as well as content-adaptive diffusion, which controls the number of diffusion steps based on image characteristics. The"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24015","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.IV","submitted_at":"2025-05-29T21:33:33Z","cross_cats_sorted":[],"title_canon_sha256":"1d52a873edfe2242bca95ab4edcd56404958b38eea087ce20ee3d35735338856","abstract_canon_sha256":"6d54af51ca1c4ca6939686a8b51249cd104a158bbc1369d30b9b809a27733e7e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:40.167114Z","signature_b64":"PLGZEpaNtfJsrUlYyOQCTJaFNlfQPdLQX7P7V40uOe052wtwABvAirKebJAcpMPJdsLXtEnCSCxkDpX/5WuwAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"613b5bc2f0beb6ac735ad8d6c027c3f5e62d67128abfe0fcafe1c5ae6157f64c","last_reissued_at":"2026-07-05T11:12:40.166693Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:40.166693Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Semantics-Guided Generative Image Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.IV","authors_text":"Cheng-Lin Wu, Hyomin Choi, Ivan V. Baji\\'c","submitted_at":"2025-05-29T21:33:33Z","abstract_excerpt":"Advancements in text-to-image generative AI with large multimodal models are spreading into the field of image compression, creating high-quality representation of images at extremely low bit rates. This work introduces novel components to the existing multimodal image semantic compression (MISC) approach, enhancing the quality of the generated images in terms of PSNR and perceptual metrics. The new components include semantic segmentation guidance for the generative decoder, as well as content-adaptive diffusion, which controls the number of diffusion steps based on image characteristics. The"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24015","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24015/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24015","created_at":"2026-07-05T11:12:40.166754+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24015v1","created_at":"2026-07-05T11:12:40.166754+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24015","created_at":"2026-07-05T11:12:40.166754+00:00"},{"alias_kind":"pith_short_12","alias_value":"ME5VXQXQX23K","created_at":"2026-07-05T11:12:40.166754+00:00"},{"alias_kind":"pith_short_16","alias_value":"ME5VXQXQX23KY422","created_at":"2026-07-05T11:12:40.166754+00:00"},{"alias_kind":"pith_short_8","alias_value":"ME5VXQXQ","created_at":"2026-07-05T11:12:40.166754+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.12622","citing_title":"Efficient Semantic Image Communication for Traffic Monitoring at the Edge","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X","json":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X.json","graph_json":"https://pith.science/api/pith-number/ME5VXQXQX23KY4223DLMAJ6D6X/graph.json","events_json":"https://pith.science/api/pith-number/ME5VXQXQX23KY4223DLMAJ6D6X/events.json","paper":"https://pith.science/paper/ME5VXQXQ"},"agent_actions":{"view_html":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X","download_json":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X.json","view_paper":"https://pith.science/paper/ME5VXQXQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24015&json=true","fetch_graph":"https://pith.science/api/pith-number/ME5VXQXQX23KY4223DLMAJ6D6X/graph.json","fetch_events":"https://pith.science/api/pith-number/ME5VXQXQX23KY4223DLMAJ6D6X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X/action/storage_attestation","attest_author":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X/action/author_attestation","sign_citation":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X/action/citation_signature","submit_replication":"https://pith.science/pith/ME5VXQXQX23KY4223DLMAJ6D6X/action/replication_record"}},"created_at":"2026-07-05T11:12:40.166754+00:00","updated_at":"2026-07-05T11:12:40.166754+00:00"}