{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WAQTAJ6VS4K2CCCSL74EJLQ5H4","short_pith_number":"pith:WAQTAJ6V","schema_version":"1.0","canonical_sha256":"b0213027d59715a108525ff844ae1d3f0feb14312e076a641b284e57080de390","source":{"kind":"arxiv","id":"2405.14598","version":2},"attestation_state":"computed","paper":{"title":"Visual Echoes: A Simple Unified Transformer for Audio-Visual Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Masato Ishii, Mengjie Zhao, Shiqi Yang, Shusuke Takahashi, Takashi Shibuya, Yuki Mitsufuji, Zhi Zhong","submitted_at":"2024-05-23T14:13:16Z","abstract_excerpt":"In recent years, with the realistic generation results and a wide range of personalized applications, diffusion-based generative models gain huge attention in both visual and audio generation areas. Compared to the considerable advancements of text2image or text2audio generation, research in audio2visual or visual2audio generation has been relatively slow. The recent audio-visual generation methods usually resort to huge large language model or composable diffusion models. Instead of designing another giant model for audio-visual generation, in this paper we take a step back showing a simple a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14598","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-23T14:13:16Z","cross_cats_sorted":["cs.LG","cs.MM","cs.SD","eess.AS"],"title_canon_sha256":"f48e3f04750db1392c2a7dd6aa61a8271b7770c57076a84f544f005bcccc127f","abstract_canon_sha256":"48fa00a08c3dcb4a3a004aa76038773f5924508173331a4307b3712fee53b412"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:41.740930Z","signature_b64":"y1nga//W2y22GLGQKwta11SI9r5zWzo88Yf5un/WpY2Utc0NjXeSDJiYNjRknnvUDprS9RIIXkNZK8Y7Xqx4Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0213027d59715a108525ff844ae1d3f0feb14312e076a641b284e57080de390","last_reissued_at":"2026-07-05T08:22:41.740459Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:41.740459Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Echoes: A Simple Unified Transformer for Audio-Visual Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Masato Ishii, Mengjie Zhao, Shiqi Yang, Shusuke Takahashi, Takashi Shibuya, Yuki Mitsufuji, Zhi Zhong","submitted_at":"2024-05-23T14:13:16Z","abstract_excerpt":"In recent years, with the realistic generation results and a wide range of personalized applications, diffusion-based generative models gain huge attention in both visual and audio generation areas. Compared to the considerable advancements of text2image or text2audio generation, research in audio2visual or visual2audio generation has been relatively slow. The recent audio-visual generation methods usually resort to huge large language model or composable diffusion models. Instead of designing another giant model for audio-visual generation, in this paper we take a step back showing a simple a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14598","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14598/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14598","created_at":"2026-07-05T08:22:41.740515+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14598v2","created_at":"2026-07-05T08:22:41.740515+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14598","created_at":"2026-07-05T08:22:41.740515+00:00"},{"alias_kind":"pith_short_12","alias_value":"WAQTAJ6VS4K2","created_at":"2026-07-05T08:22:41.740515+00:00"},{"alias_kind":"pith_short_16","alias_value":"WAQTAJ6VS4K2CCCS","created_at":"2026-07-05T08:22:41.740515+00:00"},{"alias_kind":"pith_short_8","alias_value":"WAQTAJ6V","created_at":"2026-07-05T08:22:41.740515+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04955","citing_title":"EXPOTION: Facial Expression and Motion Control for Multimodal Music Generation","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4","json":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4.json","graph_json":"https://pith.science/api/pith-number/WAQTAJ6VS4K2CCCSL74EJLQ5H4/graph.json","events_json":"https://pith.science/api/pith-number/WAQTAJ6VS4K2CCCSL74EJLQ5H4/events.json","paper":"https://pith.science/paper/WAQTAJ6V"},"agent_actions":{"view_html":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4","download_json":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4.json","view_paper":"https://pith.science/paper/WAQTAJ6V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14598&json=true","fetch_graph":"https://pith.science/api/pith-number/WAQTAJ6VS4K2CCCSL74EJLQ5H4/graph.json","fetch_events":"https://pith.science/api/pith-number/WAQTAJ6VS4K2CCCSL74EJLQ5H4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4/action/storage_attestation","attest_author":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4/action/author_attestation","sign_citation":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4/action/citation_signature","submit_replication":"https://pith.science/pith/WAQTAJ6VS4K2CCCSL74EJLQ5H4/action/replication_record"}},"created_at":"2026-07-05T08:22:41.740515+00:00","updated_at":"2026-07-05T08:22:41.740515+00:00"}