{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BIAW5AOTGKUP5MLPVYA5FETX4U","short_pith_number":"pith:BIAW5AOT","schema_version":"1.0","canonical_sha256":"0a016e81d332a8feb16fae01d29277e5187f2abeeed925c380ae0c5be1fd0aa1","source":{"kind":"arxiv","id":"2410.14101","version":2},"attestation_state":"computed","paper":{"title":"Multi-Source Spatial Knowledge Understanding for Immersive Visual Text-to-Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Rui Liu, Shuwei He","submitted_at":"2024-10-18T00:46:18Z","abstract_excerpt":"Visual Text-to-Speech (VTTS) aims to take the environmental image as the prompt to synthesize reverberant speech for the spoken content. Previous works focus on the RGB modality for global environmental modeling, overlooking the potential of multi-source spatial knowledge like depth, speaker position, and environmental semantics. To address these issues, we propose a novel multi-source spatial knowledge understanding scheme for immersive VTTS, termed MS2KU-VTTS. Specifically, we first prioritize RGB image as the dominant source and consider depth image, speaker position knowledge from object d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14101","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-10-18T00:46:18Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"f03e21fdb624212397d20baf649c862a3854fca3aa393f2825cf76db68a4b4d5","abstract_canon_sha256":"d49a8075095822767f8805db54ecac3893df96f2e628880b5b4f2cc2ce0b64ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:06.370793Z","signature_b64":"PWm1K+ftyJMH72QT2AkQa+qjPQBnQXUF0AjAhDm6+iD8sp9GtVrHXc9Mku4lARo2RVodtT36SRd6FDFUMVjuAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a016e81d332a8feb16fae01d29277e5187f2abeeed925c380ae0c5be1fd0aa1","last_reissued_at":"2026-07-05T09:53:06.370234Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:06.370234Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Source Spatial Knowledge Understanding for Immersive Visual Text-to-Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Rui Liu, Shuwei He","submitted_at":"2024-10-18T00:46:18Z","abstract_excerpt":"Visual Text-to-Speech (VTTS) aims to take the environmental image as the prompt to synthesize reverberant speech for the spoken content. Previous works focus on the RGB modality for global environmental modeling, overlooking the potential of multi-source spatial knowledge like depth, speaker position, and environmental semantics. To address these issues, we propose a novel multi-source spatial knowledge understanding scheme for immersive VTTS, termed MS2KU-VTTS. Specifically, we first prioritize RGB image as the dominant source and consider depth image, speaker position knowledge from object d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14101","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14101/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14101","created_at":"2026-07-05T09:53:06.370302+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14101v2","created_at":"2026-07-05T09:53:06.370302+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14101","created_at":"2026-07-05T09:53:06.370302+00:00"},{"alias_kind":"pith_short_12","alias_value":"BIAW5AOTGKUP","created_at":"2026-07-05T09:53:06.370302+00:00"},{"alias_kind":"pith_short_16","alias_value":"BIAW5AOTGKUP5MLP","created_at":"2026-07-05T09:53:06.370302+00:00"},{"alias_kind":"pith_short_8","alias_value":"BIAW5AOT","created_at":"2026-07-05T09:53:06.370302+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.18748","citing_title":"Towards Expressive Video Dubbing with Multiscale Multimodal Context Interaction","ref_index":33,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U","json":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U.json","graph_json":"https://pith.science/api/pith-number/BIAW5AOTGKUP5MLPVYA5FETX4U/graph.json","events_json":"https://pith.science/api/pith-number/BIAW5AOTGKUP5MLPVYA5FETX4U/events.json","paper":"https://pith.science/paper/BIAW5AOT"},"agent_actions":{"view_html":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U","download_json":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U.json","view_paper":"https://pith.science/paper/BIAW5AOT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14101&json=true","fetch_graph":"https://pith.science/api/pith-number/BIAW5AOTGKUP5MLPVYA5FETX4U/graph.json","fetch_events":"https://pith.science/api/pith-number/BIAW5AOTGKUP5MLPVYA5FETX4U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U/action/storage_attestation","attest_author":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U/action/author_attestation","sign_citation":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U/action/citation_signature","submit_replication":"https://pith.science/pith/BIAW5AOTGKUP5MLPVYA5FETX4U/action/replication_record"}},"created_at":"2026-07-05T09:53:06.370302+00:00","updated_at":"2026-07-05T09:53:06.370302+00:00"}