{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:OTGK2K363QTTVMY4US74NHSKQT","short_pith_number":"pith:OTGK2K36","schema_version":"1.0","canonical_sha256":"74ccad2b7edc273ab31ca4bfc69e4a84e32af3a079643a4d0325b419c0f8d4e3","source":{"kind":"arxiv","id":"2608.05663","version":1},"attestation_state":"computed","paper":{"title":"Vorch-Streamer: Extending Human Audio-Visual Generation to Real-Time Long-Form Streaming","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"cs.CV","authors_text":"Haoran Yu, Junyi Chen, Lin Ma, Menglin Han, Xin Ma, Yang Ding, Yaohui Wang, Yulei Lu, Zhangkai Ni","submitted_at":"2026-08-06T07:00:37Z","abstract_excerpt":"Real-time long-form avatar audio--video generation requires causal, continuous synthesis while maintaining audiovisual synchronization and visual consistency. Adapting a pretrained bidirectional model to this setting presents two key dilemmas. First, autoregressively reusing generated blocks as context creates exposure bias, causing errors and visual drift to accumulate over long rollouts. Second, a global speech utterance does not indicates a causal generator which portion should be spoken next when only limited local audio--video context is available. We present \\textbf{Vorch-Streamer}, a po"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.05663","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-08-06T07:00:37Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"0a94148d7115eb5dd57fb91a4a4d1addd62d831e0b27d99de0a6d255b2a56100","abstract_canon_sha256":"a830e4057857923876412daf6c745f0f0d09d238f805f06c1756671742691a96"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-07T00:52:19.215902Z","signature_b64":"Ej1phAt6oIDXcJoFQ85dK49X6cbXse0W7JIuUy1pAFdeTE47u4ge1IRH1loXXdPPbTELWoDXXi1Uv9s7fBRpCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"74ccad2b7edc273ab31ca4bfc69e4a84e32af3a079643a4d0325b419c0f8d4e3","last_reissued_at":"2026-08-07T00:52:19.214441Z","signature_status":"signed_v1","first_computed_at":"2026-08-07T00:52:19.214441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vorch-Streamer: Extending Human Audio-Visual Generation to Real-Time Long-Form Streaming","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"cs.CV","authors_text":"Haoran Yu, Junyi Chen, Lin Ma, Menglin Han, Xin Ma, Yang Ding, Yaohui Wang, Yulei Lu, Zhangkai Ni","submitted_at":"2026-08-06T07:00:37Z","abstract_excerpt":"Real-time long-form avatar audio--video generation requires causal, continuous synthesis while maintaining audiovisual synchronization and visual consistency. Adapting a pretrained bidirectional model to this setting presents two key dilemmas. First, autoregressively reusing generated blocks as context creates exposure bias, causing errors and visual drift to accumulate over long rollouts. Second, a global speech utterance does not indicates a causal generator which portion should be spoken next when only limited local audio--video context is available. We present \\textbf{Vorch-Streamer}, a po"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.05663","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.05663/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.05663","created_at":"2026-08-07T00:52:19.215962+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.05663v1","created_at":"2026-08-07T00:52:19.215962+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.05663","created_at":"2026-08-07T00:52:19.215962+00:00"},{"alias_kind":"pith_short_12","alias_value":"OTGK2K363QTT","created_at":"2026-08-07T00:52:19.215962+00:00"},{"alias_kind":"pith_short_16","alias_value":"OTGK2K363QTTVMY4","created_at":"2026-08-07T00:52:19.215962+00:00"},{"alias_kind":"pith_short_8","alias_value":"OTGK2K36","created_at":"2026-08-07T00:52:19.215962+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT","json":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT.json","graph_json":"https://pith.science/api/pith-number/OTGK2K363QTTVMY4US74NHSKQT/graph.json","events_json":"https://pith.science/api/pith-number/OTGK2K363QTTVMY4US74NHSKQT/events.json","paper":"https://pith.science/paper/OTGK2K36"},"agent_actions":{"view_html":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT","download_json":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT.json","view_paper":"https://pith.science/paper/OTGK2K36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.05663&json=true","fetch_graph":"https://pith.science/api/pith-number/OTGK2K363QTTVMY4US74NHSKQT/graph.json","fetch_events":"https://pith.science/api/pith-number/OTGK2K363QTTVMY4US74NHSKQT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT/action/storage_attestation","attest_author":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT/action/author_attestation","sign_citation":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT/action/citation_signature","submit_replication":"https://pith.science/pith/OTGK2K363QTTVMY4US74NHSKQT/action/replication_record"}},"created_at":"2026-08-07T00:52:19.215962+00:00","updated_at":"2026-08-07T00:52:19.215962+00:00"}