{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6UU2I5SFCI2NEXZSK43Q4K4WF5","short_pith_number":"pith:6UU2I5SF","schema_version":"1.0","canonical_sha256":"f529a476451234d25f3257370e2b962f77bcf2176baf56b7bcb56ec5a890e66a","source":{"kind":"arxiv","id":"2401.08655","version":2},"attestation_state":"computed","paper":{"title":"SAiD: Speech-driven Blendshape Facial Animation with Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Inkyu Park, Jaewoong Cho","submitted_at":"2023-12-25T04:40:32Z","abstract_excerpt":"Speech-driven 3D facial animation is challenging due to the scarcity of large-scale visual-audio datasets despite extensive research. Most prior works, typically focused on learning regression models on a small dataset using the method of least squares, encounter difficulties generating diverse lip movements from speech and require substantial effort in refining the generated outputs. To address these issues, we propose a speech-driven 3D facial animation with a diffusion model (SAiD), a lightweight Transformer-based U-Net with a cross-modality alignment bias between audio and visual to enhanc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.08655","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-25T04:40:32Z","cross_cats_sorted":["cs.AI","cs.GR","cs.LG","cs.MM"],"title_canon_sha256":"4695cf3571076647fb40cba411feb101a30e031042a27426be251546536adbfa","abstract_canon_sha256":"77188e56f8e6ae9589e308987caed23d8d58b3accea3ad936ee35d69f7e9bfdd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:37:15.143964Z","signature_b64":"jZSIxgs3wdkZL1Ow/aTXcHV8AHPTUWFWOPCynAGtE88vvQfT/0MMRGFkYi49EECrNN7V6DnH009k6Nd2flDHAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f529a476451234d25f3257370e2b962f77bcf2176baf56b7bcb56ec5a890e66a","last_reissued_at":"2026-07-05T07:37:15.143489Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:37:15.143489Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SAiD: Speech-driven Blendshape Facial Animation with Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Inkyu Park, Jaewoong Cho","submitted_at":"2023-12-25T04:40:32Z","abstract_excerpt":"Speech-driven 3D facial animation is challenging due to the scarcity of large-scale visual-audio datasets despite extensive research. Most prior works, typically focused on learning regression models on a small dataset using the method of least squares, encounter difficulties generating diverse lip movements from speech and require substantial effort in refining the generated outputs. To address these issues, we propose a speech-driven 3D facial animation with a diffusion model (SAiD), a lightweight Transformer-based U-Net with a cross-modality alignment bias between audio and visual to enhanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.08655","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.08655/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.08655","created_at":"2026-07-05T07:37:15.143548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.08655v2","created_at":"2026-07-05T07:37:15.143548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.08655","created_at":"2026-07-05T07:37:15.143548+00:00"},{"alias_kind":"pith_short_12","alias_value":"6UU2I5SFCI2N","created_at":"2026-07-05T07:37:15.143548+00:00"},{"alias_kind":"pith_short_16","alias_value":"6UU2I5SFCI2NEXZS","created_at":"2026-07-05T07:37:15.143548+00:00"},{"alias_kind":"pith_short_8","alias_value":"6UU2I5SF","created_at":"2026-07-05T07:37:15.143548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07478","citing_title":"AudioFace: Language-Assisted Speech-Driven Facial Animation with Multimodal Language Models","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5","json":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5.json","graph_json":"https://pith.science/api/pith-number/6UU2I5SFCI2NEXZSK43Q4K4WF5/graph.json","events_json":"https://pith.science/api/pith-number/6UU2I5SFCI2NEXZSK43Q4K4WF5/events.json","paper":"https://pith.science/paper/6UU2I5SF"},"agent_actions":{"view_html":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5","download_json":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5.json","view_paper":"https://pith.science/paper/6UU2I5SF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.08655&json=true","fetch_graph":"https://pith.science/api/pith-number/6UU2I5SFCI2NEXZSK43Q4K4WF5/graph.json","fetch_events":"https://pith.science/api/pith-number/6UU2I5SFCI2NEXZSK43Q4K4WF5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5/action/storage_attestation","attest_author":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5/action/author_attestation","sign_citation":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5/action/citation_signature","submit_replication":"https://pith.science/pith/6UU2I5SFCI2NEXZSK43Q4K4WF5/action/replication_record"}},"created_at":"2026-07-05T07:37:15.143548+00:00","updated_at":"2026-07-05T07:37:15.143548+00:00"}