{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KL5R7FWJQOGBTEWFGOQHJ74GTK","short_pith_number":"pith:KL5R7FWJ","schema_version":"1.0","canonical_sha256":"52fb1f96c9838c1992c533a074ff869a949c74d08a4a3b14a9a7ca70b6e4677d","source":{"kind":"arxiv","id":"2404.01339","version":1},"attestation_state":"computed","paper":{"title":"Humane Speech Synthesis through Zero-Shot Emotion and Disfluency Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Aakash Garg, Jinsil Hwaryoung Seo, Mihir Godbole, Rohan Chaudhury","submitted_at":"2024-03-31T00:38:02Z","abstract_excerpt":"Contemporary conversational systems often present a significant limitation: their responses lack the emotional depth and disfluent characteristic of human interactions. This absence becomes particularly noticeable when users seek more personalized and empathetic interactions. Consequently, this makes them seem mechanical and less relatable to human users. Recognizing this gap, we embarked on a journey to humanize machine communication, to ensure AI systems not only comprehend but also resonate. To address this shortcoming, we have designed an innovative speech synthesis pipeline. Within this f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.01339","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-31T00:38:02Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"26b42873bd5d881d74c2675c172dacd3dfcad6980c27fc1dcc39b329ca3a6c41","abstract_canon_sha256":"0b8055f05a7e5cae9a19a44024a5809610e86d67638c0f5d786a8997423900a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:03:12.833496Z","signature_b64":"CJ0O782jtFyB7WkO8GYTZdI+CfOWASTxABgPg0rgEtLgNm7z1kjT+vbmwM6t9lKsMtlFRpQTrBNIQXND1Z8GCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52fb1f96c9838c1992c533a074ff869a949c74d08a4a3b14a9a7ca70b6e4677d","last_reissued_at":"2026-07-05T08:03:12.833037Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:03:12.833037Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Humane Speech Synthesis through Zero-Shot Emotion and Disfluency Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Aakash Garg, Jinsil Hwaryoung Seo, Mihir Godbole, Rohan Chaudhury","submitted_at":"2024-03-31T00:38:02Z","abstract_excerpt":"Contemporary conversational systems often present a significant limitation: their responses lack the emotional depth and disfluent characteristic of human interactions. This absence becomes particularly noticeable when users seek more personalized and empathetic interactions. Consequently, this makes them seem mechanical and less relatable to human users. Recognizing this gap, we embarked on a journey to humanize machine communication, to ensure AI systems not only comprehend but also resonate. To address this shortcoming, we have designed an innovative speech synthesis pipeline. Within this f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01339","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01339/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.01339","created_at":"2026-07-05T08:03:12.833095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.01339v1","created_at":"2026-07-05T08:03:12.833095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01339","created_at":"2026-07-05T08:03:12.833095+00:00"},{"alias_kind":"pith_short_12","alias_value":"KL5R7FWJQOGB","created_at":"2026-07-05T08:03:12.833095+00:00"},{"alias_kind":"pith_short_16","alias_value":"KL5R7FWJQOGBTEWF","created_at":"2026-07-05T08:03:12.833095+00:00"},{"alias_kind":"pith_short_8","alias_value":"KL5R7FWJ","created_at":"2026-07-05T08:03:12.833095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.04195","citing_title":"NVSpeech: An Integrated and Scalable Pipeline for Human-Like Speech Modeling with Paralinguistic Vocalizations","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK","json":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK.json","graph_json":"https://pith.science/api/pith-number/KL5R7FWJQOGBTEWFGOQHJ74GTK/graph.json","events_json":"https://pith.science/api/pith-number/KL5R7FWJQOGBTEWFGOQHJ74GTK/events.json","paper":"https://pith.science/paper/KL5R7FWJ"},"agent_actions":{"view_html":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK","download_json":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK.json","view_paper":"https://pith.science/paper/KL5R7FWJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.01339&json=true","fetch_graph":"https://pith.science/api/pith-number/KL5R7FWJQOGBTEWFGOQHJ74GTK/graph.json","fetch_events":"https://pith.science/api/pith-number/KL5R7FWJQOGBTEWFGOQHJ74GTK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK/action/storage_attestation","attest_author":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK/action/author_attestation","sign_citation":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK/action/citation_signature","submit_replication":"https://pith.science/pith/KL5R7FWJQOGBTEWFGOQHJ74GTK/action/replication_record"}},"created_at":"2026-07-05T08:03:12.833095+00:00","updated_at":"2026-07-05T08:03:12.833095+00:00"}