{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZTANHVUVJVLRLEXPLDCOZ6O22X","short_pith_number":"pith:ZTANHVUV","schema_version":"1.0","canonical_sha256":"ccc0d3d6954d571592ef58c4ecf9dad5d7e8b1dde77d557082103cf58a09b0e0","source":{"kind":"arxiv","id":"2312.09736","version":2},"attestation_state":"computed","paper":{"title":"HEAR: Hearing Enhanced Audio Response for Video-grounded Dialogue","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chnag D. Yoo, Dahyun Kim, Eunseop Yoon, Hee Suk Yoon, Junyeong Kim, Sunjae Yoon","submitted_at":"2023-12-15T12:20:24Z","abstract_excerpt":"Video-grounded Dialogue (VGD) aims to answer questions regarding a given multi-modal input comprising video, audio, and dialogue history. Although there have been numerous efforts in developing VGD systems to improve the quality of their responses, existing systems are competent only to incorporate the information in the video and text and tend to struggle in extracting the necessary information from the audio when generating appropriate responses to the question. The VGD system seems to be deaf, and thus, we coin this symptom of current systems' ignoring audio data as a deaf response. To over"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.09736","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-15T12:20:24Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"02c5ed90b8a221331333751e77d43a1f3631a3cc8535b8ac29c55944063e6a3e","abstract_canon_sha256":"acc15615fcfe9a2098a54ff42ba1062e47417a7748a384c164c42a7d1a2c4726"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:14.075432Z","signature_b64":"Jo+GnYPCIFZq91pPsJJhejtu8ACLkwWEwsyXOMpV5YMg6tpcxDVIWZG4k36I/TbefGpxXMeP4UTPwASpnn9NBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ccc0d3d6954d571592ef58c4ecf9dad5d7e8b1dde77d557082103cf58a09b0e0","last_reissued_at":"2026-07-05T10:48:14.074931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:14.074931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HEAR: Hearing Enhanced Audio Response for Video-grounded Dialogue","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chnag D. Yoo, Dahyun Kim, Eunseop Yoon, Hee Suk Yoon, Junyeong Kim, Sunjae Yoon","submitted_at":"2023-12-15T12:20:24Z","abstract_excerpt":"Video-grounded Dialogue (VGD) aims to answer questions regarding a given multi-modal input comprising video, audio, and dialogue history. Although there have been numerous efforts in developing VGD systems to improve the quality of their responses, existing systems are competent only to incorporate the information in the video and text and tend to struggle in extracting the necessary information from the audio when generating appropriate responses to the question. The VGD system seems to be deaf, and thus, we coin this symptom of current systems' ignoring audio data as a deaf response. To over"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09736","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09736/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.09736","created_at":"2026-07-05T10:48:14.074984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.09736v2","created_at":"2026-07-05T10:48:14.074984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09736","created_at":"2026-07-05T10:48:14.074984+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZTANHVUVJVLR","created_at":"2026-07-05T10:48:14.074984+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZTANHVUVJVLRLEXP","created_at":"2026-07-05T10:48:14.074984+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZTANHVUV","created_at":"2026-07-05T10:48:14.074984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.00398","citing_title":"Occlusion-robust Stylization for Drawing-based 3D Animation","ref_index":62,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X","json":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X.json","graph_json":"https://pith.science/api/pith-number/ZTANHVUVJVLRLEXPLDCOZ6O22X/graph.json","events_json":"https://pith.science/api/pith-number/ZTANHVUVJVLRLEXPLDCOZ6O22X/events.json","paper":"https://pith.science/paper/ZTANHVUV"},"agent_actions":{"view_html":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X","download_json":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X.json","view_paper":"https://pith.science/paper/ZTANHVUV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.09736&json=true","fetch_graph":"https://pith.science/api/pith-number/ZTANHVUVJVLRLEXPLDCOZ6O22X/graph.json","fetch_events":"https://pith.science/api/pith-number/ZTANHVUVJVLRLEXPLDCOZ6O22X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X/action/storage_attestation","attest_author":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X/action/author_attestation","sign_citation":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X/action/citation_signature","submit_replication":"https://pith.science/pith/ZTANHVUVJVLRLEXPLDCOZ6O22X/action/replication_record"}},"created_at":"2026-07-05T10:48:14.074984+00:00","updated_at":"2026-07-05T10:48:14.074984+00:00"}