{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4KFIGPLCN35EGGHGSJM6P72VMN","short_pith_number":"pith:4KFIGPLC","schema_version":"1.0","canonical_sha256":"e28a833d626efa4318e69259e7ff55637174661a0db37eb5014a567ac8ad93ce","source":{"kind":"arxiv","id":"2412.10002","version":1},"attestation_state":"computed","paper":{"title":"NowYouSee Me: Context-Aware Automatic Audio Description","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"David Fan, Jue Wang, Linda Liu, Seon-Ho Lee, Vimal Bhat, Xiang Hao, Xinyu Li, Zhikang Zhang","submitted_at":"2024-12-13T09:40:37Z","abstract_excerpt":"Audio Description (AD) plays a pivotal role as an application system aimed at guaranteeing accessibility in multimedia content, which provides additional narrations at suitable intervals to describe visual elements, catering specifically to the needs of visually impaired audiences. In this paper, we introduce $\\mathrm{CA^3D}$, the pioneering unified Context-Aware Automatic Audio Description system that provides AD event scripts with precise locations in the long cinematic content. Specifically, $\\mathrm{CA^3D}$ system consists of: 1) a Temporal Feature Enhancement Module to efficiently capture"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.10002","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-13T09:40:37Z","cross_cats_sorted":[],"title_canon_sha256":"05438a321d977504e148585e42bb59f935ccd9e85f57036b031715fe24d964a1","abstract_canon_sha256":"a6c761e1d6cd4d7c7c90b0507cc34ced3f10a78da3d558fca2e98a6ccd97136e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:45.361029Z","signature_b64":"fK+l6gDU+Py3J361Wnr32S/WSH/ZnnTijlLgaZ3xH+y5Ga151r0SvjWzsCzTSsOedj5U+ub2GRYqxFO0L9eEDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e28a833d626efa4318e69259e7ff55637174661a0db37eb5014a567ac8ad93ce","last_reissued_at":"2026-07-05T09:48:45.360597Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:45.360597Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NowYouSee Me: Context-Aware Automatic Audio Description","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"David Fan, Jue Wang, Linda Liu, Seon-Ho Lee, Vimal Bhat, Xiang Hao, Xinyu Li, Zhikang Zhang","submitted_at":"2024-12-13T09:40:37Z","abstract_excerpt":"Audio Description (AD) plays a pivotal role as an application system aimed at guaranteeing accessibility in multimedia content, which provides additional narrations at suitable intervals to describe visual elements, catering specifically to the needs of visually impaired audiences. In this paper, we introduce $\\mathrm{CA^3D}$, the pioneering unified Context-Aware Automatic Audio Description system that provides AD event scripts with precise locations in the long cinematic content. Specifically, $\\mathrm{CA^3D}$ system consists of: 1) a Temporal Feature Enhancement Module to efficiently capture"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.10002","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.10002/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.10002","created_at":"2026-07-05T09:48:45.360652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.10002v1","created_at":"2026-07-05T09:48:45.360652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.10002","created_at":"2026-07-05T09:48:45.360652+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KFIGPLCN35E","created_at":"2026-07-05T09:48:45.360652+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KFIGPLCN35EGGHG","created_at":"2026-07-05T09:48:45.360652+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KFIGPLC","created_at":"2026-07-05T09:48:45.360652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.22045","citing_title":"Mitigating Audiovisual Mismatch in Visual-Guide Audio Captioning","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN","json":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN.json","graph_json":"https://pith.science/api/pith-number/4KFIGPLCN35EGGHGSJM6P72VMN/graph.json","events_json":"https://pith.science/api/pith-number/4KFIGPLCN35EGGHGSJM6P72VMN/events.json","paper":"https://pith.science/paper/4KFIGPLC"},"agent_actions":{"view_html":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN","download_json":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN.json","view_paper":"https://pith.science/paper/4KFIGPLC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.10002&json=true","fetch_graph":"https://pith.science/api/pith-number/4KFIGPLCN35EGGHGSJM6P72VMN/graph.json","fetch_events":"https://pith.science/api/pith-number/4KFIGPLCN35EGGHGSJM6P72VMN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN/action/storage_attestation","attest_author":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN/action/author_attestation","sign_citation":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN/action/citation_signature","submit_replication":"https://pith.science/pith/4KFIGPLCN35EGGHGSJM6P72VMN/action/replication_record"}},"created_at":"2026-07-05T09:48:45.360652+00:00","updated_at":"2026-07-05T09:48:45.360652+00:00"}