{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GGJM3ODA2S55E6XJVE5P4635CI","short_pith_number":"pith:GGJM3ODA","schema_version":"1.0","canonical_sha256":"3192cdb860d4bbd27ae9a93afe7b7d123b0a54236ac15dce8c55d8f9616fdaad","source":{"kind":"arxiv","id":"2306.12559","version":1},"attestation_state":"computed","paper":{"title":"Exploring the Role of Audio in Video Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Ehsan Elhamifar, Haichao Yu, Heng Wang, Linjie Yang, Longyin Wen, Yuhan Shen","submitted_at":"2023-06-21T20:54:52Z","abstract_excerpt":"Recent focus in video captioning has been on designing architectures that can consume both video and text modalities, and using large-scale video datasets with text transcripts for pre-training, such as HowTo100M. Though these approaches have achieved significant improvement, the audio modality is often ignored in video captioning. In this work, we present an audio-visual framework, which aims to fully exploit the potential of the audio modality for captioning. Instead of relying on text transcripts extracted via automatic speech recognition (ASR), we argue that learning with raw audio signals"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.12559","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-06-21T20:54:52Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"93db4d7f1b4d3c9ee4c86593938545bcf9e0f06ce9d7f7b0476c16c10c486986","abstract_canon_sha256":"688ce24ed0bdfc3050d49246f8ed544ebe9260a39a5049b477af5598e7e3cbc4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:23:39.896895Z","signature_b64":"X9dT3Guiork+/9OuJ9/YTi3yIZ3hBOo6M9oOJE72m6ZUSLxIfdx2vSOR/actByxp9BliP/Jd9YAZnxE+sLcgAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3192cdb860d4bbd27ae9a93afe7b7d123b0a54236ac15dce8c55d8f9616fdaad","last_reissued_at":"2026-07-05T06:23:39.896523Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:23:39.896523Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the Role of Audio in Video Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Ehsan Elhamifar, Haichao Yu, Heng Wang, Linjie Yang, Longyin Wen, Yuhan Shen","submitted_at":"2023-06-21T20:54:52Z","abstract_excerpt":"Recent focus in video captioning has been on designing architectures that can consume both video and text modalities, and using large-scale video datasets with text transcripts for pre-training, such as HowTo100M. Though these approaches have achieved significant improvement, the audio modality is often ignored in video captioning. In this work, we present an audio-visual framework, which aims to fully exploit the potential of the audio modality for captioning. Instead of relying on text transcripts extracted via automatic speech recognition (ASR), we argue that learning with raw audio signals"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.12559","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.12559/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.12559","created_at":"2026-07-05T06:23:39.896592+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.12559v1","created_at":"2026-07-05T06:23:39.896592+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.12559","created_at":"2026-07-05T06:23:39.896592+00:00"},{"alias_kind":"pith_short_12","alias_value":"GGJM3ODA2S55","created_at":"2026-07-05T06:23:39.896592+00:00"},{"alias_kind":"pith_short_16","alias_value":"GGJM3ODA2S55E6XJ","created_at":"2026-07-05T06:23:39.896592+00:00"},{"alias_kind":"pith_short_8","alias_value":"GGJM3ODA","created_at":"2026-07-05T06:23:39.896592+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI","json":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI.json","graph_json":"https://pith.science/api/pith-number/GGJM3ODA2S55E6XJVE5P4635CI/graph.json","events_json":"https://pith.science/api/pith-number/GGJM3ODA2S55E6XJVE5P4635CI/events.json","paper":"https://pith.science/paper/GGJM3ODA"},"agent_actions":{"view_html":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI","download_json":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI.json","view_paper":"https://pith.science/paper/GGJM3ODA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.12559&json=true","fetch_graph":"https://pith.science/api/pith-number/GGJM3ODA2S55E6XJVE5P4635CI/graph.json","fetch_events":"https://pith.science/api/pith-number/GGJM3ODA2S55E6XJVE5P4635CI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI/action/storage_attestation","attest_author":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI/action/author_attestation","sign_citation":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI/action/citation_signature","submit_replication":"https://pith.science/pith/GGJM3ODA2S55E6XJVE5P4635CI/action/replication_record"}},"created_at":"2026-07-05T06:23:39.896592+00:00","updated_at":"2026-07-05T06:23:39.896592+00:00"}