{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:TS6E2JDMW466ZOIUJOKACTNR7U","short_pith_number":"pith:TS6E2JDM","schema_version":"1.0","canonical_sha256":"9cbc4d246cb73decb9144b94014db1fd2a5f9838656325d048b997428bb446a4","source":{"kind":"arxiv","id":"2202.01405","version":1},"attestation_state":"computed","paper":{"title":"Joint Speech Recognition and Audio Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chaitanya Narisetty, Emiru Tsunoo, Michael Hentschel, Shinji Watanabe, Xuankai Chang, Yosuke Kashiwagi","submitted_at":"2022-02-03T04:42:43Z","abstract_excerpt":"Speech samples recorded in both indoor and outdoor environments are often contaminated with secondary audio sources. Most end-to-end monaural speech recognition systems either remove these background sounds using speech enhancement or train noise-robust models. For better model interpretability and holistic understanding, we aim to bring together the growing field of automated audio captioning (AAC) and the thoroughly studied automatic speech recognition (ASR). The goal of AAC is to generate natural language descriptions of contents in audio samples. We propose several approaches for end-to-en"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.01405","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2022-02-03T04:42:43Z","cross_cats_sorted":["cs.CL","cs.SD"],"title_canon_sha256":"3da5d1d71d9d33524ad85624fc36647f006fc142d7812ad136514dadd420d5af","abstract_canon_sha256":"a42e27fb4c97218992a1a7563b00a57adbf8a775dc8b76e77d78e72366bb81b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:53:53.939062Z","signature_b64":"1h+Rg1tGuk77HJbQKRwPZDuVv3WBTre+THTJQMogKmAmt6o4oxzYWQGYDwKQOe+TZoAzmcAc/ZutWun5tSrbAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9cbc4d246cb73decb9144b94014db1fd2a5f9838656325d048b997428bb446a4","last_reissued_at":"2026-07-05T03:53:53.938553Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:53:53.938553Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Joint Speech Recognition and Audio Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chaitanya Narisetty, Emiru Tsunoo, Michael Hentschel, Shinji Watanabe, Xuankai Chang, Yosuke Kashiwagi","submitted_at":"2022-02-03T04:42:43Z","abstract_excerpt":"Speech samples recorded in both indoor and outdoor environments are often contaminated with secondary audio sources. Most end-to-end monaural speech recognition systems either remove these background sounds using speech enhancement or train noise-robust models. For better model interpretability and holistic understanding, we aim to bring together the growing field of automated audio captioning (AAC) and the thoroughly studied automatic speech recognition (ASR). The goal of AAC is to generate natural language descriptions of contents in audio samples. We propose several approaches for end-to-en"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.01405","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.01405/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.01405","created_at":"2026-07-05T03:53:53.938615+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.01405v1","created_at":"2026-07-05T03:53:53.938615+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.01405","created_at":"2026-07-05T03:53:53.938615+00:00"},{"alias_kind":"pith_short_12","alias_value":"TS6E2JDMW466","created_at":"2026-07-05T03:53:53.938615+00:00"},{"alias_kind":"pith_short_16","alias_value":"TS6E2JDMW466ZOIU","created_at":"2026-07-05T03:53:53.938615+00:00"},{"alias_kind":"pith_short_8","alias_value":"TS6E2JDM","created_at":"2026-07-05T03:53:53.938615+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U","json":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U.json","graph_json":"https://pith.science/api/pith-number/TS6E2JDMW466ZOIUJOKACTNR7U/graph.json","events_json":"https://pith.science/api/pith-number/TS6E2JDMW466ZOIUJOKACTNR7U/events.json","paper":"https://pith.science/paper/TS6E2JDM"},"agent_actions":{"view_html":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U","download_json":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U.json","view_paper":"https://pith.science/paper/TS6E2JDM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.01405&json=true","fetch_graph":"https://pith.science/api/pith-number/TS6E2JDMW466ZOIUJOKACTNR7U/graph.json","fetch_events":"https://pith.science/api/pith-number/TS6E2JDMW466ZOIUJOKACTNR7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U/action/storage_attestation","attest_author":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U/action/author_attestation","sign_citation":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U/action/citation_signature","submit_replication":"https://pith.science/pith/TS6E2JDMW466ZOIUJOKACTNR7U/action/replication_record"}},"created_at":"2026-07-05T03:53:53.938615+00:00","updated_at":"2026-07-05T03:53:53.938615+00:00"}