{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:TDZHSLOENPQQAM35QWGYZ55EFV","short_pith_number":"pith:TDZHSLOE","schema_version":"1.0","canonical_sha256":"98f2792dc46be100337d858d8cf7a42d5661be3243102d15a6259bec7952006e","source":{"kind":"arxiv","id":"2111.00009","version":2},"attestation_state":"computed","paper":{"title":"Revisiting joint decoding based multi-talker speech recognition with DNN acoustic model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jan \\v{C}ernock\\'y, J\\'an \\v{S}vec, Kate\\v{r}ina \\v{Z}mol\\'ikov\\'a, Lucas Ondel, Luk\\'a\\v{s} Burget, Marc Delcroix, Martin Kocour, Tsubasa Ochiai","submitted_at":"2021-10-31T09:28:04Z","abstract_excerpt":"In typical multi-talker speech recognition systems, a neural network-based acoustic model predicts senone state posteriors for each speaker. These are later used by a single-talker decoder which is applied on each speaker-specific output stream separately. In this work, we argue that such a scheme is sub-optimal and propose a principled solution that decodes all speakers jointly. We modify the acoustic model to predict joint state posteriors for all speakers, enabling the network to express uncertainty about the attribution of parts of the speech signal to the speakers. We employ a joint decod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.00009","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2021-10-31T09:28:04Z","cross_cats_sorted":["cs.LG","cs.SD"],"title_canon_sha256":"8457349d7d7c3f3830b10494d07bfb25a888691a500ca3695c7f22a4d79a8f12","abstract_canon_sha256":"26c63aeb5c8a1179ddc1c170278f40a1848f3b671b380ee06ea76ab56229f962"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:14:59.889924Z","signature_b64":"t6TSnXipc1PLbCNlYkY4KNaxmuRBXGNWClsdGzC59kBBLR7W1ur5FgJsma5mwlf8YDTlqE+sKiNOehF4lWWuAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98f2792dc46be100337d858d8cf7a42d5661be3243102d15a6259bec7952006e","last_reissued_at":"2026-07-05T04:14:59.889446Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:14:59.889446Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting joint decoding based multi-talker speech recognition with DNN acoustic model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jan \\v{C}ernock\\'y, J\\'an \\v{S}vec, Kate\\v{r}ina \\v{Z}mol\\'ikov\\'a, Lucas Ondel, Luk\\'a\\v{s} Burget, Marc Delcroix, Martin Kocour, Tsubasa Ochiai","submitted_at":"2021-10-31T09:28:04Z","abstract_excerpt":"In typical multi-talker speech recognition systems, a neural network-based acoustic model predicts senone state posteriors for each speaker. These are later used by a single-talker decoder which is applied on each speaker-specific output stream separately. In this work, we argue that such a scheme is sub-optimal and propose a principled solution that decodes all speakers jointly. We modify the acoustic model to predict joint state posteriors for all speakers, enabling the network to express uncertainty about the attribution of parts of the speech signal to the speakers. We employ a joint decod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.00009","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.00009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.00009","created_at":"2026-07-05T04:14:59.889503+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.00009v2","created_at":"2026-07-05T04:14:59.889503+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.00009","created_at":"2026-07-05T04:14:59.889503+00:00"},{"alias_kind":"pith_short_12","alias_value":"TDZHSLOENPQQ","created_at":"2026-07-05T04:14:59.889503+00:00"},{"alias_kind":"pith_short_16","alias_value":"TDZHSLOENPQQAM35","created_at":"2026-07-05T04:14:59.889503+00:00"},{"alias_kind":"pith_short_8","alias_value":"TDZHSLOE","created_at":"2026-07-05T04:14:59.889503+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV","json":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV.json","graph_json":"https://pith.science/api/pith-number/TDZHSLOENPQQAM35QWGYZ55EFV/graph.json","events_json":"https://pith.science/api/pith-number/TDZHSLOENPQQAM35QWGYZ55EFV/events.json","paper":"https://pith.science/paper/TDZHSLOE"},"agent_actions":{"view_html":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV","download_json":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV.json","view_paper":"https://pith.science/paper/TDZHSLOE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.00009&json=true","fetch_graph":"https://pith.science/api/pith-number/TDZHSLOENPQQAM35QWGYZ55EFV/graph.json","fetch_events":"https://pith.science/api/pith-number/TDZHSLOENPQQAM35QWGYZ55EFV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV/action/storage_attestation","attest_author":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV/action/author_attestation","sign_citation":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV/action/citation_signature","submit_replication":"https://pith.science/pith/TDZHSLOENPQQAM35QWGYZ55EFV/action/replication_record"}},"created_at":"2026-07-05T04:14:59.889503+00:00","updated_at":"2026-07-05T04:14:59.889503+00:00"}