{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IXRD2FHLXKS23CGHGATGREL7LZ","short_pith_number":"pith:IXRD2FHL","schema_version":"1.0","canonical_sha256":"45e23d14ebbaa5ad88c7302668917f5e7cdc46ee91a577972faefc2ae6ab2ff5","source":{"kind":"arxiv","id":"2303.05397","version":2},"attestation_state":"computed","paper":{"title":"TOLD: A Novel Two-Stage Overlap-Aware Framework for Speaker Diarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jiaming Wang, Shiliang Zhang, Zhihao Du","submitted_at":"2023-03-08T05:05:26Z","abstract_excerpt":"Recently, end-to-end neural diarization (EEND) is introduced and achieves promising results in speaker-overlapped scenarios. In EEND, speaker diarization is formulated as a multi-label prediction problem, where speaker activities are estimated independently and their dependency are not well considered. To overcome these disadvantages, we employ the power set encoding to reformulate speaker diarization as a single-label classification problem and propose the overlap-aware EEND (EEND-OLA) model, in which speaker overlaps and dependency can be modeled explicitly. Inspired by the success of two-st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.05397","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-03-08T05:05:26Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"158ee4f8601fe77093b239c9d75f366ef0b3ceee8cda4a0c35e8d0b3e9882352","abstract_canon_sha256":"83d897c2778d3d275acacce79923b657d7abfc484563b09678f8659d553dcd9c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:23:24.736658Z","signature_b64":"uOKXvfye0juVpz6z0k5Vb/OeuRKu/v3UCH6NzIqfYiBNgySZOZoDJp+TmTFjTD8W6l8C2PSf0TPIvsirJpmWAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45e23d14ebbaa5ad88c7302668917f5e7cdc46ee91a577972faefc2ae6ab2ff5","last_reissued_at":"2026-07-05T07:23:24.736225Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:23:24.736225Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TOLD: A Novel Two-Stage Overlap-Aware Framework for Speaker Diarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jiaming Wang, Shiliang Zhang, Zhihao Du","submitted_at":"2023-03-08T05:05:26Z","abstract_excerpt":"Recently, end-to-end neural diarization (EEND) is introduced and achieves promising results in speaker-overlapped scenarios. In EEND, speaker diarization is formulated as a multi-label prediction problem, where speaker activities are estimated independently and their dependency are not well considered. To overcome these disadvantages, we employ the power set encoding to reformulate speaker diarization as a single-label classification problem and propose the overlap-aware EEND (EEND-OLA) model, in which speaker overlaps and dependency can be modeled explicitly. Inspired by the success of two-st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.05397","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.05397/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.05397","created_at":"2026-07-05T07:23:24.736285+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.05397v2","created_at":"2026-07-05T07:23:24.736285+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.05397","created_at":"2026-07-05T07:23:24.736285+00:00"},{"alias_kind":"pith_short_12","alias_value":"IXRD2FHLXKS2","created_at":"2026-07-05T07:23:24.736285+00:00"},{"alias_kind":"pith_short_16","alias_value":"IXRD2FHLXKS23CGH","created_at":"2026-07-05T07:23:24.736285+00:00"},{"alias_kind":"pith_short_8","alias_value":"IXRD2FHL","created_at":"2026-07-05T07:23:24.736285+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.11344","citing_title":"Do We Still Need Audio? Rethinking Speaker Diarization with a Text-Based Approach Using Multiple Prediction Models","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ","json":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ.json","graph_json":"https://pith.science/api/pith-number/IXRD2FHLXKS23CGHGATGREL7LZ/graph.json","events_json":"https://pith.science/api/pith-number/IXRD2FHLXKS23CGHGATGREL7LZ/events.json","paper":"https://pith.science/paper/IXRD2FHL"},"agent_actions":{"view_html":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ","download_json":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ.json","view_paper":"https://pith.science/paper/IXRD2FHL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.05397&json=true","fetch_graph":"https://pith.science/api/pith-number/IXRD2FHLXKS23CGHGATGREL7LZ/graph.json","fetch_events":"https://pith.science/api/pith-number/IXRD2FHLXKS23CGHGATGREL7LZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ/action/storage_attestation","attest_author":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ/action/author_attestation","sign_citation":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ/action/citation_signature","submit_replication":"https://pith.science/pith/IXRD2FHLXKS23CGHGATGREL7LZ/action/replication_record"}},"created_at":"2026-07-05T07:23:24.736285+00:00","updated_at":"2026-07-05T07:23:24.736285+00:00"}