{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MDSFM5JY3ZMS26DZZBGMG3ZMOF","short_pith_number":"pith:MDSFM5JY","schema_version":"1.0","canonical_sha256":"60e4567538de592d7879c84cc36f2c717c596b6b862d6df887d69ce424fe7c51","source":{"kind":"arxiv","id":"2505.10879","version":2},"attestation_state":"computed","paper":{"title":"Multi-Stage Speaker Diarization for Noisy Classrooms","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ahmed Adel Attia, Ali Sartaz Khan, Dorottya Demszky, Tolulope Ogunremi","submitted_at":"2025-05-16T05:35:06Z","abstract_excerpt":"Speaker diarization, the process of identifying \"who spoke when\" in audio recordings, is essential for understanding classroom dynamics. However, classroom settings present distinct challenges, including poor recording quality, high levels of background noise, overlapping speech, and the difficulty of accurately capturing children's voices. This study investigates the effectiveness of multi-stage diarization models using Nvidia's NeMo diarization pipeline. We assess the impact of denoising on diarization accuracy and compare various voice activity detection (VAD) models, including self-supervi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10879","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2025-05-16T05:35:06Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"dc017b0b631e9022fc65a74ffa69432c747cc18ca645998f17913f4b7b634bc0","abstract_canon_sha256":"ebab04b08e7b713ce8d753e5af7107cfb15d6f6bec73760abba9129a55d633fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:08.091866Z","signature_b64":"5vHNIWOsgQAkQetVVAG6e2QBTEiDt6Dv3cENRf9qCypHOp84QClRIHVV4Hdz59deFKllweSo4Pc/kHz9wUwZAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60e4567538de592d7879c84cc36f2c717c596b6b862d6df887d69ce424fe7c51","last_reissued_at":"2026-07-05T11:10:08.091389Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:08.091389Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Stage Speaker Diarization for Noisy Classrooms","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ahmed Adel Attia, Ali Sartaz Khan, Dorottya Demszky, Tolulope Ogunremi","submitted_at":"2025-05-16T05:35:06Z","abstract_excerpt":"Speaker diarization, the process of identifying \"who spoke when\" in audio recordings, is essential for understanding classroom dynamics. However, classroom settings present distinct challenges, including poor recording quality, high levels of background noise, overlapping speech, and the difficulty of accurately capturing children's voices. This study investigates the effectiveness of multi-stage diarization models using Nvidia's NeMo diarization pipeline. We assess the impact of denoising on diarization accuracy and compare various voice activity detection (VAD) models, including self-supervi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10879","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10879/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10879","created_at":"2026-07-05T11:10:08.091444+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10879v2","created_at":"2026-07-05T11:10:08.091444+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10879","created_at":"2026-07-05T11:10:08.091444+00:00"},{"alias_kind":"pith_short_12","alias_value":"MDSFM5JY3ZMS","created_at":"2026-07-05T11:10:08.091444+00:00"},{"alias_kind":"pith_short_16","alias_value":"MDSFM5JY3ZMS26DZ","created_at":"2026-07-05T11:10:08.091444+00:00"},{"alias_kind":"pith_short_8","alias_value":"MDSFM5JY","created_at":"2026-07-05T11:10:08.091444+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF","json":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF.json","graph_json":"https://pith.science/api/pith-number/MDSFM5JY3ZMS26DZZBGMG3ZMOF/graph.json","events_json":"https://pith.science/api/pith-number/MDSFM5JY3ZMS26DZZBGMG3ZMOF/events.json","paper":"https://pith.science/paper/MDSFM5JY"},"agent_actions":{"view_html":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF","download_json":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF.json","view_paper":"https://pith.science/paper/MDSFM5JY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10879&json=true","fetch_graph":"https://pith.science/api/pith-number/MDSFM5JY3ZMS26DZZBGMG3ZMOF/graph.json","fetch_events":"https://pith.science/api/pith-number/MDSFM5JY3ZMS26DZZBGMG3ZMOF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF/action/storage_attestation","attest_author":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF/action/author_attestation","sign_citation":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF/action/citation_signature","submit_replication":"https://pith.science/pith/MDSFM5JY3ZMS26DZZBGMG3ZMOF/action/replication_record"}},"created_at":"2026-07-05T11:10:08.091444+00:00","updated_at":"2026-07-05T11:10:08.091444+00:00"}