{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3X7RUZIQGMPRQ27RDCWMUZBYL4","short_pith_number":"pith:3X7RUZIQ","schema_version":"1.0","canonical_sha256":"ddff1a6510331f186bf118acca64385f28dc2a604f909f4eb8e933d61ebc84a8","source":{"kind":"arxiv","id":"2409.17656","version":1},"attestation_state":"computed","paper":{"title":"Prototype based Masked Audio Model for Self-Supervised Learning of Sound Event Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ian McLoughlin, Nan Jiang, Pengfei Cai, Qing Gu, Yan Song","submitted_at":"2024-09-26T09:07:20Z","abstract_excerpt":"A significant challenge in sound event detection (SED) is the effective utilization of unlabeled data, given the limited availability of labeled data due to high annotation costs. Semi-supervised algorithms rely on labeled data to learn from unlabeled data, and the performance is constrained by the quality and size of the former. In this paper, we introduce the Prototype based Masked Audio Model~(PMAM) algorithm for self-supervised representation learning in SED, to better exploit unlabeled data. Specifically, semantically rich frame-level pseudo labels are constructed from a Gaussian mixture "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.17656","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-09-26T09:07:20Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"134dbbd83d460c4482e86bde12eed2c630dc5154ff70452d768a84132f52c787","abstract_canon_sha256":"25de4ca910b88da92fe0648e1a5955a75d35f2ea6487fa8341cb0a71c10b745f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:13.395337Z","signature_b64":"zAB7fg5lGjQxtkRCnCm1sYOHhRU23DwS4e86ZD+ytS/k2uLZC3DIHHkob1WQy8O8YCzGPEllPZygFnPr+cwvCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddff1a6510331f186bf118acca64385f28dc2a604f909f4eb8e933d61ebc84a8","last_reissued_at":"2026-07-05T09:12:13.394925Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:13.394925Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prototype based Masked Audio Model for Self-Supervised Learning of Sound Event Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ian McLoughlin, Nan Jiang, Pengfei Cai, Qing Gu, Yan Song","submitted_at":"2024-09-26T09:07:20Z","abstract_excerpt":"A significant challenge in sound event detection (SED) is the effective utilization of unlabeled data, given the limited availability of labeled data due to high annotation costs. Semi-supervised algorithms rely on labeled data to learn from unlabeled data, and the performance is constrained by the quality and size of the former. In this paper, we introduce the Prototype based Masked Audio Model~(PMAM) algorithm for self-supervised representation learning in SED, to better exploit unlabeled data. Specifically, semantically rich frame-level pseudo labels are constructed from a Gaussian mixture "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.17656","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.17656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.17656","created_at":"2026-07-05T09:12:13.394982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.17656v1","created_at":"2026-07-05T09:12:13.394982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.17656","created_at":"2026-07-05T09:12:13.394982+00:00"},{"alias_kind":"pith_short_12","alias_value":"3X7RUZIQGMPR","created_at":"2026-07-05T09:12:13.394982+00:00"},{"alias_kind":"pith_short_16","alias_value":"3X7RUZIQGMPRQ27R","created_at":"2026-07-05T09:12:13.394982+00:00"},{"alias_kind":"pith_short_8","alias_value":"3X7RUZIQ","created_at":"2026-07-05T09:12:13.394982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.12785","citing_title":"Frequency Dynamic Convolutions for Sound Event Detection","ref_index":86,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4","json":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4.json","graph_json":"https://pith.science/api/pith-number/3X7RUZIQGMPRQ27RDCWMUZBYL4/graph.json","events_json":"https://pith.science/api/pith-number/3X7RUZIQGMPRQ27RDCWMUZBYL4/events.json","paper":"https://pith.science/paper/3X7RUZIQ"},"agent_actions":{"view_html":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4","download_json":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4.json","view_paper":"https://pith.science/paper/3X7RUZIQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.17656&json=true","fetch_graph":"https://pith.science/api/pith-number/3X7RUZIQGMPRQ27RDCWMUZBYL4/graph.json","fetch_events":"https://pith.science/api/pith-number/3X7RUZIQGMPRQ27RDCWMUZBYL4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4/action/storage_attestation","attest_author":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4/action/author_attestation","sign_citation":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4/action/citation_signature","submit_replication":"https://pith.science/pith/3X7RUZIQGMPRQ27RDCWMUZBYL4/action/replication_record"}},"created_at":"2026-07-05T09:12:13.394982+00:00","updated_at":"2026-07-05T09:12:13.394982+00:00"}