{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F73C4KKMUWAS62HH5LYZQWSTUK","short_pith_number":"pith:F73C4KKM","schema_version":"1.0","canonical_sha256":"2ff62e294ca5812f68e7eaf1985a53a2a991fbac045e16fcb7b72085ac1b4560","source":{"kind":"arxiv","id":"2503.22088","version":2},"attestation_state":"computed","paper":{"title":"Baseline Systems and Evaluation Metrics for Spatial Semantic Segmentation of Sound Scenes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Binh Thien Nguyen, Daiki Takeuchi, Daisuke Niizumi, Masahiro Yasuda, Noboru Harada, Yasunori Ohishi","submitted_at":"2025-03-28T02:08:58Z","abstract_excerpt":"Immersive communication has made significant advancements, especially with the release of the codec for Immersive Voice and Audio Services. Aiming at its further realization, the DCASE 2025 Challenge has recently introduced a task for spatial semantic segmentation of sound scenes (S5), which focuses on detecting and separating sound events in spatial sound scenes. In this paper, we explore methods for addressing the S5 task. Specifically, we present baseline S5 systems that combine audio tagging (AT) and label-queried source separation (LSS) models. We investigate two LSS approaches based on t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.22088","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2025-03-28T02:08:58Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"d3ec6cf598e09c37a17053ffd178b00a42208bb4133713e9d070b3c9fe62109a","abstract_canon_sha256":"b7b57d9afeedb09749d0c63bd8c1e193aa0be7c279a29876dbfc45e7e18a6875"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:19.903804Z","signature_b64":"wNS/ncW7V756q0pANOCYA4m0Ho2ph9kXnVLbOBQ7cl9aweO5Ek81ySPPNIShCkpwba6TOelLLgv1tnVJk/0uBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ff62e294ca5812f68e7eaf1985a53a2a991fbac045e16fcb7b72085ac1b4560","last_reissued_at":"2026-07-05T11:18:19.903399Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:19.903399Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Baseline Systems and Evaluation Metrics for Spatial Semantic Segmentation of Sound Scenes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Binh Thien Nguyen, Daiki Takeuchi, Daisuke Niizumi, Masahiro Yasuda, Noboru Harada, Yasunori Ohishi","submitted_at":"2025-03-28T02:08:58Z","abstract_excerpt":"Immersive communication has made significant advancements, especially with the release of the codec for Immersive Voice and Audio Services. Aiming at its further realization, the DCASE 2025 Challenge has recently introduced a task for spatial semantic segmentation of sound scenes (S5), which focuses on detecting and separating sound events in spatial sound scenes. In this paper, we explore methods for addressing the S5 task. Specifically, we present baseline S5 systems that combine audio tagging (AT) and label-queried source separation (LSS) models. We investigate two LSS approaches based on t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.22088","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.22088/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.22088","created_at":"2026-07-05T11:18:19.903459+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.22088v2","created_at":"2026-07-05T11:18:19.903459+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.22088","created_at":"2026-07-05T11:18:19.903459+00:00"},{"alias_kind":"pith_short_12","alias_value":"F73C4KKMUWAS","created_at":"2026-07-05T11:18:19.903459+00:00"},{"alias_kind":"pith_short_16","alias_value":"F73C4KKMUWAS62HH","created_at":"2026-07-05T11:18:19.903459+00:00"},{"alias_kind":"pith_short_8","alias_value":"F73C4KKM","created_at":"2026-07-05T11:18:19.903459+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.17297","citing_title":"On Temporal Guidance and Iterative Refinement in Audio Source Separation","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK","json":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK.json","graph_json":"https://pith.science/api/pith-number/F73C4KKMUWAS62HH5LYZQWSTUK/graph.json","events_json":"https://pith.science/api/pith-number/F73C4KKMUWAS62HH5LYZQWSTUK/events.json","paper":"https://pith.science/paper/F73C4KKM"},"agent_actions":{"view_html":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK","download_json":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK.json","view_paper":"https://pith.science/paper/F73C4KKM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.22088&json=true","fetch_graph":"https://pith.science/api/pith-number/F73C4KKMUWAS62HH5LYZQWSTUK/graph.json","fetch_events":"https://pith.science/api/pith-number/F73C4KKMUWAS62HH5LYZQWSTUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK/action/storage_attestation","attest_author":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK/action/author_attestation","sign_citation":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK/action/citation_signature","submit_replication":"https://pith.science/pith/F73C4KKMUWAS62HH5LYZQWSTUK/action/replication_record"}},"created_at":"2026-07-05T11:18:19.903459+00:00","updated_at":"2026-07-05T11:18:19.903459+00:00"}