{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AELKPUHXLFVXSJDMHMZSMZECJE","short_pith_number":"pith:AELKPUHX","schema_version":"1.0","canonical_sha256":"0116a7d0f7596b79246c3b33266482493255544be3ad42ab88e218de769fffcb","source":{"kind":"arxiv","id":"2303.08342","version":2},"attestation_state":"computed","paper":{"title":"Autonomous Soundscape Augmentation with Multimodal Fusion of Visual and Participant-linked Inputs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Bhan Lam, Karn N. Watcharasupat, Kenneth Ooi, Woon-Seng Gan, Zhen-Ting Ong","submitted_at":"2023-03-15T03:17:31Z","abstract_excerpt":"Autonomous soundscape augmentation systems typically use trained models to pick optimal maskers to effect a desired perceptual change. While acoustic information is paramount to such systems, contextual information, including participant demographics and the visual environment, also influences acoustic perception. Hence, we propose modular modifications to an existing attention-based deep neural network, to allow early, mid-level, and late feature fusion of participant-linked, visual, and acoustic features. Ablation studies on module configurations and corresponding fusion methods using the AR"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.08342","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2023-03-15T03:17:31Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"92a106ce16032722602dd8d43f697ba3bad4981bbd79a6822da9502f146d7c45","abstract_canon_sha256":"772b31e853a508dd99c96099b163cf186c5649b73b1d93685744c62e49e34269"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:51.416017Z","signature_b64":"toIjhDGGAkHNmEOv/ipn6iJbKPONOWT/OFhVs8ciXYgBaMLwfL+qFn74Pe0muV2Y7gBl52MxZ0nsOErM2lA4Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0116a7d0f7596b79246c3b33266482493255544be3ad42ab88e218de769fffcb","last_reissued_at":"2026-07-05T08:38:51.415573Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:51.415573Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Autonomous Soundscape Augmentation with Multimodal Fusion of Visual and Participant-linked Inputs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Bhan Lam, Karn N. Watcharasupat, Kenneth Ooi, Woon-Seng Gan, Zhen-Ting Ong","submitted_at":"2023-03-15T03:17:31Z","abstract_excerpt":"Autonomous soundscape augmentation systems typically use trained models to pick optimal maskers to effect a desired perceptual change. While acoustic information is paramount to such systems, contextual information, including participant demographics and the visual environment, also influences acoustic perception. Hence, we propose modular modifications to an existing attention-based deep neural network, to allow early, mid-level, and late feature fusion of participant-linked, visual, and acoustic features. Ablation studies on module configurations and corresponding fusion methods using the AR"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.08342","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.08342/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.08342","created_at":"2026-07-05T08:38:51.415636+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.08342v2","created_at":"2026-07-05T08:38:51.415636+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.08342","created_at":"2026-07-05T08:38:51.415636+00:00"},{"alias_kind":"pith_short_12","alias_value":"AELKPUHXLFVX","created_at":"2026-07-05T08:38:51.415636+00:00"},{"alias_kind":"pith_short_16","alias_value":"AELKPUHXLFVXSJDM","created_at":"2026-07-05T08:38:51.415636+00:00"},{"alias_kind":"pith_short_8","alias_value":"AELKPUHX","created_at":"2026-07-05T08:38:51.415636+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE","json":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE.json","graph_json":"https://pith.science/api/pith-number/AELKPUHXLFVXSJDMHMZSMZECJE/graph.json","events_json":"https://pith.science/api/pith-number/AELKPUHXLFVXSJDMHMZSMZECJE/events.json","paper":"https://pith.science/paper/AELKPUHX"},"agent_actions":{"view_html":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE","download_json":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE.json","view_paper":"https://pith.science/paper/AELKPUHX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.08342&json=true","fetch_graph":"https://pith.science/api/pith-number/AELKPUHXLFVXSJDMHMZSMZECJE/graph.json","fetch_events":"https://pith.science/api/pith-number/AELKPUHXLFVXSJDMHMZSMZECJE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE/action/storage_attestation","attest_author":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE/action/author_attestation","sign_citation":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE/action/citation_signature","submit_replication":"https://pith.science/pith/AELKPUHXLFVXSJDMHMZSMZECJE/action/replication_record"}},"created_at":"2026-07-05T08:38:51.415636+00:00","updated_at":"2026-07-05T08:38:51.415636+00:00"}