{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QXDAPKGI2YVAFGUXPFKZT5OBSM","short_pith_number":"pith:QXDAPKGI","schema_version":"1.0","canonical_sha256":"85c607a8c8d62a029a97795599f5c1932ebe6f844d3dcd5bb9ef254268744779","source":{"kind":"arxiv","id":"2507.20169","version":1},"attestation_state":"computed","paper":{"title":"Self-Improvement for Audio Large Language Model using Unlabeled Speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Shaowen Wang, Xinyuan Chen, Yao Xu","submitted_at":"2025-07-27T08:16:23Z","abstract_excerpt":"Recent audio LLMs have emerged rapidly, demonstrating strong generalization across various speech tasks. However, given the inherent complexity of speech signals, these models inevitably suffer from performance degradation in specific target domains. To address this, we focus on enhancing audio LLMs in target domains without any labeled data. We propose a self-improvement method called SI-SDA, leveraging the information embedded in large-model decoding to evaluate the quality of generated pseudo labels and then perform domain adaptation based on reinforcement learning optimization. Experimenta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.20169","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2025-07-27T08:16:23Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"18097fa128e49c111891afbeb1ff03630a775e06a8c4d273fd1028273e73b9e2","abstract_canon_sha256":"3e8db650fd96a5938d4490d52d5190e781e33ced2c24f7b5aeadee139fe2c86c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:14.578485Z","signature_b64":"QajPA5n5sS+6eOJE3SUBGhy1UbzYBzB13gZwf7TyVsIiiLHRH3liHin2mp0SrUQAp9dxhl0vssQih5ypfCxtBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85c607a8c8d62a029a97795599f5c1932ebe6f844d3dcd5bb9ef254268744779","last_reissued_at":"2026-07-05T11:44:14.578056Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:14.578056Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Improvement for Audio Large Language Model using Unlabeled Speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Shaowen Wang, Xinyuan Chen, Yao Xu","submitted_at":"2025-07-27T08:16:23Z","abstract_excerpt":"Recent audio LLMs have emerged rapidly, demonstrating strong generalization across various speech tasks. However, given the inherent complexity of speech signals, these models inevitably suffer from performance degradation in specific target domains. To address this, we focus on enhancing audio LLMs in target domains without any labeled data. We propose a self-improvement method called SI-SDA, leveraging the information embedded in large-model decoding to evaluate the quality of generated pseudo labels and then perform domain adaptation based on reinforcement learning optimization. Experimenta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.20169","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.20169/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.20169","created_at":"2026-07-05T11:44:14.578119+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.20169v1","created_at":"2026-07-05T11:44:14.578119+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.20169","created_at":"2026-07-05T11:44:14.578119+00:00"},{"alias_kind":"pith_short_12","alias_value":"QXDAPKGI2YVA","created_at":"2026-07-05T11:44:14.578119+00:00"},{"alias_kind":"pith_short_16","alias_value":"QXDAPKGI2YVAFGUX","created_at":"2026-07-05T11:44:14.578119+00:00"},{"alias_kind":"pith_short_8","alias_value":"QXDAPKGI","created_at":"2026-07-05T11:44:14.578119+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.01164","citing_title":"A Multimodal Deep Learning Framework for Early Diagnosis of Liver Cancer via Optimized BiLSTM-AM-VMD Architecture","ref_index":98,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM","json":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM.json","graph_json":"https://pith.science/api/pith-number/QXDAPKGI2YVAFGUXPFKZT5OBSM/graph.json","events_json":"https://pith.science/api/pith-number/QXDAPKGI2YVAFGUXPFKZT5OBSM/events.json","paper":"https://pith.science/paper/QXDAPKGI"},"agent_actions":{"view_html":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM","download_json":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM.json","view_paper":"https://pith.science/paper/QXDAPKGI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.20169&json=true","fetch_graph":"https://pith.science/api/pith-number/QXDAPKGI2YVAFGUXPFKZT5OBSM/graph.json","fetch_events":"https://pith.science/api/pith-number/QXDAPKGI2YVAFGUXPFKZT5OBSM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM/action/storage_attestation","attest_author":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM/action/author_attestation","sign_citation":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM/action/citation_signature","submit_replication":"https://pith.science/pith/QXDAPKGI2YVAFGUXPFKZT5OBSM/action/replication_record"}},"created_at":"2026-07-05T11:44:14.578119+00:00","updated_at":"2026-07-05T11:44:14.578119+00:00"}