{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PAP3QSG4LLHTGEWY7UUDUUATUG","short_pith_number":"pith:PAP3QSG4","schema_version":"1.0","canonical_sha256":"781fb848dc5acf3312d8fd283a5013a1a9f3020f1f89296f89190eab290f4865","source":{"kind":"arxiv","id":"2309.00916","version":2},"attestation_state":"computed","paper":{"title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chengqing Zong, Chen Wang, Jiajun Zhang, Jinliang Lu, Junhong Wu, Minpeng Liao, Yuchen Liu, Zhongqiang Huang","submitted_at":"2023-09-02T11:46:05Z","abstract_excerpt":"The emergence of large language models (LLMs) has sparked significant interest in extending their remarkable language capabilities to speech. However, modality alignment between speech and text still remains an open problem. Current solutions can be categorized into two strategies. One is a cascaded approach where outputs (tokens or states) of a separately trained speech recognition system are used as inputs for LLMs, which limits their potential in modeling alignment between speech and text. The other is an end-to-end approach that relies on speech instruction data, which is very difficult to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.00916","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-02T11:46:05Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"d167c5b1de06a4977421ccc6e735cdc60aa998ab0249f9187efa70d681ea6e7b","abstract_canon_sha256":"cdbff23cbe4d81f8416c2db2d92b6f863a65df2467fb3a4bb513718a775f7076"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:54.894626Z","signature_b64":"Rn+VQk4pT4rhb7XQ6vE4cZlgPiZFtP3DpjdePcCiVOOSUAhlzczyrUU4p5Ix1TZgxA1Yl+iLe0xZhhEua9lDCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"781fb848dc5acf3312d8fd283a5013a1a9f3020f1f89296f89190eab290f4865","last_reissued_at":"2026-07-05T08:23:54.894105Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:54.894105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chengqing Zong, Chen Wang, Jiajun Zhang, Jinliang Lu, Junhong Wu, Minpeng Liao, Yuchen Liu, Zhongqiang Huang","submitted_at":"2023-09-02T11:46:05Z","abstract_excerpt":"The emergence of large language models (LLMs) has sparked significant interest in extending their remarkable language capabilities to speech. However, modality alignment between speech and text still remains an open problem. Current solutions can be categorized into two strategies. One is a cascaded approach where outputs (tokens or states) of a separately trained speech recognition system are used as inputs for LLMs, which limits their potential in modeling alignment between speech and text. The other is an end-to-end approach that relies on speech instruction data, which is very difficult to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.00916","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.00916/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.00916","created_at":"2026-07-05T08:23:54.894165+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.00916v2","created_at":"2026-07-05T08:23:54.894165+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.00916","created_at":"2026-07-05T08:23:54.894165+00:00"},{"alias_kind":"pith_short_12","alias_value":"PAP3QSG4LLHT","created_at":"2026-07-05T08:23:54.894165+00:00"},{"alias_kind":"pith_short_16","alias_value":"PAP3QSG4LLHTGEWY","created_at":"2026-07-05T08:23:54.894165+00:00"},{"alias_kind":"pith_short_8","alias_value":"PAP3QSG4","created_at":"2026-07-05T08:23:54.894165+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23313","citing_title":"Uncertainty-based Debiasing and Unlearning for Decontamination","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21453","citing_title":"CORTIS: Text-Only Adaptation of Spoken Language Models for Task-Oriented Voice Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12199","citing_title":"Which Speech Representation Better Matches Text-Native Reasoning? A Study of Speech-Text Alignment on Frame Rate and Representation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11033","citing_title":"AuRA: Internalizing Audio Understanding into LLMs as LoRA","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03526","citing_title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21008","citing_title":"A Survey of Audio Reasoning in Multimodal Foundation Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23511","citing_title":"MECAT: A Multi-Experts Constructed Benchmark for Fine-Grained Audio Understanding Tasks","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25591","citing_title":"Walking Through Uncertainty: An Empirical Study of Uncertainty Estimation for Audio-Aware Large Language Models","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG","json":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG.json","graph_json":"https://pith.science/api/pith-number/PAP3QSG4LLHTGEWY7UUDUUATUG/graph.json","events_json":"https://pith.science/api/pith-number/PAP3QSG4LLHTGEWY7UUDUUATUG/events.json","paper":"https://pith.science/paper/PAP3QSG4"},"agent_actions":{"view_html":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG","download_json":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG.json","view_paper":"https://pith.science/paper/PAP3QSG4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.00916&json=true","fetch_graph":"https://pith.science/api/pith-number/PAP3QSG4LLHTGEWY7UUDUUATUG/graph.json","fetch_events":"https://pith.science/api/pith-number/PAP3QSG4LLHTGEWY7UUDUUATUG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG/action/storage_attestation","attest_author":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG/action/author_attestation","sign_citation":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG/action/citation_signature","submit_replication":"https://pith.science/pith/PAP3QSG4LLHTGEWY7UUDUUATUG/action/replication_record"}},"created_at":"2026-07-05T08:23:54.894165+00:00","updated_at":"2026-07-05T08:23:54.894165+00:00"}