{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OJDOILEOYLAQB7DVJLKDNPJJFM","short_pith_number":"pith:OJDOILEO","schema_version":"1.0","canonical_sha256":"7246e42c8ec2c100fc754ad436bd292b2bc8f759c503d66ebc626e2a887882ef","source":{"kind":"arxiv","id":"2512.05933","version":2},"attestation_state":"computed","paper":{"title":"Speech World Model: Causal State-Action Planning with Explicit Reasoning for Speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Gopala Anumanchipalli, Henry Hong, Jiachen Lian, Xinyi Yang, Xuanru Zhou","submitted_at":"2025-12-05T18:19:36Z","abstract_excerpt":"Current speech-language models (SLMs) typically use a cascade of speech encoder and large language model, treating speech understanding as a single black box. They analyze the content of speech well but reason weakly about other aspects, especially under sparse supervision. Thus, we argue for explicit reasoning over speech states and actions with modular and transparent decisions. Inspired by cognitive science we adopt a modular perspective and a world model view in which the system learns forward dynamics over latent states. We factorize speech understanding into four modules that communicate"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.05933","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2025-12-05T18:19:36Z","cross_cats_sorted":[],"title_canon_sha256":"89f7fda261503ede59b597676a03cc8bc5d78d31e63088d4cc0cd133446bf0f9","abstract_canon_sha256":"9ceac40c87d493fd7df389e49fa7fe96b55605bf5300e165730a3b076270036a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:20:49.478059Z","signature_b64":"QEZRRRlJEWQ49uVRRuQzk5VN9IBLfo5Z67n1IaPC+x6sCGut5LlDG/NR9GDuA27NtpjWXfeJtbG0uwP77S+CCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7246e42c8ec2c100fc754ad436bd292b2bc8f759c503d66ebc626e2a887882ef","last_reissued_at":"2026-07-14T01:20:49.477120Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:20:49.477120Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speech World Model: Causal State-Action Planning with Explicit Reasoning for Speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Gopala Anumanchipalli, Henry Hong, Jiachen Lian, Xinyi Yang, Xuanru Zhou","submitted_at":"2025-12-05T18:19:36Z","abstract_excerpt":"Current speech-language models (SLMs) typically use a cascade of speech encoder and large language model, treating speech understanding as a single black box. They analyze the content of speech well but reason weakly about other aspects, especially under sparse supervision. Thus, we argue for explicit reasoning over speech states and actions with modular and transparent decisions. Inspired by cognitive science we adopt a modular perspective and a world model view in which the system learns forward dynamics over latent states. We factorize speech understanding into four modules that communicate"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.05933","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.05933/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.05933","created_at":"2026-07-14T01:20:49.477579+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.05933v2","created_at":"2026-07-14T01:20:49.477579+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.05933","created_at":"2026-07-14T01:20:49.477579+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJDOILEOYLAQ","created_at":"2026-07-14T01:20:49.477579+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJDOILEOYLAQB7DV","created_at":"2026-07-14T01:20:49.477579+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJDOILEO","created_at":"2026-07-14T01:20:49.477579+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2605.20266","citing_title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","ref_index":94,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM","json":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM.json","graph_json":"https://pith.science/api/pith-number/OJDOILEOYLAQB7DVJLKDNPJJFM/graph.json","events_json":"https://pith.science/api/pith-number/OJDOILEOYLAQB7DVJLKDNPJJFM/events.json","paper":"https://pith.science/paper/OJDOILEO"},"agent_actions":{"view_html":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM","download_json":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM.json","view_paper":"https://pith.science/paper/OJDOILEO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.05933&json=true","fetch_graph":"https://pith.science/api/pith-number/OJDOILEOYLAQB7DVJLKDNPJJFM/graph.json","fetch_events":"https://pith.science/api/pith-number/OJDOILEOYLAQB7DVJLKDNPJJFM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM/action/storage_attestation","attest_author":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM/action/author_attestation","sign_citation":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM/action/citation_signature","submit_replication":"https://pith.science/pith/OJDOILEOYLAQB7DVJLKDNPJJFM/action/replication_record"}},"created_at":"2026-07-14T01:20:49.477579+00:00","updated_at":"2026-07-14T01:20:49.477579+00:00"}