{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:LJTPDE2XZNN6J76Z44MHDFCEV4","short_pith_number":"pith:LJTPDE2X","schema_version":"1.0","canonical_sha256":"5a66f19357cb5be4ffd9e718719444af107008d38bd1a84f2afcb7e7bfd62a9a","source":{"kind":"arxiv","id":"2608.06718","version":1},"attestation_state":"computed","paper":{"title":"Do Audio Language Models Use Paralinguistic Evidence? Counterfactual Audits for Response Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arjun Chandra, Kevin Miller, Venkatesh Saligrama","submitted_at":"2026-08-07T02:24:12Z","abstract_excerpt":"Audio-language models (ALMs) are increasingly used as judges for speech-to-speech systems, but a judge that receives audio may not actually use paralinguistic evidence. We introduce counterfactual audits for paralinguistic response evaluation. Each audit item holds the transcript fixed while varying affect, prosody, or the timing of an affective shift, forcing a valid judge to track the audio cue rather than lexical content or response style. We evaluate ALM judges using a native one-context judgment protocol and a contrastive recoverability control, then further decompose each item into its c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.06718","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-08-07T02:24:12Z","cross_cats_sorted":[],"title_canon_sha256":"2e0cb5d74f101c93229a9914ba6d34ac2dacb43a37df1e50d7c1eb22b8e6ae41","abstract_canon_sha256":"8f554e2ea9b8e314fed0b30639bd1f0aa107ef805fdd53cbbf1c6a898fb036b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-10T01:11:48.733415Z","signature_b64":"JgB+M8ElGM5YTu+8DtX/9LVtNKChMWuozgprzUjYjCBBRMMwuMB8Zs00q8fWEuB8TrUfUHjaPGRK6vVJg+9sBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a66f19357cb5be4ffd9e718719444af107008d38bd1a84f2afcb7e7bfd62a9a","last_reissued_at":"2026-08-10T01:11:48.730717Z","signature_status":"signed_v1","first_computed_at":"2026-08-10T01:11:48.730717Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Audio Language Models Use Paralinguistic Evidence? Counterfactual Audits for Response Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arjun Chandra, Kevin Miller, Venkatesh Saligrama","submitted_at":"2026-08-07T02:24:12Z","abstract_excerpt":"Audio-language models (ALMs) are increasingly used as judges for speech-to-speech systems, but a judge that receives audio may not actually use paralinguistic evidence. We introduce counterfactual audits for paralinguistic response evaluation. Each audit item holds the transcript fixed while varying affect, prosody, or the timing of an affective shift, forcing a valid judge to track the audio cue rather than lexical content or response style. We evaluate ALM judges using a native one-context judgment protocol and a contrastive recoverability control, then further decompose each item into its c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.06718","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.06718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.06718","created_at":"2026-08-10T01:11:48.732006+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.06718v1","created_at":"2026-08-10T01:11:48.732006+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.06718","created_at":"2026-08-10T01:11:48.732006+00:00"},{"alias_kind":"pith_short_12","alias_value":"LJTPDE2XZNN6","created_at":"2026-08-10T01:11:48.732006+00:00"},{"alias_kind":"pith_short_16","alias_value":"LJTPDE2XZNN6J76Z","created_at":"2026-08-10T01:11:48.732006+00:00"},{"alias_kind":"pith_short_8","alias_value":"LJTPDE2X","created_at":"2026-08-10T01:11:48.732006+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4","json":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4.json","graph_json":"https://pith.science/api/pith-number/LJTPDE2XZNN6J76Z44MHDFCEV4/graph.json","events_json":"https://pith.science/api/pith-number/LJTPDE2XZNN6J76Z44MHDFCEV4/events.json","paper":"https://pith.science/paper/LJTPDE2X"},"agent_actions":{"view_html":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4","download_json":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4.json","view_paper":"https://pith.science/paper/LJTPDE2X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.06718&json=true","fetch_graph":"https://pith.science/api/pith-number/LJTPDE2XZNN6J76Z44MHDFCEV4/graph.json","fetch_events":"https://pith.science/api/pith-number/LJTPDE2XZNN6J76Z44MHDFCEV4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4/action/storage_attestation","attest_author":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4/action/author_attestation","sign_citation":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4/action/citation_signature","submit_replication":"https://pith.science/pith/LJTPDE2XZNN6J76Z44MHDFCEV4/action/replication_record"}},"created_at":"2026-08-10T01:11:48.732006+00:00","updated_at":"2026-08-10T01:11:48.732006+00:00"}