{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3WAUBHXKS4N5VINYWJNL2EYVPQ","short_pith_number":"pith:3WAUBHXK","schema_version":"1.0","canonical_sha256":"dd81409eea971bdaa1b8b25abd13157c1e6ab89c17f15a8a1e3301f891b57143","source":{"kind":"arxiv","id":"2512.09066","version":2},"attestation_state":"computed","paper":{"title":"ORCA: Open-ended Response Correctness Assessment for Audio Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.SD","authors_text":"Alicia Lozano-Diez, Allison Ferner, Bolaji Yusuf, Cecilia Bola\\~nos, Fernando L\\'opez, Jan \\v{C}ernock\\'y, Laura Herrera-Alarc\\'on, Ramani Duraiswami, Santosh Kesiraju, Sara Barahona, Sathvik Udupa, \\v{S}imon Sedl\\'a\\v{c}ek","submitted_at":"2025-11-28T14:41:48Z","abstract_excerpt":"Reliable assessment of the abilities of large audio language models (LALMs) is essential to advancing the state of the art. As benchmarks rapidly evolve to incorporate complex reasoning and subjective tasks, they increasingly necessitate open-ended responses from LALMs. We present Open-ended Response Correctness Assessment (ORCA) -- a reliable and lightweight model-based approach for answer correctness and disagreement modeling. We employ a three-stage annotation pipeline combining human judgment, structured feedback, and human-AI correction, yielding 9,663 annotations across 3,699 question-an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.09066","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2025-11-28T14:41:48Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"9436952b7de2c6da8e49b429f027eac03d1f706f4fa72b3d0c1b605d33c0dc3e","abstract_canon_sha256":"b0804716ba9b4dcf5a3ae2ac2c12f4e4415c1112ccd92dac2652bb314c03deea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-30T02:17:12.744989Z","signature_b64":"aS/SU78sx45ys5GCujs1Dbag0x6Y2KqrD1BSmI56af+mHdcnZXLvceWqBQ3sepdplZuYPDJg72sdzpvWlD4IAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd81409eea971bdaa1b8b25abd13157c1e6ab89c17f15a8a1e3301f891b57143","last_reissued_at":"2026-06-30T02:17:12.744398Z","signature_status":"signed_v1","first_computed_at":"2026-06-30T02:17:12.744398Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ORCA: Open-ended Response Correctness Assessment for Audio Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.SD","authors_text":"Alicia Lozano-Diez, Allison Ferner, Bolaji Yusuf, Cecilia Bola\\~nos, Fernando L\\'opez, Jan \\v{C}ernock\\'y, Laura Herrera-Alarc\\'on, Ramani Duraiswami, Santosh Kesiraju, Sara Barahona, Sathvik Udupa, \\v{S}imon Sedl\\'a\\v{c}ek","submitted_at":"2025-11-28T14:41:48Z","abstract_excerpt":"Reliable assessment of the abilities of large audio language models (LALMs) is essential to advancing the state of the art. As benchmarks rapidly evolve to incorporate complex reasoning and subjective tasks, they increasingly necessitate open-ended responses from LALMs. We present Open-ended Response Correctness Assessment (ORCA) -- a reliable and lightweight model-based approach for answer correctness and disagreement modeling. We employ a three-stage annotation pipeline combining human judgment, structured feedback, and human-AI correction, yielding 9,663 annotations across 3,699 question-an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.09066","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.09066/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.09066","created_at":"2026-06-30T02:17:12.744473+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.09066v2","created_at":"2026-06-30T02:17:12.744473+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.09066","created_at":"2026-06-30T02:17:12.744473+00:00"},{"alias_kind":"pith_short_12","alias_value":"3WAUBHXKS4N5","created_at":"2026-06-30T02:17:12.744473+00:00"},{"alias_kind":"pith_short_16","alias_value":"3WAUBHXKS4N5VINY","created_at":"2026-06-30T02:17:12.744473+00:00"},{"alias_kind":"pith_short_8","alias_value":"3WAUBHXK","created_at":"2026-06-30T02:17:12.744473+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ","json":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ.json","graph_json":"https://pith.science/api/pith-number/3WAUBHXKS4N5VINYWJNL2EYVPQ/graph.json","events_json":"https://pith.science/api/pith-number/3WAUBHXKS4N5VINYWJNL2EYVPQ/events.json","paper":"https://pith.science/paper/3WAUBHXK"},"agent_actions":{"view_html":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ","download_json":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ.json","view_paper":"https://pith.science/paper/3WAUBHXK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.09066&json=true","fetch_graph":"https://pith.science/api/pith-number/3WAUBHXKS4N5VINYWJNL2EYVPQ/graph.json","fetch_events":"https://pith.science/api/pith-number/3WAUBHXKS4N5VINYWJNL2EYVPQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ/action/storage_attestation","attest_author":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ/action/author_attestation","sign_citation":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ/action/citation_signature","submit_replication":"https://pith.science/pith/3WAUBHXKS4N5VINYWJNL2EYVPQ/action/replication_record"}},"created_at":"2026-06-30T02:17:12.744473+00:00","updated_at":"2026-06-30T02:17:12.744473+00:00"}