{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YLLIZ5CPXOSDZ6JLH7TJYCC7B6","short_pith_number":"pith:YLLIZ5CP","schema_version":"1.0","canonical_sha256":"c2d68cf44fbba43cf92b3fe69c085f0f935fa5901b10ea02bf3b661cdc5c9673","source":{"kind":"arxiv","id":"2305.14091","version":3},"attestation_state":"computed","paper":{"title":"Revisiting Acceptability Judgements","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aini Li, Chien-Jer Charles Lin, Hai Hu, Jackie Yan-Ki Lai, Jiahui Huang, Peng Zhang, Rui Wang, Weifang Huang, Yina Patterson, Ziyin Zhang","submitted_at":"2023-05-23T14:16:22Z","abstract_excerpt":"In this work, we revisit linguistic acceptability in the context of large language models. We introduce CoLAC - Corpus of Linguistic Acceptability in Chinese, the first large-scale acceptability dataset for a non-Indo-European language. It is verified by native speakers and is the first acceptability dataset that comes with two sets of labels: a linguist label and a crowd label. Our experiments show that even the largest InstructGPT model performs only at chance level on CoLAC, while ChatGPT's performance (48.30 MCC) is also much below supervised models (59.03 MCC) and human (65.11 MCC). Throu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14091","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T14:16:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e58afde59de3ff8b4de60aefa7baa603caf4f1356a9afe7f2989aee4daa26e97","abstract_canon_sha256":"80d80a4120cc62ab814192d5ffef946d425ea909d3391d9065758becfcb5e35b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:03.788911Z","signature_b64":"mzn1keJnbOC/pctXMTipTtMMPZlFq0ohSqI3YkIdzGHr1d7uVBbUQQgOb2hBP/muXGCFENgw1t7wts2tAxcWCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2d68cf44fbba43cf92b3fe69c085f0f935fa5901b10ea02bf3b661cdc5c9673","last_reissued_at":"2026-07-05T06:55:03.788402Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:03.788402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Acceptability Judgements","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aini Li, Chien-Jer Charles Lin, Hai Hu, Jackie Yan-Ki Lai, Jiahui Huang, Peng Zhang, Rui Wang, Weifang Huang, Yina Patterson, Ziyin Zhang","submitted_at":"2023-05-23T14:16:22Z","abstract_excerpt":"In this work, we revisit linguistic acceptability in the context of large language models. We introduce CoLAC - Corpus of Linguistic Acceptability in Chinese, the first large-scale acceptability dataset for a non-Indo-European language. It is verified by native speakers and is the first acceptability dataset that comes with two sets of labels: a linguist label and a crowd label. Our experiments show that even the largest InstructGPT model performs only at chance level on CoLAC, while ChatGPT's performance (48.30 MCC) is also much below supervised models (59.03 MCC) and human (65.11 MCC). Throu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14091","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14091/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14091","created_at":"2026-07-05T06:55:03.788467+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14091v3","created_at":"2026-07-05T06:55:03.788467+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14091","created_at":"2026-07-05T06:55:03.788467+00:00"},{"alias_kind":"pith_short_12","alias_value":"YLLIZ5CPXOSD","created_at":"2026-07-05T06:55:03.788467+00:00"},{"alias_kind":"pith_short_16","alias_value":"YLLIZ5CPXOSDZ6JL","created_at":"2026-07-05T06:55:03.788467+00:00"},{"alias_kind":"pith_short_8","alias_value":"YLLIZ5CP","created_at":"2026-07-05T06:55:03.788467+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.11867","citing_title":"COLA-GEC: A Bidirectional Framework for Enhancing Grammatical Acceptability and Error Correction","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6","json":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6.json","graph_json":"https://pith.science/api/pith-number/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/graph.json","events_json":"https://pith.science/api/pith-number/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/events.json","paper":"https://pith.science/paper/YLLIZ5CP"},"agent_actions":{"view_html":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6","download_json":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6.json","view_paper":"https://pith.science/paper/YLLIZ5CP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14091&json=true","fetch_graph":"https://pith.science/api/pith-number/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/graph.json","fetch_events":"https://pith.science/api/pith-number/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/action/storage_attestation","attest_author":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/action/author_attestation","sign_citation":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/action/citation_signature","submit_replication":"https://pith.science/pith/YLLIZ5CPXOSDZ6JLH7TJYCC7B6/action/replication_record"}},"created_at":"2026-07-05T06:55:03.788467+00:00","updated_at":"2026-07-05T06:55:03.788467+00:00"}