{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7LVWLDPG6T6FQDJAJ3IXESNN35","short_pith_number":"pith:7LVWLDPG","schema_version":"1.0","canonical_sha256":"faeb658de6f4fc580d204ed17249addf6455ab3838aeb6b7e9c63e4b338de02b","source":{"kind":"arxiv","id":"2408.09639","version":2},"attestation_state":"computed","paper":{"title":"How to Make the Most of LLMs' Grammatical Knowledge for Acceptability Judgments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hidetaka Kamigaito, Justin Vasselli, Miyu Oba, Taro Watanabe, Yusuke Ide, Yusuke Sakai, Yuto Nishida","submitted_at":"2024-08-19T01:53:47Z","abstract_excerpt":"The grammatical knowledge of language models (LMs) is often measured using a benchmark of linguistic minimal pairs, where the LMs are presented with a pair of acceptable and unacceptable sentences and required to judge which is more acceptable. Conventional approaches directly compare sentence probabilities assigned by LMs, but recent large language models (LLMs) are trained to perform tasks via prompting, and thus, the raw probabilities they assign may not fully reflect their grammatical knowledge. In this study, we attempt to derive more accurate acceptability judgments from LLMs using promp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.09639","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-19T01:53:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8771ff474f8baa5ea6b4792c86a907bea7eac8f5c72540c2a3f462766dcfde93","abstract_canon_sha256":"3181a8e42b1b8a3507565f255d36e2384d81126ae37faee684d501ac3221d59f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:43.245195Z","signature_b64":"Nh7MjUF6kUtC58F+7m3+dWNOrlQHKlDrVkrBnvezbyuOnmiIG302aZP8/NkiepHAwWATqBFdjwk7ApPKnasJDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"faeb658de6f4fc580d204ed17249addf6455ab3838aeb6b7e9c63e4b338de02b","last_reissued_at":"2026-07-05T10:10:43.244719Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:43.244719Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to Make the Most of LLMs' Grammatical Knowledge for Acceptability Judgments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hidetaka Kamigaito, Justin Vasselli, Miyu Oba, Taro Watanabe, Yusuke Ide, Yusuke Sakai, Yuto Nishida","submitted_at":"2024-08-19T01:53:47Z","abstract_excerpt":"The grammatical knowledge of language models (LMs) is often measured using a benchmark of linguistic minimal pairs, where the LMs are presented with a pair of acceptable and unacceptable sentences and required to judge which is more acceptable. Conventional approaches directly compare sentence probabilities assigned by LMs, but recent large language models (LLMs) are trained to perform tasks via prompting, and thus, the raw probabilities they assign may not fully reflect their grammatical knowledge. In this study, we attempt to derive more accurate acceptability judgments from LLMs using promp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.09639","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.09639/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.09639","created_at":"2026-07-05T10:10:43.244778+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.09639v2","created_at":"2026-07-05T10:10:43.244778+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.09639","created_at":"2026-07-05T10:10:43.244778+00:00"},{"alias_kind":"pith_short_12","alias_value":"7LVWLDPG6T6F","created_at":"2026-07-05T10:10:43.244778+00:00"},{"alias_kind":"pith_short_16","alias_value":"7LVWLDPG6T6FQDJA","created_at":"2026-07-05T10:10:43.244778+00:00"},{"alias_kind":"pith_short_8","alias_value":"7LVWLDPG","created_at":"2026-07-05T10:10:43.244778+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.15175","citing_title":"Linear representations of grammaticality in neural language models","ref_index":100,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35","json":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35.json","graph_json":"https://pith.science/api/pith-number/7LVWLDPG6T6FQDJAJ3IXESNN35/graph.json","events_json":"https://pith.science/api/pith-number/7LVWLDPG6T6FQDJAJ3IXESNN35/events.json","paper":"https://pith.science/paper/7LVWLDPG"},"agent_actions":{"view_html":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35","download_json":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35.json","view_paper":"https://pith.science/paper/7LVWLDPG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.09639&json=true","fetch_graph":"https://pith.science/api/pith-number/7LVWLDPG6T6FQDJAJ3IXESNN35/graph.json","fetch_events":"https://pith.science/api/pith-number/7LVWLDPG6T6FQDJAJ3IXESNN35/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35/action/storage_attestation","attest_author":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35/action/author_attestation","sign_citation":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35/action/citation_signature","submit_replication":"https://pith.science/pith/7LVWLDPG6T6FQDJAJ3IXESNN35/action/replication_record"}},"created_at":"2026-07-05T10:10:43.244778+00:00","updated_at":"2026-07-05T10:10:43.244778+00:00"}