{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PEZFCSQW7JTWMVO2BNSIOTIYVX","short_pith_number":"pith:PEZFCSQW","schema_version":"1.0","canonical_sha256":"7932514a16fa676655da0b64874d18adc009593e3ccf1a7e17544b5dcabb4fc3","source":{"kind":"arxiv","id":"2411.11260","version":2},"attestation_state":"computed","paper":{"title":"Large corpora and large language models: a replicable method for automating grammatical annotation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cameron Morin, Matti Marttinen Larsson","submitted_at":"2024-11-18T03:29:48Z","abstract_excerpt":"Much linguistic research relies on annotated datasets of features extracted from text corpora, but the rapid quantitative growth of these corpora has created practical difficulties for linguists to manually annotate large data samples. In this paper, we present a replicable, supervised method that leverages large language models for assisting the linguist in grammatical annotation through prompt engineering, training, and evaluation. We introduce a methodological pipeline applied to the case study of formal variation in the English evaluative verb construction 'consider X (as) (to be) Y', base"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11260","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-18T03:29:48Z","cross_cats_sorted":[],"title_canon_sha256":"5853b8b7ca4b0c37dc0f8f1d2a143d618fa725462935755b1fd4ee4b834ccefe","abstract_canon_sha256":"46b16ca4f1f29172ae15b155a39bfa28f37e5d4bbe4a272a8b7893e10d98e902"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:58.587474Z","signature_b64":"TIpPX1lkoFHN+yKXMw/vxGtbCwNVs5ytvvC2sXEZV6cZhjgmZ+ZMh+2eo9bhcM5sGKSHsHCc2aFZ+dwv4xZ2AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7932514a16fa676655da0b64874d18adc009593e3ccf1a7e17544b5dcabb4fc3","last_reissued_at":"2026-07-05T10:46:58.586858Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:58.586858Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large corpora and large language models: a replicable method for automating grammatical annotation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cameron Morin, Matti Marttinen Larsson","submitted_at":"2024-11-18T03:29:48Z","abstract_excerpt":"Much linguistic research relies on annotated datasets of features extracted from text corpora, but the rapid quantitative growth of these corpora has created practical difficulties for linguists to manually annotate large data samples. In this paper, we present a replicable, supervised method that leverages large language models for assisting the linguist in grammatical annotation through prompt engineering, training, and evaluation. We introduce a methodological pipeline applied to the case study of formal variation in the English evaluative verb construction 'consider X (as) (to be) Y', base"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11260","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11260/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11260","created_at":"2026-07-05T10:46:58.586954+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11260v2","created_at":"2026-07-05T10:46:58.586954+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11260","created_at":"2026-07-05T10:46:58.586954+00:00"},{"alias_kind":"pith_short_12","alias_value":"PEZFCSQW7JTW","created_at":"2026-07-05T10:46:58.586954+00:00"},{"alias_kind":"pith_short_16","alias_value":"PEZFCSQW7JTWMVO2","created_at":"2026-07-05T10:46:58.586954+00:00"},{"alias_kind":"pith_short_8","alias_value":"PEZFCSQW","created_at":"2026-07-05T10:46:58.586954+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX","json":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX.json","graph_json":"https://pith.science/api/pith-number/PEZFCSQW7JTWMVO2BNSIOTIYVX/graph.json","events_json":"https://pith.science/api/pith-number/PEZFCSQW7JTWMVO2BNSIOTIYVX/events.json","paper":"https://pith.science/paper/PEZFCSQW"},"agent_actions":{"view_html":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX","download_json":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX.json","view_paper":"https://pith.science/paper/PEZFCSQW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11260&json=true","fetch_graph":"https://pith.science/api/pith-number/PEZFCSQW7JTWMVO2BNSIOTIYVX/graph.json","fetch_events":"https://pith.science/api/pith-number/PEZFCSQW7JTWMVO2BNSIOTIYVX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX/action/storage_attestation","attest_author":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX/action/author_attestation","sign_citation":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX/action/citation_signature","submit_replication":"https://pith.science/pith/PEZFCSQW7JTWMVO2BNSIOTIYVX/action/replication_record"}},"created_at":"2026-07-05T10:46:58.586954+00:00","updated_at":"2026-07-05T10:46:58.586954+00:00"}