{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DBQQLTM74WLKKE6XDHIJZPX656","short_pith_number":"pith:DBQQLTM7","schema_version":"1.0","canonical_sha256":"186105cd9fe596a513d719d09cbefeefbd7205e54903059c6429ff8d7b9b8e48","source":{"kind":"arxiv","id":"2409.13120","version":1},"attestation_state":"computed","paper":{"title":"Are Large Language Models Good Essay Graders?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Anindita Kundu, Denilson Barbosa","submitted_at":"2024-09-19T23:20:49Z","abstract_excerpt":"We evaluate the effectiveness of Large Language Models (LLMs) in assessing essay quality, focusing on their alignment with human grading. More precisely, we evaluate ChatGPT and Llama in the Automated Essay Scoring (AES) task, a crucial natural language processing (NLP) application in Education. We consider both zero-shot and few-shot learning and different prompting approaches. We compare the numeric grade provided by the LLMs to human rater-provided scores utilizing the ASAP dataset, a well-known benchmark for the AES task. Our research reveals that both LLMs generally assign lower scores co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13120","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-19T23:20:49Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"3ccb58f108ae0b126a8f9d4b600154ae8beb6fa4592f77ad3d063ddd93bab104","abstract_canon_sha256":"d8fd638cf7e187fb2bd6e917cc80a0353d4deb0466387b5e07e9a383ee551f08"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:09:29.396433Z","signature_b64":"r3/oXleVwI1YHjaXxWROiyEZn9K29w/SmcidGmdlcyScpGBu0i6tnNXjDUbxq1gw3NgOsenyqsnUotqzG/h6DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"186105cd9fe596a513d719d09cbefeefbd7205e54903059c6429ff8d7b9b8e48","last_reissued_at":"2026-07-05T09:09:29.395907Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:09:29.395907Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Large Language Models Good Essay Graders?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Anindita Kundu, Denilson Barbosa","submitted_at":"2024-09-19T23:20:49Z","abstract_excerpt":"We evaluate the effectiveness of Large Language Models (LLMs) in assessing essay quality, focusing on their alignment with human grading. More precisely, we evaluate ChatGPT and Llama in the Automated Essay Scoring (AES) task, a crucial natural language processing (NLP) application in Education. We consider both zero-shot and few-shot learning and different prompting approaches. We compare the numeric grade provided by the LLMs to human rater-provided scores utilizing the ASAP dataset, a well-known benchmark for the AES task. Our research reveals that both LLMs generally assign lower scores co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13120","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13120/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13120","created_at":"2026-07-05T09:09:29.395967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13120v1","created_at":"2026-07-05T09:09:29.395967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13120","created_at":"2026-07-05T09:09:29.395967+00:00"},{"alias_kind":"pith_short_12","alias_value":"DBQQLTM74WLK","created_at":"2026-07-05T09:09:29.395967+00:00"},{"alias_kind":"pith_short_16","alias_value":"DBQQLTM74WLKKE6X","created_at":"2026-07-05T09:09:29.395967+00:00"},{"alias_kind":"pith_short_8","alias_value":"DBQQLTM7","created_at":"2026-07-05T09:09:29.395967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20152","citing_title":"From Texts to Scores: Tracing the Emergence of Essay Quality Representations in Large Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20287","citing_title":"PsyScore: A Psychometrically-Aware Framework for Trait-Adaptive Essay Scoring and ZPD-Scaffolded Feedback","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656","json":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656.json","graph_json":"https://pith.science/api/pith-number/DBQQLTM74WLKKE6XDHIJZPX656/graph.json","events_json":"https://pith.science/api/pith-number/DBQQLTM74WLKKE6XDHIJZPX656/events.json","paper":"https://pith.science/paper/DBQQLTM7"},"agent_actions":{"view_html":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656","download_json":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656.json","view_paper":"https://pith.science/paper/DBQQLTM7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13120&json=true","fetch_graph":"https://pith.science/api/pith-number/DBQQLTM74WLKKE6XDHIJZPX656/graph.json","fetch_events":"https://pith.science/api/pith-number/DBQQLTM74WLKKE6XDHIJZPX656/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656/action/storage_attestation","attest_author":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656/action/author_attestation","sign_citation":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656/action/citation_signature","submit_replication":"https://pith.science/pith/DBQQLTM74WLKKE6XDHIJZPX656/action/replication_record"}},"created_at":"2026-07-05T09:09:29.395967+00:00","updated_at":"2026-07-05T09:09:29.395967+00:00"}