{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JF2RVPXLY7373HSUYYYJUP2WD2","short_pith_number":"pith:JF2RVPXL","schema_version":"1.0","canonical_sha256":"49751abeebc7f7fd9e54c6309a3f561e9962094627ac96b5bf1a120f88764391","source":{"kind":"arxiv","id":"2403.06149","version":2},"attestation_state":"computed","paper":{"title":"Can Large Language Models Automatically Score Proficiency of Written Essays?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Salam Albatarni, Sohaila Eltanbouly, Tamer Elsayed, Watheq Mansour","submitted_at":"2024-03-10T09:39:00Z","abstract_excerpt":"Although several methods were proposed to address the problem of automated essay scoring (AES) in the last 50 years, there is still much to desire in terms of effectiveness. Large Language Models (LLMs) are transformer-based models that demonstrate extraordinary capabilities on various tasks. In this paper, we test the ability of LLMs, given their powerful linguistic knowledge, to analyze and effectively score written essays. We experimented with two popular LLMs, namely ChatGPT and Llama. We aim to check if these models can do this task and, if so, how their performance is positioned among th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.06149","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-10T09:39:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"72b57ab3ed9fa789157fa49a926b1146c7d883dbbc063f32852c8bbed73ee3da","abstract_canon_sha256":"8aebd00228d8a268c588b30917b2446bb849af426955638167106bef8c91a664"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:21.397653Z","signature_b64":"QgqlElsEJdHnECAscYEx1rJUEqXwaasnDVEzw2+/8ebd3OxFNvt0ePl3lSdyQLQv/MxPbcCBHpXSvidt5qpzAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49751abeebc7f7fd9e54c6309a3f561e9962094627ac96b5bf1a120f88764391","last_reissued_at":"2026-07-05T08:08:21.397157Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:21.397157Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Automatically Score Proficiency of Written Essays?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Salam Albatarni, Sohaila Eltanbouly, Tamer Elsayed, Watheq Mansour","submitted_at":"2024-03-10T09:39:00Z","abstract_excerpt":"Although several methods were proposed to address the problem of automated essay scoring (AES) in the last 50 years, there is still much to desire in terms of effectiveness. Large Language Models (LLMs) are transformer-based models that demonstrate extraordinary capabilities on various tasks. In this paper, we test the ability of LLMs, given their powerful linguistic knowledge, to analyze and effectively score written essays. We experimented with two popular LLMs, namely ChatGPT and Llama. We aim to check if these models can do this task and, if so, how their performance is positioned among th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.06149","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.06149/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.06149","created_at":"2026-07-05T08:08:21.397217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.06149v2","created_at":"2026-07-05T08:08:21.397217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.06149","created_at":"2026-07-05T08:08:21.397217+00:00"},{"alias_kind":"pith_short_12","alias_value":"JF2RVPXLY737","created_at":"2026-07-05T08:08:21.397217+00:00"},{"alias_kind":"pith_short_16","alias_value":"JF2RVPXLY7373HSU","created_at":"2026-07-05T08:08:21.397217+00:00"},{"alias_kind":"pith_short_8","alias_value":"JF2RVPXL","created_at":"2026-07-05T08:08:21.397217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24973","citing_title":"LLM Performance on a Real, Double-Marked GCSE Benchmark","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20152","citing_title":"From Texts to Scores: Tracing the Emergence of Essay Quality Representations in Large Language Models","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2","json":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2.json","graph_json":"https://pith.science/api/pith-number/JF2RVPXLY7373HSUYYYJUP2WD2/graph.json","events_json":"https://pith.science/api/pith-number/JF2RVPXLY7373HSUYYYJUP2WD2/events.json","paper":"https://pith.science/paper/JF2RVPXL"},"agent_actions":{"view_html":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2","download_json":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2.json","view_paper":"https://pith.science/paper/JF2RVPXL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.06149&json=true","fetch_graph":"https://pith.science/api/pith-number/JF2RVPXLY7373HSUYYYJUP2WD2/graph.json","fetch_events":"https://pith.science/api/pith-number/JF2RVPXLY7373HSUYYYJUP2WD2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2/action/storage_attestation","attest_author":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2/action/author_attestation","sign_citation":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2/action/citation_signature","submit_replication":"https://pith.science/pith/JF2RVPXLY7373HSUYYYJUP2WD2/action/replication_record"}},"created_at":"2026-07-05T08:08:21.397217+00:00","updated_at":"2026-07-05T08:08:21.397217+00:00"}