{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:USWMHS32VF5YI7IRR37UXFCTK2","short_pith_number":"pith:USWMHS32","schema_version":"1.0","canonical_sha256":"a4acc3cb7aa97b847d118eff4b945356880cb33e4a337cf5c194236d3eff4514","source":{"kind":"arxiv","id":"2309.11508","version":2},"attestation_state":"computed","paper":{"title":"Towards LLM-based Autograding for Short Textual Answers","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bernd Schenk, Christina Niklaus, Johannes Schneider","submitted_at":"2023-09-09T22:25:56Z","abstract_excerpt":"Grading exams is an important, labor-intensive, subjective, repetitive, and frequently challenging task. The feasibility of autograding textual responses has greatly increased thanks to the availability of large language models (LLMs) such as ChatGPT and the substantial influx of data brought about by digitalization. However, entrusting AI models with decision-making roles raises ethical considerations, mainly stemming from potential biases and issues related to generating false information. Thus, in this manuscript, we provide an evaluation of a large language model for the purpose of autogra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.11508","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-09T22:25:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"111538be38d496d5a8c6fb023cc1fef0d7a732b46713131c4811e6b8fd4b07d7","abstract_canon_sha256":"e56be3c026124dd2318a59ddcdb5e2f4299af9e12841b3d57db170e3a99235c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:55.104209Z","signature_b64":"/SI4QvgcNIWeICCkTRf8DKCOmrAHswQEmmXLYM6GbjjoqqShyzKVerA6+pMKq+V05aXkSTncXvfImxCS/W85Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a4acc3cb7aa97b847d118eff4b945356880cb33e4a337cf5c194236d3eff4514","last_reissued_at":"2026-07-05T08:40:55.103726Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:55.103726Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards LLM-based Autograding for Short Textual Answers","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bernd Schenk, Christina Niklaus, Johannes Schneider","submitted_at":"2023-09-09T22:25:56Z","abstract_excerpt":"Grading exams is an important, labor-intensive, subjective, repetitive, and frequently challenging task. The feasibility of autograding textual responses has greatly increased thanks to the availability of large language models (LLMs) such as ChatGPT and the substantial influx of data brought about by digitalization. However, entrusting AI models with decision-making roles raises ethical considerations, mainly stemming from potential biases and issues related to generating false information. Thus, in this manuscript, we provide an evaluation of a large language model for the purpose of autogra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.11508","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.11508/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.11508","created_at":"2026-07-05T08:40:55.103786+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.11508v2","created_at":"2026-07-05T08:40:55.103786+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.11508","created_at":"2026-07-05T08:40:55.103786+00:00"},{"alias_kind":"pith_short_12","alias_value":"USWMHS32VF5Y","created_at":"2026-07-05T08:40:55.103786+00:00"},{"alias_kind":"pith_short_16","alias_value":"USWMHS32VF5YI7IR","created_at":"2026-07-05T08:40:55.103786+00:00"},{"alias_kind":"pith_short_8","alias_value":"USWMHS32","created_at":"2026-07-05T08:40:55.103786+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01247","citing_title":"LLMs as Teaching Assistants for Mathematics Exam Grading: Reliability, and Practical Usability","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03090","citing_title":"\"**Important** You should give me full credits!\": Exploring Prompt Injection Attacks on LLM-Based Automatic Grading Systems","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2","json":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2.json","graph_json":"https://pith.science/api/pith-number/USWMHS32VF5YI7IRR37UXFCTK2/graph.json","events_json":"https://pith.science/api/pith-number/USWMHS32VF5YI7IRR37UXFCTK2/events.json","paper":"https://pith.science/paper/USWMHS32"},"agent_actions":{"view_html":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2","download_json":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2.json","view_paper":"https://pith.science/paper/USWMHS32","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.11508&json=true","fetch_graph":"https://pith.science/api/pith-number/USWMHS32VF5YI7IRR37UXFCTK2/graph.json","fetch_events":"https://pith.science/api/pith-number/USWMHS32VF5YI7IRR37UXFCTK2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2/action/storage_attestation","attest_author":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2/action/author_attestation","sign_citation":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2/action/citation_signature","submit_replication":"https://pith.science/pith/USWMHS32VF5YI7IRR37UXFCTK2/action/replication_record"}},"created_at":"2026-07-05T08:40:55.103786+00:00","updated_at":"2026-07-05T08:40:55.103786+00:00"}