{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:P4PHYPPNDHS4O2SOO26IJUADGV","short_pith_number":"pith:P4PHYPPN","schema_version":"1.0","canonical_sha256":"7f1e7c3ded19e5c76a4e76bc84d0033545a5bbcbd72ae83737b0eaf8990996e4","source":{"kind":"arxiv","id":"2502.11916","version":2},"attestation_state":"computed","paper":{"title":"EssayJudge: A Multi-Granular Benchmark for Assessing Automated Essay Scoring Capabilities of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fangteng Fu, Han Zhang, Huiyu Zhou, Jiahao Huo, Jiamin Su, Jingheng Ye, Xiang Liu, Xuming Hu, Yibo Yan","submitted_at":"2025-02-17T15:31:59Z","abstract_excerpt":"Automated Essay Scoring (AES) plays a crucial role in educational assessment by providing scalable and consistent evaluations of writing tasks. However, traditional AES systems face three major challenges: (1) reliance on handcrafted features that limit generalizability, (2) difficulty in capturing fine-grained traits like coherence and argumentation, and (3) inability to handle multimodal contexts. In the era of Multimodal Large Language Models (MLLMs), we propose EssayJudge, the first multimodal benchmark to evaluate AES capabilities across lexical-, sentence-, and discourse-level traits. By"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11916","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-17T15:31:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"aeb57fffa093e5852a18572048dc7abeb23d9df713d56aff78ffbe053d170757","abstract_canon_sha256":"f1528d38000cdb06a45d49ab8f6ddc4ec2bf46a55f327ebb1b7ab3eabb39249c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:40.182953Z","signature_b64":"IYYd5j9thyoYGRjHRr0K1KT49BbhhDuW/0vLU5xjxwdI1BycVr73h6ShN5YOwgWcPZa7nOn6mJ0OfQfoZpDfAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f1e7c3ded19e5c76a4e76bc84d0033545a5bbcbd72ae83737b0eaf8990996e4","last_reissued_at":"2026-07-05T11:05:40.182461Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:40.182461Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EssayJudge: A Multi-Granular Benchmark for Assessing Automated Essay Scoring Capabilities of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fangteng Fu, Han Zhang, Huiyu Zhou, Jiahao Huo, Jiamin Su, Jingheng Ye, Xiang Liu, Xuming Hu, Yibo Yan","submitted_at":"2025-02-17T15:31:59Z","abstract_excerpt":"Automated Essay Scoring (AES) plays a crucial role in educational assessment by providing scalable and consistent evaluations of writing tasks. However, traditional AES systems face three major challenges: (1) reliance on handcrafted features that limit generalizability, (2) difficulty in capturing fine-grained traits like coherence and argumentation, and (3) inability to handle multimodal contexts. In the era of Multimodal Large Language Models (MLLMs), we propose EssayJudge, the first multimodal benchmark to evaluate AES capabilities across lexical-, sentence-, and discourse-level traits. By"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11916","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11916/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11916","created_at":"2026-07-05T11:05:40.182523+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11916v2","created_at":"2026-07-05T11:05:40.182523+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11916","created_at":"2026-07-05T11:05:40.182523+00:00"},{"alias_kind":"pith_short_12","alias_value":"P4PHYPPNDHS4","created_at":"2026-07-05T11:05:40.182523+00:00"},{"alias_kind":"pith_short_16","alias_value":"P4PHYPPNDHS4O2SO","created_at":"2026-07-05T11:05:40.182523+00:00"},{"alias_kind":"pith_short_8","alias_value":"P4PHYPPN","created_at":"2026-07-05T11:05:40.182523+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20287","citing_title":"PsyScore: A Psychometrically-Aware Framework for Trait-Adaptive Essay Scoring and ZPD-Scaffolded Feedback","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV","json":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV.json","graph_json":"https://pith.science/api/pith-number/P4PHYPPNDHS4O2SOO26IJUADGV/graph.json","events_json":"https://pith.science/api/pith-number/P4PHYPPNDHS4O2SOO26IJUADGV/events.json","paper":"https://pith.science/paper/P4PHYPPN"},"agent_actions":{"view_html":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV","download_json":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV.json","view_paper":"https://pith.science/paper/P4PHYPPN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11916&json=true","fetch_graph":"https://pith.science/api/pith-number/P4PHYPPNDHS4O2SOO26IJUADGV/graph.json","fetch_events":"https://pith.science/api/pith-number/P4PHYPPNDHS4O2SOO26IJUADGV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV/action/storage_attestation","attest_author":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV/action/author_attestation","sign_citation":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV/action/citation_signature","submit_replication":"https://pith.science/pith/P4PHYPPNDHS4O2SOO26IJUADGV/action/replication_record"}},"created_at":"2026-07-05T11:05:40.182523+00:00","updated_at":"2026-07-05T11:05:40.182523+00:00"}