{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:MFU2KJRAYNWGV6U4E3N5BNZXGL","short_pith_number":"pith:MFU2KJRA","schema_version":"1.0","canonical_sha256":"6169a52620c36c6afa9c26dbd0b73732c2842c0094067e595b11dda5624cced2","source":{"kind":"arxiv","id":"2202.11558","version":1},"attestation_state":"computed","paper":{"title":"Short-answer scoring with ensembles of pretrained language models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christopher Ormerod","submitted_at":"2022-02-23T15:12:20Z","abstract_excerpt":"We investigate the effectiveness of ensembles of pretrained transformer-based language models on short answer questions using the Kaggle Automated Short Answer Scoring dataset. We fine-tune a collection of popular small, base, and large pretrained transformer-based language models, and train one feature-base model on the dataset with the aim of testing ensembles of these models. We used an early stopping mechanism and hyperparameter optimization in training. We observe that generally that the larger models perform slightly better, however, they still fall short of state-of-the-art results one "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.11558","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-23T15:12:20Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cb8eee22e7ca6ddd33c871c06bb92fa398a4143ad042f50c7c0f1499bda06c71","abstract_canon_sha256":"7dad9d82e38e41a663ecbc427777cc7bd46f88584dc99acc890fb2e4d0ab8f67"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:59:34.418871Z","signature_b64":"kaBTamlhH5+5Nscj+gDnc2aZw1aMqRKMgp+QLnXykNDge98U9DJQoPXLO5aZAu54KYDvQdV9kITZLX+L+rlPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6169a52620c36c6afa9c26dbd0b73732c2842c0094067e595b11dda5624cced2","last_reissued_at":"2026-07-05T03:59:34.418407Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:59:34.418407Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Short-answer scoring with ensembles of pretrained language models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christopher Ormerod","submitted_at":"2022-02-23T15:12:20Z","abstract_excerpt":"We investigate the effectiveness of ensembles of pretrained transformer-based language models on short answer questions using the Kaggle Automated Short Answer Scoring dataset. We fine-tune a collection of popular small, base, and large pretrained transformer-based language models, and train one feature-base model on the dataset with the aim of testing ensembles of these models. We used an early stopping mechanism and hyperparameter optimization in training. We observe that generally that the larger models perform slightly better, however, they still fall short of state-of-the-art results one "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.11558","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.11558/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.11558","created_at":"2026-07-05T03:59:34.418466+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.11558v1","created_at":"2026-07-05T03:59:34.418466+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.11558","created_at":"2026-07-05T03:59:34.418466+00:00"},{"alias_kind":"pith_short_12","alias_value":"MFU2KJRAYNWG","created_at":"2026-07-05T03:59:34.418466+00:00"},{"alias_kind":"pith_short_16","alias_value":"MFU2KJRAYNWGV6U4","created_at":"2026-07-05T03:59:34.418466+00:00"},{"alias_kind":"pith_short_8","alias_value":"MFU2KJRA","created_at":"2026-07-05T03:59:34.418466+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.22771","citing_title":"Automated Essay Scoring Incorporating Annotations from Automated Feedback Systems","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL","json":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL.json","graph_json":"https://pith.science/api/pith-number/MFU2KJRAYNWGV6U4E3N5BNZXGL/graph.json","events_json":"https://pith.science/api/pith-number/MFU2KJRAYNWGV6U4E3N5BNZXGL/events.json","paper":"https://pith.science/paper/MFU2KJRA"},"agent_actions":{"view_html":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL","download_json":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL.json","view_paper":"https://pith.science/paper/MFU2KJRA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.11558&json=true","fetch_graph":"https://pith.science/api/pith-number/MFU2KJRAYNWGV6U4E3N5BNZXGL/graph.json","fetch_events":"https://pith.science/api/pith-number/MFU2KJRAYNWGV6U4E3N5BNZXGL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL/action/storage_attestation","attest_author":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL/action/author_attestation","sign_citation":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL/action/citation_signature","submit_replication":"https://pith.science/pith/MFU2KJRAYNWGV6U4E3N5BNZXGL/action/replication_record"}},"created_at":"2026-07-05T03:59:34.418466+00:00","updated_at":"2026-07-05T03:59:34.418466+00:00"}