{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:IJCR4YDLZ4L4HCDF3EV6JAOMV6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5facbdf9852de9c26be35689917acf0779e8cdee3982513f1413f5103457e02d","cross_cats_sorted":["cs.CY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T09:58:57Z","title_canon_sha256":"ab7a1d63430cab9ea80c0a5000e9f6e01a18932fcb9ee42ed3519649d9fc40b9"},"schema_version":"1.0","source":{"id":"2404.01799","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.01799","created_at":"2026-07-05T11:26:22Z"},{"alias_kind":"arxiv_version","alias_value":"2404.01799v3","created_at":"2026-07-05T11:26:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01799","created_at":"2026-07-05T11:26:22Z"},{"alias_kind":"pith_short_12","alias_value":"IJCR4YDLZ4L4","created_at":"2026-07-05T11:26:22Z"},{"alias_kind":"pith_short_16","alias_value":"IJCR4YDLZ4L4HCDF","created_at":"2026-07-05T11:26:22Z"},{"alias_kind":"pith_short_8","alias_value":"IJCR4YDL","created_at":"2026-07-05T11:26:22Z"}],"graph_snapshots":[{"event_id":"sha256:fbab45d89307a17a1861e88b536e69fafc90acea56e657b9c435918b0eeb56f6","target":"graph","created_at":"2026-07-05T11:26:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2404.01799/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Many existing benchmarks of large (multimodal) language models (LLMs) focus on measuring LLMs' academic proficiency, often with also an interest in comparing model performance with human test takers'. While such benchmarks have proven key to the development of LLMs, they suffer from several limitations, including questionable measurement quality (e.g., Do they measure what they are supposed to in a reliable way?), lack of quality assessment on the item level (e.g., Are some items more important or difficult than others?) and unclear human population reference (e.g., To whom can the model be co","authors_text":"Daniel L. Oberski, Dong Nguyen, Qixiang Fang","cross_cats":["cs.CY"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T09:58:57Z","title":"PATCH! {P}sychometrics-{A}ssis{T}ed Ben{CH}marking of Large Language Models against Human Populations: A Case Study of Proficiency in 8th Grade Mathematics"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01799","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0689e55de7b08ae65d4728a7c4109523140f537ce3b3861c7e77115e035b2fe7","target":"record","created_at":"2026-07-05T11:26:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5facbdf9852de9c26be35689917acf0779e8cdee3982513f1413f5103457e02d","cross_cats_sorted":["cs.CY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T09:58:57Z","title_canon_sha256":"ab7a1d63430cab9ea80c0a5000e9f6e01a18932fcb9ee42ed3519649d9fc40b9"},"schema_version":"1.0","source":{"id":"2404.01799","kind":"arxiv","version":3}},"canonical_sha256":"42451e606bcf17c38865d92be481ccaf81d777f3d5cc682b4fadf00b4652b5ca","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"42451e606bcf17c38865d92be481ccaf81d777f3d5cc682b4fadf00b4652b5ca","first_computed_at":"2026-07-05T11:26:22.249002Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:26:22.249002Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"9NcFXZTySxIn3ULnG+agzZ0m108BMv3r5FefLQ1lM+5idUdqSFzlov0W936MdNlGpDVHJNHSNt9HBRyRh30wDA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:26:22.249441Z","signed_message":"canonical_sha256_bytes"},"source_id":"2404.01799","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0689e55de7b08ae65d4728a7c4109523140f537ce3b3861c7e77115e035b2fe7","sha256:fbab45d89307a17a1861e88b536e69fafc90acea56e657b9c435918b0eeb56f6"],"state_sha256":"8db7c184ec4b6019cbba7f783d6d186e74e9a34ab5333b881b1041a35ce5a65a"}