{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:753BPG5QMV6WSXELCZ2OFE7V6C","short_pith_number":"pith:753BPG5Q","schema_version":"1.0","canonical_sha256":"ff76179bb0657d695c8b1674e293f5f09d9d3ea56c13bab2b092aa15ae433f58","source":{"kind":"arxiv","id":"2407.07890","version":3},"attestation_state":"computed","paper":{"title":"Training on the Test Task Confounds Evaluation and Emergence","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Florian E. Dorner, Moritz Hardt, Ricardo Dominguez-Olmedo","submitted_at":"2024-07-10T17:57:58Z","abstract_excerpt":"We study a fundamental problem in the evaluation of large language models that we call training on the test task. Unlike wrongful practices like training on the test data, leakage, or data contamination, training on the test task is not a malpractice. Rather, the term describes a growing set of practices that utilize knowledge about evaluation tasks at training time. We demonstrate that training on the test task confounds both relative model evaluations and claims about emergent capabilities. We argue that the seeming superiority of one model family over another may be explained by a different"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07890","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-10T17:57:58Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9ef4ef72eba68fc147a2983407f6ae2f94eec644981e7e9eef82e06b950d0cfa","abstract_canon_sha256":"b2c14cfae54cb7bec3177bdc96407bf3ef46f57d089ac8705156d3d9780304ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:23.863861Z","signature_b64":"fg9EeGk/iQb6yzJ8QZWObelLWwBfQaG0vxkifUr7XZ1fXE1ZSLGowpFN6O1q7k0boEgEB5vaXykGTbJkTTniBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff76179bb0657d695c8b1674e293f5f09d9d3ea56c13bab2b092aa15ae433f58","last_reissued_at":"2026-07-05T10:51:23.863357Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:23.863357Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training on the Test Task Confounds Evaluation and Emergence","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Florian E. Dorner, Moritz Hardt, Ricardo Dominguez-Olmedo","submitted_at":"2024-07-10T17:57:58Z","abstract_excerpt":"We study a fundamental problem in the evaluation of large language models that we call training on the test task. Unlike wrongful practices like training on the test data, leakage, or data contamination, training on the test task is not a malpractice. Rather, the term describes a growing set of practices that utilize knowledge about evaluation tasks at training time. We demonstrate that training on the test task confounds both relative model evaluations and claims about emergent capabilities. We argue that the seeming superiority of one model family over another may be explained by a different"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07890","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07890/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07890","created_at":"2026-07-05T10:51:23.863415+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07890v3","created_at":"2026-07-05T10:51:23.863415+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07890","created_at":"2026-07-05T10:51:23.863415+00:00"},{"alias_kind":"pith_short_12","alias_value":"753BPG5QMV6W","created_at":"2026-07-05T10:51:23.863415+00:00"},{"alias_kind":"pith_short_16","alias_value":"753BPG5QMV6WSXEL","created_at":"2026-07-05T10:51:23.863415+00:00"},{"alias_kind":"pith_short_8","alias_value":"753BPG5Q","created_at":"2026-07-05T10:51:23.863415+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05029","citing_title":"Validity Threats for Foundation Model Research","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20448","citing_title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20448","citing_title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02833","citing_title":"Generalizing Verifiable Instruction Following","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23009","citing_title":"Position: Stop Evaluating AI with Human Tests, Develop Principled, AI-specific Tests instead","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14164","citing_title":"Unsteady Metrics and Benchmarking Cultures of AI Model Builders","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C","json":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C.json","graph_json":"https://pith.science/api/pith-number/753BPG5QMV6WSXELCZ2OFE7V6C/graph.json","events_json":"https://pith.science/api/pith-number/753BPG5QMV6WSXELCZ2OFE7V6C/events.json","paper":"https://pith.science/paper/753BPG5Q"},"agent_actions":{"view_html":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C","download_json":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C.json","view_paper":"https://pith.science/paper/753BPG5Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07890&json=true","fetch_graph":"https://pith.science/api/pith-number/753BPG5QMV6WSXELCZ2OFE7V6C/graph.json","fetch_events":"https://pith.science/api/pith-number/753BPG5QMV6WSXELCZ2OFE7V6C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C/action/storage_attestation","attest_author":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C/action/author_attestation","sign_citation":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C/action/citation_signature","submit_replication":"https://pith.science/pith/753BPG5QMV6WSXELCZ2OFE7V6C/action/replication_record"}},"created_at":"2026-07-05T10:51:23.863415+00:00","updated_at":"2026-07-05T10:51:23.863415+00:00"}