{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LCPC6HNQSFFR62SOOVN56CLT2X","short_pith_number":"pith:LCPC6HNQ","schema_version":"1.0","canonical_sha256":"589e2f1db0914b1f6a4e755bdf0973d5d67649b9437343dc0072a36a2ff75423","source":{"kind":"arxiv","id":"2401.06751","version":2},"attestation_state":"computed","paper":{"title":"The Unreasonable Effectiveness of Easy Training Data for Hard Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Mohit Bansal, Peter Clark, Peter Hase, Sarah Wiegreffe","submitted_at":"2024-01-12T18:36:29Z","abstract_excerpt":"How can we train models to perform well on hard test data when hard training data is by definition difficult to label correctly? This question has been termed the scalable oversight problem and has drawn increasing attention as language models have continually improved. In this paper, we present the surprising conclusion that current pretrained language models often generalize relatively well from easy to hard data, even performing as well as oracle models finetuned on hard data. We demonstrate this kind of easy-to-hard generalization using simple finetuning methods like in-context learning, l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06751","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-12T18:36:29Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"975c8b17e5b4958395eea46fae0607f37df89819939d233effa6e4f0e7ccbd50","abstract_canon_sha256":"85ba87a9ae0c58a800de655de4d83f42cf3b286403ae5bb2d57df1ded17e1352"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:45.462023Z","signature_b64":"spNhrK6bhPjGG6C03YwR2dYYrC7vYOuqK2uMANX6ey1OxjUy34CwOJdOitc0Kx+rEk+Yh3uy6LLVs0lJ20W+Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"589e2f1db0914b1f6a4e755bdf0973d5d67649b9437343dc0072a36a2ff75423","last_reissued_at":"2026-07-05T08:27:45.461513Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:45.461513Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Unreasonable Effectiveness of Easy Training Data for Hard Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Mohit Bansal, Peter Clark, Peter Hase, Sarah Wiegreffe","submitted_at":"2024-01-12T18:36:29Z","abstract_excerpt":"How can we train models to perform well on hard test data when hard training data is by definition difficult to label correctly? This question has been termed the scalable oversight problem and has drawn increasing attention as language models have continually improved. In this paper, we present the surprising conclusion that current pretrained language models often generalize relatively well from easy to hard data, even performing as well as oracle models finetuned on hard data. We demonstrate this kind of easy-to-hard generalization using simple finetuning methods like in-context learning, l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06751","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06751/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06751","created_at":"2026-07-05T08:27:45.461578+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06751v2","created_at":"2026-07-05T08:27:45.461578+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06751","created_at":"2026-07-05T08:27:45.461578+00:00"},{"alias_kind":"pith_short_12","alias_value":"LCPC6HNQSFFR","created_at":"2026-07-05T08:27:45.461578+00:00"},{"alias_kind":"pith_short_16","alias_value":"LCPC6HNQSFFR62SO","created_at":"2026-07-05T08:27:45.461578+00:00"},{"alias_kind":"pith_short_8","alias_value":"LCPC6HNQ","created_at":"2026-07-05T08:27:45.461578+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.01612","citing_title":"Self-Improving Transformers Overcome Easy-to-Hard and Length Generalization Challenges","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X","json":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X.json","graph_json":"https://pith.science/api/pith-number/LCPC6HNQSFFR62SOOVN56CLT2X/graph.json","events_json":"https://pith.science/api/pith-number/LCPC6HNQSFFR62SOOVN56CLT2X/events.json","paper":"https://pith.science/paper/LCPC6HNQ"},"agent_actions":{"view_html":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X","download_json":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X.json","view_paper":"https://pith.science/paper/LCPC6HNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06751&json=true","fetch_graph":"https://pith.science/api/pith-number/LCPC6HNQSFFR62SOOVN56CLT2X/graph.json","fetch_events":"https://pith.science/api/pith-number/LCPC6HNQSFFR62SOOVN56CLT2X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X/action/storage_attestation","attest_author":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X/action/author_attestation","sign_citation":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X/action/citation_signature","submit_replication":"https://pith.science/pith/LCPC6HNQSFFR62SOOVN56CLT2X/action/replication_record"}},"created_at":"2026-07-05T08:27:45.461578+00:00","updated_at":"2026-07-05T08:27:45.461578+00:00"}