{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UEVERYKY7OXLEA46Z3XM5ECYBS","short_pith_number":"pith:UEVERYKY","schema_version":"1.0","canonical_sha256":"a12a48e158fbaeb2039eceeece90580cb32b6cb12d0e4ce1dc474bbd5ef329c5","source":{"kind":"arxiv","id":"2401.01867","version":1},"attestation_state":"computed","paper":{"title":"Dataset Difficulty and the Role of Inductive Bias","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"David Rolnick, Devin Kwok, Gintare Karolina Dziugaite, Jonathan Frankle, Nikhil Anand","submitted_at":"2024-01-03T18:19:51Z","abstract_excerpt":"Motivated by the goals of dataset pruning and defect identification, a growing body of methods have been developed to score individual examples within a dataset. These methods, which we call \"example difficulty scores\", are typically used to rank or categorize examples, but the consistency of rankings between different training runs, scoring methods, and model architectures is generally unknown. To determine how example rankings vary due to these random and controlled effects, we systematically compare different formulations of scores over a range of runs and model architectures. We find that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.01867","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-01-03T18:19:51Z","cross_cats_sorted":[],"title_canon_sha256":"09776f9b2c43be25c4cfdb2c2e5dbb6a5dc25bfa9c0ed473d23c20206583b6fe","abstract_canon_sha256":"4888f46735741b9e781db9f1e08ef0c64c0a679d57bfab37298dec37e7f6b689"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:58.479495Z","signature_b64":"xaidrXgbBGjFYyIfoq2T/Xj7b9fDxvbDTEmoQ0zpyUcykz00DBLwm2IOTSAPu1jd8XPQ3wm4F7WoVIdmsnb4Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a12a48e158fbaeb2039eceeece90580cb32b6cb12d0e4ce1dc474bbd5ef329c5","last_reissued_at":"2026-07-05T07:29:58.479092Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:58.479092Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dataset Difficulty and the Role of Inductive Bias","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"David Rolnick, Devin Kwok, Gintare Karolina Dziugaite, Jonathan Frankle, Nikhil Anand","submitted_at":"2024-01-03T18:19:51Z","abstract_excerpt":"Motivated by the goals of dataset pruning and defect identification, a growing body of methods have been developed to score individual examples within a dataset. These methods, which we call \"example difficulty scores\", are typically used to rank or categorize examples, but the consistency of rankings between different training runs, scoring methods, and model architectures is generally unknown. To determine how example rankings vary due to these random and controlled effects, we systematically compare different formulations of scores over a range of runs and model architectures. We find that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.01867","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.01867/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.01867","created_at":"2026-07-05T07:29:58.479149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.01867v1","created_at":"2026-07-05T07:29:58.479149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.01867","created_at":"2026-07-05T07:29:58.479149+00:00"},{"alias_kind":"pith_short_12","alias_value":"UEVERYKY7OXL","created_at":"2026-07-05T07:29:58.479149+00:00"},{"alias_kind":"pith_short_16","alias_value":"UEVERYKY7OXLEA46","created_at":"2026-07-05T07:29:58.479149+00:00"},{"alias_kind":"pith_short_8","alias_value":"UEVERYKY","created_at":"2026-07-05T07:29:58.479149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.03648","citing_title":"Disentangling the Roles of Representation and Selection in Data Pruning","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS","json":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS.json","graph_json":"https://pith.science/api/pith-number/UEVERYKY7OXLEA46Z3XM5ECYBS/graph.json","events_json":"https://pith.science/api/pith-number/UEVERYKY7OXLEA46Z3XM5ECYBS/events.json","paper":"https://pith.science/paper/UEVERYKY"},"agent_actions":{"view_html":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS","download_json":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS.json","view_paper":"https://pith.science/paper/UEVERYKY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.01867&json=true","fetch_graph":"https://pith.science/api/pith-number/UEVERYKY7OXLEA46Z3XM5ECYBS/graph.json","fetch_events":"https://pith.science/api/pith-number/UEVERYKY7OXLEA46Z3XM5ECYBS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS/action/storage_attestation","attest_author":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS/action/author_attestation","sign_citation":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS/action/citation_signature","submit_replication":"https://pith.science/pith/UEVERYKY7OXLEA46Z3XM5ECYBS/action/replication_record"}},"created_at":"2026-07-05T07:29:58.479149+00:00","updated_at":"2026-07-05T07:29:58.479149+00:00"}