{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KF4FLEHP5OWBHZEK2JYRVJNWN2","short_pith_number":"pith:KF4FLEHP","schema_version":"1.0","canonical_sha256":"51785590efebac13e48ad2711aa5b66ea4d49dc23a236d2af27a3d4b63461262","source":{"kind":"arxiv","id":"2410.03735","version":2},"attestation_state":"computed","paper":{"title":"Task-Adaptive Pretrained Language Models via Clustered-Importance Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"David Grangier, Pierre Ablin, Simin Fan, Skyler Seto","submitted_at":"2024-09-30T20:49:54Z","abstract_excerpt":"Specialist language models (LMs) focus on a specific task or domain on which they often outperform generalist LMs of the same size. However, the specialist data needed to pretrain these models is only available in limited amount for most tasks. In this work, we build specialist models from large generalist training sets instead. We propose a novel method, ClusteRed Importance SamPling (CRISP). CRISP clusters the generalist dataset and samples from these clusters based on their frequencies in the smaller specialist dataset. It is scalable, suitable for both pretraining and continued pretraining"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03735","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-30T20:49:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3fdeccd5a7d937677c9f2a87759cf3dd31fbae797b17af4d3bcacd1ce1e9c1ea","abstract_canon_sha256":"b6a3181321bb841d1cb86d9f034a0b15a3e1ccd871c7717817fe1accef8b9080"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:28:23.566853Z","signature_b64":"Vsu3rGHFjdM6Pj6hXjvCPr9rrFS3XGxKKai0L3Pk/miDaD0avtI8Dw2TZUXPM6kF3L+O76ODcBsPrQR97lzIDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"51785590efebac13e48ad2711aa5b66ea4d49dc23a236d2af27a3d4b63461262","last_reissued_at":"2026-07-05T10:28:23.566145Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:28:23.566145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Task-Adaptive Pretrained Language Models via Clustered-Importance Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"David Grangier, Pierre Ablin, Simin Fan, Skyler Seto","submitted_at":"2024-09-30T20:49:54Z","abstract_excerpt":"Specialist language models (LMs) focus on a specific task or domain on which they often outperform generalist LMs of the same size. However, the specialist data needed to pretrain these models is only available in limited amount for most tasks. In this work, we build specialist models from large generalist training sets instead. We propose a novel method, ClusteRed Importance SamPling (CRISP). CRISP clusters the generalist dataset and samples from these clusters based on their frequencies in the smaller specialist dataset. It is scalable, suitable for both pretraining and continued pretraining"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03735","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03735/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03735","created_at":"2026-07-05T10:28:23.566256+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03735v2","created_at":"2026-07-05T10:28:23.566256+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03735","created_at":"2026-07-05T10:28:23.566256+00:00"},{"alias_kind":"pith_short_12","alias_value":"KF4FLEHP5OWB","created_at":"2026-07-05T10:28:23.566256+00:00"},{"alias_kind":"pith_short_16","alias_value":"KF4FLEHP5OWBHZEK","created_at":"2026-07-05T10:28:23.566256+00:00"},{"alias_kind":"pith_short_8","alias_value":"KF4FLEHP","created_at":"2026-07-05T10:28:23.566256+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.08519","citing_title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2","json":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2.json","graph_json":"https://pith.science/api/pith-number/KF4FLEHP5OWBHZEK2JYRVJNWN2/graph.json","events_json":"https://pith.science/api/pith-number/KF4FLEHP5OWBHZEK2JYRVJNWN2/events.json","paper":"https://pith.science/paper/KF4FLEHP"},"agent_actions":{"view_html":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2","download_json":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2.json","view_paper":"https://pith.science/paper/KF4FLEHP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03735&json=true","fetch_graph":"https://pith.science/api/pith-number/KF4FLEHP5OWBHZEK2JYRVJNWN2/graph.json","fetch_events":"https://pith.science/api/pith-number/KF4FLEHP5OWBHZEK2JYRVJNWN2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2/action/storage_attestation","attest_author":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2/action/author_attestation","sign_citation":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2/action/citation_signature","submit_replication":"https://pith.science/pith/KF4FLEHP5OWBHZEK2JYRVJNWN2/action/replication_record"}},"created_at":"2026-07-05T10:28:23.566256+00:00","updated_at":"2026-07-05T10:28:23.566256+00:00"}