{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:7RZ4SHTBQS5Z6DA3OEQFNBJ4OK","short_pith_number":"pith:7RZ4SHTB","schema_version":"1.0","canonical_sha256":"fc73c91e6184bb9f0c1b712056853c72b39856344b488ad4e23380ea5cd98e50","source":{"kind":"arxiv","id":"2010.09535","version":2},"attestation_state":"computed","paper":{"title":"Cold-start Active Learning through Self-supervised Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hsuan-Tien Lin, Jordan Boyd-Graber, Michelle Yuan","submitted_at":"2020-10-19T14:09:17Z","abstract_excerpt":"Active learning strives to reduce annotation costs by choosing the most critical examples to label. Typically, the active learning strategy is contingent on the classification model. For instance, uncertainty sampling depends on poorly calibrated model confidence scores. In the cold-start setting, active learning is impractical because of model instability and data scarcity. Fortunately, modern NLP provides an additional source of information: pre-trained language models. The pre-training loss can find examples that surprise the model and should be labeled for efficient fine-tuning. Therefore,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.09535","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-19T14:09:17Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3504550758d793a248946acf53dadf0b517f286a61bb217e9c6f2083ec5db533","abstract_canon_sha256":"94a341bed706c68eeda65a012a676043ba4c3613b755f3845ff1367c9f06bad6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:45:31.207299Z","signature_b64":"XuKZf+mtS0vd2ISb/sh1dmRNvAjtZ3+kIZvueTiJJ3SCMtFygtzyJs4NXRruxTmA0dyHz4C9eTj/whrb5W4eCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc73c91e6184bb9f0c1b712056853c72b39856344b488ad4e23380ea5cd98e50","last_reissued_at":"2026-07-05T01:45:31.206862Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:45:31.206862Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cold-start Active Learning through Self-supervised Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hsuan-Tien Lin, Jordan Boyd-Graber, Michelle Yuan","submitted_at":"2020-10-19T14:09:17Z","abstract_excerpt":"Active learning strives to reduce annotation costs by choosing the most critical examples to label. Typically, the active learning strategy is contingent on the classification model. For instance, uncertainty sampling depends on poorly calibrated model confidence scores. In the cold-start setting, active learning is impractical because of model instability and data scarcity. Fortunately, modern NLP provides an additional source of information: pre-trained language models. The pre-training loss can find examples that surprise the model and should be labeled for efficient fine-tuning. Therefore,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.09535","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.09535/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.09535","created_at":"2026-07-05T01:45:31.206919+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.09535v2","created_at":"2026-07-05T01:45:31.206919+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.09535","created_at":"2026-07-05T01:45:31.206919+00:00"},{"alias_kind":"pith_short_12","alias_value":"7RZ4SHTBQS5Z","created_at":"2026-07-05T01:45:31.206919+00:00"},{"alias_kind":"pith_short_16","alias_value":"7RZ4SHTBQS5Z6DA3","created_at":"2026-07-05T01:45:31.206919+00:00"},{"alias_kind":"pith_short_8","alias_value":"7RZ4SHTB","created_at":"2026-07-05T01:45:31.206919+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02423","citing_title":"Neuron-Aware Active Few-Shot Learning for LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12303","citing_title":"Labeled TrustSet Guided: Batch Active Learning with Reinforcement Learning","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK","json":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK.json","graph_json":"https://pith.science/api/pith-number/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/graph.json","events_json":"https://pith.science/api/pith-number/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/events.json","paper":"https://pith.science/paper/7RZ4SHTB"},"agent_actions":{"view_html":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK","download_json":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK.json","view_paper":"https://pith.science/paper/7RZ4SHTB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.09535&json=true","fetch_graph":"https://pith.science/api/pith-number/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/graph.json","fetch_events":"https://pith.science/api/pith-number/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/action/storage_attestation","attest_author":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/action/author_attestation","sign_citation":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/action/citation_signature","submit_replication":"https://pith.science/pith/7RZ4SHTBQS5Z6DA3OEQFNBJ4OK/action/replication_record"}},"created_at":"2026-07-05T01:45:31.206919+00:00","updated_at":"2026-07-05T01:45:31.206919+00:00"}