{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4LVJREPZHGEIJR4E6JUXDL64C5","short_pith_number":"pith:4LVJREPZ","schema_version":"1.0","canonical_sha256":"e2ea9891f9398884c784f26971afdc174d163474cdb5c835f88ac57dd47e2273","source":{"kind":"arxiv","id":"2308.03279","version":2},"attestation_state":"computed","paper":{"title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hoifung Poon, Muhao Chen, Sheng Zhang, Wenxuan Zhou, Yu Gu","submitted_at":"2023-08-07T03:39:52Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable generalizability, such as understanding arbitrary entities and relations. Instruction tuning has proven effective for distilling LLMs into more cost-efficient models such as Alpaca and Vicuna. Yet such student models still trail the original LLMs by large margins in downstream applications. In this paper, we explore targeted distillation with mission-focused instruction tuning to train student models that can excel in a broad application class such as open information extraction. Using named entity recognition (NER) for case study, we s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.03279","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-07T03:39:52Z","cross_cats_sorted":[],"title_canon_sha256":"3b7897fa8319c7fc973e6c395b515b1fe26c5619e00d0774f44cf8a011a9e1d7","abstract_canon_sha256":"c768760ccd9a38b8c71839e99a377fcb5f5f72d4a77aded3bdf73a8e66fb502b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:35:20.205925Z","signature_b64":"zkJpeBzjEDHhInvwR7tPEgyAVXAMz+5RMk2WNizp/E+AsC86klzLQ8XopoYaTHyNNPR4LnwDC8//oiJ4/XR9BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e2ea9891f9398884c784f26971afdc174d163474cdb5c835f88ac57dd47e2273","last_reissued_at":"2026-07-05T07:35:20.205491Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:35:20.205491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hoifung Poon, Muhao Chen, Sheng Zhang, Wenxuan Zhou, Yu Gu","submitted_at":"2023-08-07T03:39:52Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable generalizability, such as understanding arbitrary entities and relations. Instruction tuning has proven effective for distilling LLMs into more cost-efficient models such as Alpaca and Vicuna. Yet such student models still trail the original LLMs by large margins in downstream applications. In this paper, we explore targeted distillation with mission-focused instruction tuning to train student models that can excel in a broad application class such as open information extraction. Using named entity recognition (NER) for case study, we s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.03279","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.03279/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.03279","created_at":"2026-07-05T07:35:20.205549+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.03279v2","created_at":"2026-07-05T07:35:20.205549+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.03279","created_at":"2026-07-05T07:35:20.205549+00:00"},{"alias_kind":"pith_short_12","alias_value":"4LVJREPZHGEI","created_at":"2026-07-05T07:35:20.205549+00:00"},{"alias_kind":"pith_short_16","alias_value":"4LVJREPZHGEIJR4E","created_at":"2026-07-05T07:35:20.205549+00:00"},{"alias_kind":"pith_short_8","alias_value":"4LVJREPZ","created_at":"2026-07-05T07:35:20.205549+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08268","citing_title":"Different Teachers, Different Capabilities: Sub-1B On-Device Distillation for Structured Text Enrichment","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18782","citing_title":"RedactionBench","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01293","citing_title":"RuleChef: Grounding LLM Task Knowledge in Human-Editable Rules","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04952","citing_title":"Quantifying Geospatial in the Common Crawl Corpus","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03706","citing_title":"SAM-NER: Semantic Archetype Mediation for Zero-Shot Named Entity Recognition","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06142","citing_title":"IRC-Bench: Recognizing Entities from Contextual Cues in First-Person Reminiscences","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06403","citing_title":"FMI@SU ToxHabits: Evaluating LLMs Performance on Toxic Habit Extraction in Spanish Clinical Texts","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20447","citing_title":"Decoding Text Spans for Efficient and Accurate Named-Entity Recognition","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5","json":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5.json","graph_json":"https://pith.science/api/pith-number/4LVJREPZHGEIJR4E6JUXDL64C5/graph.json","events_json":"https://pith.science/api/pith-number/4LVJREPZHGEIJR4E6JUXDL64C5/events.json","paper":"https://pith.science/paper/4LVJREPZ"},"agent_actions":{"view_html":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5","download_json":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5.json","view_paper":"https://pith.science/paper/4LVJREPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.03279&json=true","fetch_graph":"https://pith.science/api/pith-number/4LVJREPZHGEIJR4E6JUXDL64C5/graph.json","fetch_events":"https://pith.science/api/pith-number/4LVJREPZHGEIJR4E6JUXDL64C5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5/action/storage_attestation","attest_author":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5/action/author_attestation","sign_citation":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5/action/citation_signature","submit_replication":"https://pith.science/pith/4LVJREPZHGEIJR4E6JUXDL64C5/action/replication_record"}},"created_at":"2026-07-05T07:35:20.205549+00:00","updated_at":"2026-07-05T07:35:20.205549+00:00"}