{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DS4C3VBD6CDWWE62CORZIXNO62","short_pith_number":"pith:DS4C3VBD","schema_version":"1.0","canonical_sha256":"1cb82dd423f0876b13da13a3945daef699e65e4dd575968e2c2ba22a42dad1e9","source":{"kind":"arxiv","id":"2402.15343","version":1},"attestation_state":"computed","paper":{"title":"NuNER: Entity Recognition Encoder Pre-training via LLM-Annotated Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandre Constantin, Benoit Crabb\\'e, Etienne Bernard, Sergei Bogdanov, Timoth\\'ee Bernard","submitted_at":"2024-02-23T14:23:51Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive abilities in data annotation, opening the way for new approaches to solve classic NLP problems. In this paper, we show how to use LLMs to create NuNER, a compact language representation model specialized in the Named Entity Recognition (NER) task. NuNER can be fine-tuned to solve downstream NER problems in a data-efficient way, outperforming similar-sized foundation models in the few-shot regime and competing with much larger LLMs. We find that the size and entity-type diversity of the pre-training dataset are key to achieving good performance"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.15343","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-23T14:23:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f99d1d0b4c43c352b536da414bad223fc1fe03208b947836bb1a644b5365f6aa","abstract_canon_sha256":"c9de48a978b5d39735b550d6d4f9fae0710152f3b79f00d3ed49cf2b7eecfec1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:39.394430Z","signature_b64":"anN/TEOTVMnEcAzU+iwwZINzBZIzUmiUWhjc5QrJDCqkoXO4pIpCU5WMXLnpk9r2vj/Rdo+3tLnXRo+SxBWIDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1cb82dd423f0876b13da13a3945daef699e65e4dd575968e2c2ba22a42dad1e9","last_reissued_at":"2026-07-05T07:48:39.393999Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:39.393999Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NuNER: Entity Recognition Encoder Pre-training via LLM-Annotated Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandre Constantin, Benoit Crabb\\'e, Etienne Bernard, Sergei Bogdanov, Timoth\\'ee Bernard","submitted_at":"2024-02-23T14:23:51Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive abilities in data annotation, opening the way for new approaches to solve classic NLP problems. In this paper, we show how to use LLMs to create NuNER, a compact language representation model specialized in the Named Entity Recognition (NER) task. NuNER can be fine-tuned to solve downstream NER problems in a data-efficient way, outperforming similar-sized foundation models in the few-shot regime and competing with much larger LLMs. We find that the size and entity-type diversity of the pre-training dataset are key to achieving good performance"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.15343","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.15343/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.15343","created_at":"2026-07-05T07:48:39.394055+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.15343v1","created_at":"2026-07-05T07:48:39.394055+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.15343","created_at":"2026-07-05T07:48:39.394055+00:00"},{"alias_kind":"pith_short_12","alias_value":"DS4C3VBD6CDW","created_at":"2026-07-05T07:48:39.394055+00:00"},{"alias_kind":"pith_short_16","alias_value":"DS4C3VBD6CDWWE62","created_at":"2026-07-05T07:48:39.394055+00:00"},{"alias_kind":"pith_short_8","alias_value":"DS4C3VBD","created_at":"2026-07-05T07:48:39.394055+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29659","citing_title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10108","citing_title":"GLiNER-Relex: A Unified Framework for Joint Named Entity Recognition and Relation Extraction","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06940","citing_title":"MultiSoc-4D: A Benchmark for Diagnosing Instruction-Induced Label Collapse in Closed-Set LLM Annotation of Bengali Social Media","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15866","citing_title":"DiZiNER: Disagreement-guided Instruction Refinement via Pilot Annotation Simulation for Zero-shot Named Entity Recognition","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62","json":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62.json","graph_json":"https://pith.science/api/pith-number/DS4C3VBD6CDWWE62CORZIXNO62/graph.json","events_json":"https://pith.science/api/pith-number/DS4C3VBD6CDWWE62CORZIXNO62/events.json","paper":"https://pith.science/paper/DS4C3VBD"},"agent_actions":{"view_html":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62","download_json":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62.json","view_paper":"https://pith.science/paper/DS4C3VBD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.15343&json=true","fetch_graph":"https://pith.science/api/pith-number/DS4C3VBD6CDWWE62CORZIXNO62/graph.json","fetch_events":"https://pith.science/api/pith-number/DS4C3VBD6CDWWE62CORZIXNO62/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62/action/storage_attestation","attest_author":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62/action/author_attestation","sign_citation":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62/action/citation_signature","submit_replication":"https://pith.science/pith/DS4C3VBD6CDWWE62CORZIXNO62/action/replication_record"}},"created_at":"2026-07-05T07:48:39.394055+00:00","updated_at":"2026-07-05T07:48:39.394055+00:00"}