{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:F7U35DSATN4A4TLATHUVNENQFB","short_pith_number":"pith:F7U35DSA","schema_version":"1.0","canonical_sha256":"2fe9be8e409b780e4d6099e95691b0284ad6c5609d3e3d6df07b4d270978d0e3","source":{"kind":"arxiv","id":"2310.19596","version":2},"attestation_state":"computed","paper":{"title":"LLMaAA: Making Large Language Models as Active Annotators","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Lei Zou, Ming Zhou, Ruoyu Zhang, Yanzeng Li, Yongliang Ma","submitted_at":"2023-10-30T14:54:15Z","abstract_excerpt":"Prevalent supervised learning methods in natural language processing (NLP) are notoriously data-hungry, which demand large amounts of high-quality annotated data. In practice, acquiring such data is a costly endeavor. Recently, the superior few-shot performance of large language models (LLMs) has propelled the development of dataset generation, where the training data are solely synthesized from LLMs. However, such an approach usually suffers from low-quality issues, and requires orders of magnitude more labeled data to achieve satisfactory performance. To fully exploit the potential of LLMs a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.19596","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-30T14:54:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"166f4a52fac2e624f21c9f80e2d81a08e25fd797a4c5cf7e8f4e023c109e6ea6","abstract_canon_sha256":"8bf09bf55d3af085b1c561d24d16ae772382e71759f9d2636643382031450852"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:07:19.398907Z","signature_b64":"vJDYmKrKAbXPxe10XfDjOqaUtco/nn9hGNKpC+PxdGcLga48/qZCnouSIg4ZCTmFmGIvuzpytBo/gGlg5nyDAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fe9be8e409b780e4d6099e95691b0284ad6c5609d3e3d6df07b4d270978d0e3","last_reissued_at":"2026-07-05T07:07:19.398426Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:07:19.398426Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMaAA: Making Large Language Models as Active Annotators","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Lei Zou, Ming Zhou, Ruoyu Zhang, Yanzeng Li, Yongliang Ma","submitted_at":"2023-10-30T14:54:15Z","abstract_excerpt":"Prevalent supervised learning methods in natural language processing (NLP) are notoriously data-hungry, which demand large amounts of high-quality annotated data. In practice, acquiring such data is a costly endeavor. Recently, the superior few-shot performance of large language models (LLMs) has propelled the development of dataset generation, where the training data are solely synthesized from LLMs. However, such an approach usually suffers from low-quality issues, and requires orders of magnitude more labeled data to achieve satisfactory performance. To fully exploit the potential of LLMs a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19596","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19596/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.19596","created_at":"2026-07-05T07:07:19.398480+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.19596v2","created_at":"2026-07-05T07:07:19.398480+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19596","created_at":"2026-07-05T07:07:19.398480+00:00"},{"alias_kind":"pith_short_12","alias_value":"F7U35DSATN4A","created_at":"2026-07-05T07:07:19.398480+00:00"},{"alias_kind":"pith_short_16","alias_value":"F7U35DSATN4A4TLA","created_at":"2026-07-05T07:07:19.398480+00:00"},{"alias_kind":"pith_short_8","alias_value":"F7U35DSA","created_at":"2026-07-05T07:07:19.398480+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.01482","citing_title":"Towards Consistent Detection of Cognitive Distortions: LLM-Based Annotation and Dataset-Agnostic Evaluation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16654","citing_title":"A Scalable Tool for Measuring Manner and Result Verbs in Developmental Language Research","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08578","citing_title":"Structured Exploration and Exploitation of Label Functions for Automated Data Annotation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":293,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05483","citing_title":"Can We Trust a Black-box LLM? LLM Untrustworthy Boundary Detection via Bias-Diffusion and Multi-Agent Reinforcement Learning","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB","json":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB.json","graph_json":"https://pith.science/api/pith-number/F7U35DSATN4A4TLATHUVNENQFB/graph.json","events_json":"https://pith.science/api/pith-number/F7U35DSATN4A4TLATHUVNENQFB/events.json","paper":"https://pith.science/paper/F7U35DSA"},"agent_actions":{"view_html":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB","download_json":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB.json","view_paper":"https://pith.science/paper/F7U35DSA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.19596&json=true","fetch_graph":"https://pith.science/api/pith-number/F7U35DSATN4A4TLATHUVNENQFB/graph.json","fetch_events":"https://pith.science/api/pith-number/F7U35DSATN4A4TLATHUVNENQFB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB/action/storage_attestation","attest_author":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB/action/author_attestation","sign_citation":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB/action/citation_signature","submit_replication":"https://pith.science/pith/F7U35DSATN4A4TLATHUVNENQFB/action/replication_record"}},"created_at":"2026-07-05T07:07:19.398480+00:00","updated_at":"2026-07-05T07:07:19.398480+00:00"}