{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MYU4I7HP3FNJ3XTZRDYE6WUP5S","short_pith_number":"pith:MYU4I7HP","schema_version":"1.0","canonical_sha256":"6629c47cefd95a9dde7988f04f5a8fec8eecad9fdb9a4a0e46c370bc8a4c086f","source":{"kind":"arxiv","id":"2506.16393","version":1},"attestation_state":"computed","paper":{"title":"From LLM-anation to LLM-orchestrator: Coordinating Small Models for Data Labeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiawei Du, Qi Xuan, Tianyi Zhou, Yao Lu, Yu Shanqing, Zhaiyuan Ji","submitted_at":"2025-06-19T15:26:08Z","abstract_excerpt":"Although the annotation paradigm based on Large Language Models (LLMs) has made significant breakthroughs in recent years, its actual deployment still has two core bottlenecks: first, the cost of calling commercial APIs in large-scale annotation is very expensive; second, in scenarios that require fine-grained semantic understanding, such as sentiment classification and toxicity classification, the annotation accuracy of LLMs is even lower than that of Small Language Models (SLMs) dedicated to this field. To address these problems, we propose a new paradigm of multi-model cooperative annotatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.16393","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-19T15:26:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"daf428616a8249c7484bf4f0e1e630193ef457ac90012107baf193b38035f204","abstract_canon_sha256":"a5c1fa445d5272d189cabf3ed4f6247003af5db84ef06c350c3a9baee31265e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:35.715253Z","signature_b64":"N3ywRwmU4d88i7XL/L6gjRAN9HTPNCgKVGdRKHp4XGx1Mq4fXWLkZ+rVqKPG0AkkqiC7kdp1DhCeysi2idpXCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6629c47cefd95a9dde7988f04f5a8fec8eecad9fdb9a4a0e46c370bc8a4c086f","last_reissued_at":"2026-07-05T11:24:35.714708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:35.714708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From LLM-anation to LLM-orchestrator: Coordinating Small Models for Data Labeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiawei Du, Qi Xuan, Tianyi Zhou, Yao Lu, Yu Shanqing, Zhaiyuan Ji","submitted_at":"2025-06-19T15:26:08Z","abstract_excerpt":"Although the annotation paradigm based on Large Language Models (LLMs) has made significant breakthroughs in recent years, its actual deployment still has two core bottlenecks: first, the cost of calling commercial APIs in large-scale annotation is very expensive; second, in scenarios that require fine-grained semantic understanding, such as sentiment classification and toxicity classification, the annotation accuracy of LLMs is even lower than that of Small Language Models (SLMs) dedicated to this field. To address these problems, we propose a new paradigm of multi-model cooperative annotatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.16393","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.16393/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.16393","created_at":"2026-07-05T11:24:35.714763+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.16393v1","created_at":"2026-07-05T11:24:35.714763+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.16393","created_at":"2026-07-05T11:24:35.714763+00:00"},{"alias_kind":"pith_short_12","alias_value":"MYU4I7HP3FNJ","created_at":"2026-07-05T11:24:35.714763+00:00"},{"alias_kind":"pith_short_16","alias_value":"MYU4I7HP3FNJ3XTZ","created_at":"2026-07-05T11:24:35.714763+00:00"},{"alias_kind":"pith_short_8","alias_value":"MYU4I7HP","created_at":"2026-07-05T11:24:35.714763+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.01161","citing_title":"Multi-Modal Machine Learning Framework for Predicting Early Recurrence of Brain Tumors Using MRI and Clinical Biomarkers","ref_index":54,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S","json":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S.json","graph_json":"https://pith.science/api/pith-number/MYU4I7HP3FNJ3XTZRDYE6WUP5S/graph.json","events_json":"https://pith.science/api/pith-number/MYU4I7HP3FNJ3XTZRDYE6WUP5S/events.json","paper":"https://pith.science/paper/MYU4I7HP"},"agent_actions":{"view_html":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S","download_json":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S.json","view_paper":"https://pith.science/paper/MYU4I7HP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.16393&json=true","fetch_graph":"https://pith.science/api/pith-number/MYU4I7HP3FNJ3XTZRDYE6WUP5S/graph.json","fetch_events":"https://pith.science/api/pith-number/MYU4I7HP3FNJ3XTZRDYE6WUP5S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S/action/storage_attestation","attest_author":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S/action/author_attestation","sign_citation":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S/action/citation_signature","submit_replication":"https://pith.science/pith/MYU4I7HP3FNJ3XTZRDYE6WUP5S/action/replication_record"}},"created_at":"2026-07-05T11:24:35.714763+00:00","updated_at":"2026-07-05T11:24:35.714763+00:00"}