{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:U45XPVT6YGMS3GFHN72TTOEN6U","short_pith_number":"pith:U45XPVT6","schema_version":"1.0","canonical_sha256":"a73b77d67ec1992d98a76ff539b88df53118c0ba54931b07430ed8ae461a3c00","source":{"kind":"arxiv","id":"2404.02261","version":2},"attestation_state":"computed","paper":{"title":"LLMs in the Loop: Leveraging Large Language Model Annotations for Active Learning in Low-Resource Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Michael Granitzer, Mohammad Khodadadi, Muhammed Nurullah Gumus, Nataliia Kholodna, Sahib Julka","submitted_at":"2024-04-02T19:34:22Z","abstract_excerpt":"Low-resource languages face significant barriers in AI development due to limited linguistic resources and expertise for data labeling, rendering them rare and costly. The scarcity of data and the absence of preexisting tools exacerbate these challenges, especially since these languages may not be adequately represented in various NLP datasets. To address this gap, we propose leveraging the potential of LLMs in the active learning loop for data annotation. Initially, we conduct evaluations to assess inter-annotator agreement and consistency, facilitating the selection of a suitable LLM annotat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.02261","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T19:34:22Z","cross_cats_sorted":["cs.AI","cs.IR","cs.LG"],"title_canon_sha256":"3f0a5f4499215356e83bacebc967562f3bb25bfbcbd8a2365812afb0268856fd","abstract_canon_sha256":"aaac615d1fd25e083dff67b3f08ea291f033b6c1beb579d36c8d48267f305564"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:45.024587Z","signature_b64":"+fCgD+5ohLQmqpIC84ofyMN2ilVbV8fgNqP0Pic3jBa3nf+pTUwSKBg3qUkMReHBJVjQZs43L47D0vtPbRyODw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a73b77d67ec1992d98a76ff539b88df53118c0ba54931b07430ed8ae461a3c00","last_reissued_at":"2026-07-05T08:35:45.024145Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:45.024145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs in the Loop: Leveraging Large Language Model Annotations for Active Learning in Low-Resource Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Michael Granitzer, Mohammad Khodadadi, Muhammed Nurullah Gumus, Nataliia Kholodna, Sahib Julka","submitted_at":"2024-04-02T19:34:22Z","abstract_excerpt":"Low-resource languages face significant barriers in AI development due to limited linguistic resources and expertise for data labeling, rendering them rare and costly. The scarcity of data and the absence of preexisting tools exacerbate these challenges, especially since these languages may not be adequately represented in various NLP datasets. To address this gap, we propose leveraging the potential of LLMs in the active learning loop for data annotation. Initially, we conduct evaluations to assess inter-annotator agreement and consistency, facilitating the selection of a suitable LLM annotat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.02261","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.02261/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.02261","created_at":"2026-07-05T08:35:45.024201+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.02261v2","created_at":"2026-07-05T08:35:45.024201+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.02261","created_at":"2026-07-05T08:35:45.024201+00:00"},{"alias_kind":"pith_short_12","alias_value":"U45XPVT6YGMS","created_at":"2026-07-05T08:35:45.024201+00:00"},{"alias_kind":"pith_short_16","alias_value":"U45XPVT6YGMS3GFH","created_at":"2026-07-05T08:35:45.024201+00:00"},{"alias_kind":"pith_short_8","alias_value":"U45XPVT6","created_at":"2026-07-05T08:35:45.024201+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.13383","citing_title":"Whose View of Safety? A Deep DIVE Dataset for Pluralistic Alignment of Text-to-Image Models","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U","json":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U.json","graph_json":"https://pith.science/api/pith-number/U45XPVT6YGMS3GFHN72TTOEN6U/graph.json","events_json":"https://pith.science/api/pith-number/U45XPVT6YGMS3GFHN72TTOEN6U/events.json","paper":"https://pith.science/paper/U45XPVT6"},"agent_actions":{"view_html":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U","download_json":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U.json","view_paper":"https://pith.science/paper/U45XPVT6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.02261&json=true","fetch_graph":"https://pith.science/api/pith-number/U45XPVT6YGMS3GFHN72TTOEN6U/graph.json","fetch_events":"https://pith.science/api/pith-number/U45XPVT6YGMS3GFHN72TTOEN6U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U/action/storage_attestation","attest_author":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U/action/author_attestation","sign_citation":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U/action/citation_signature","submit_replication":"https://pith.science/pith/U45XPVT6YGMS3GFHN72TTOEN6U/action/replication_record"}},"created_at":"2026-07-05T08:35:45.024201+00:00","updated_at":"2026-07-05T08:35:45.024201+00:00"}