{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2HHPWRQASU5D2P4NCOEMCJ5DMP","short_pith_number":"pith:2HHPWRQA","schema_version":"1.0","canonical_sha256":"d1cefb4600953a3d3f8d1388c127a363e595aa6327ad5417056302aa52f61341","source":{"kind":"arxiv","id":"2310.20111","version":1},"attestation_state":"computed","paper":{"title":"Making Large Language Models Better Data Creators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dong-Ho Lee, Jay Pujara, Mohit Sewak, Ryen W. White, Sujay Kumar Jauhar","submitted_at":"2023-10-31T01:08:34Z","abstract_excerpt":"Although large language models (LLMs) have advanced the state-of-the-art in NLP significantly, deploying them for downstream applications is still challenging due to cost, responsiveness, control, or concerns around privacy and security. As such, trainable models are still the preferred option in some cases. However, these models still require human-labeled data for optimal performance, which is expensive and time-consuming to obtain. In order to address this issue, several techniques to reduce human effort involve labeling or generating data using LLMs. Although these methods are effective fo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.20111","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-31T01:08:34Z","cross_cats_sorted":[],"title_canon_sha256":"9279feaf735a8ac8515a748b50797e3d6ca8e2311ef4c6b634d5bd7ab32c1d17","abstract_canon_sha256":"c8f91745a1764ad3a99e9a63d9ab6744672f1a32eaab63da3e60a7f480e63f52"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:07:24.187711Z","signature_b64":"V/n1ThQ77NUtt/Aeq7amHtq9KZRFSzNHSxHxqqek6ZlmFoG6MRZsKjI0C5ZpdVvzo8H082s3RmfS/+kYvHSYDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1cefb4600953a3d3f8d1388c127a363e595aa6327ad5417056302aa52f61341","last_reissued_at":"2026-07-05T07:07:24.187223Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:07:24.187223Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Making Large Language Models Better Data Creators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dong-Ho Lee, Jay Pujara, Mohit Sewak, Ryen W. White, Sujay Kumar Jauhar","submitted_at":"2023-10-31T01:08:34Z","abstract_excerpt":"Although large language models (LLMs) have advanced the state-of-the-art in NLP significantly, deploying them for downstream applications is still challenging due to cost, responsiveness, control, or concerns around privacy and security. As such, trainable models are still the preferred option in some cases. However, these models still require human-labeled data for optimal performance, which is expensive and time-consuming to obtain. In order to address this issue, several techniques to reduce human effort involve labeling or generating data using LLMs. Although these methods are effective fo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.20111","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.20111/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.20111","created_at":"2026-07-05T07:07:24.187280+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.20111v1","created_at":"2026-07-05T07:07:24.187280+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.20111","created_at":"2026-07-05T07:07:24.187280+00:00"},{"alias_kind":"pith_short_12","alias_value":"2HHPWRQASU5D","created_at":"2026-07-05T07:07:24.187280+00:00"},{"alias_kind":"pith_short_16","alias_value":"2HHPWRQASU5D2P4N","created_at":"2026-07-05T07:07:24.187280+00:00"},{"alias_kind":"pith_short_8","alias_value":"2HHPWRQA","created_at":"2026-07-05T07:07:24.187280+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.18252","citing_title":"Multimodal Behavioral Patterns Analysis with Eye-Tracking and LLM-Based Reasoning","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP","json":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP.json","graph_json":"https://pith.science/api/pith-number/2HHPWRQASU5D2P4NCOEMCJ5DMP/graph.json","events_json":"https://pith.science/api/pith-number/2HHPWRQASU5D2P4NCOEMCJ5DMP/events.json","paper":"https://pith.science/paper/2HHPWRQA"},"agent_actions":{"view_html":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP","download_json":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP.json","view_paper":"https://pith.science/paper/2HHPWRQA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.20111&json=true","fetch_graph":"https://pith.science/api/pith-number/2HHPWRQASU5D2P4NCOEMCJ5DMP/graph.json","fetch_events":"https://pith.science/api/pith-number/2HHPWRQASU5D2P4NCOEMCJ5DMP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP/action/storage_attestation","attest_author":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP/action/author_attestation","sign_citation":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP/action/citation_signature","submit_replication":"https://pith.science/pith/2HHPWRQASU5D2P4NCOEMCJ5DMP/action/replication_record"}},"created_at":"2026-07-05T07:07:24.187280+00:00","updated_at":"2026-07-05T07:07:24.187280+00:00"}