{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5X3QGFQLN5NJJ3I2BK7AXXIF6V","short_pith_number":"pith:5X3QGFQL","schema_version":"1.0","canonical_sha256":"edf703160b6f5a94ed1a0abe0bdd05f555d265a9c28e2e59cf1b6f17e36ea7e9","source":{"kind":"arxiv","id":"2412.16089","version":1},"attestation_state":"computed","paper":{"title":"The Evolution of LLM Adoption in Industry Data Curation Practices","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"Carrie J. Cai, Crystal Qian, Emily Reif, Grady Simon, James Wexler, Michael Terry, Michael Xieyang Liu, Minsuk Kahng, Nada Hussein, Nathan Clement","submitted_at":"2024-12-20T17:34:16Z","abstract_excerpt":"As large language models (LLMs) grow increasingly adept at processing unstructured text data, they offer new opportunities to enhance data curation workflows. This paper explores the evolution of LLM adoption among practitioners at a large technology company, evaluating the impact of LLMs in data curation tasks through participants' perceptions, integration strategies, and reported usage scenarios. Through a series of surveys, interviews, and user studies, we provide a timely snapshot of how organizations are navigating a pivotal moment in LLM evolution. In Q2 2023, we conducted a survey to as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16089","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.HC","submitted_at":"2024-12-20T17:34:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3eba270e93393c1c748465580fcc01e916806ab3104ccc191d3d85ba510e65ea","abstract_canon_sha256":"5c2b38d0ebe44eb30a22c89a5124d4e45ecb1d5781186ac551c68cbbb1b83d2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:34.786598Z","signature_b64":"0DqdV020sph6R1jmQVSzo6JT5syn5p3JUXOJPQl3Y/u0tN7Vmwd0A4Dn1UiOfm069Le/b7RP8KSDfWdvAhu4AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"edf703160b6f5a94ed1a0abe0bdd05f555d265a9c28e2e59cf1b6f17e36ea7e9","last_reissued_at":"2026-07-05T09:52:34.786095Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:34.786095Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Evolution of LLM Adoption in Industry Data Curation Practices","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"Carrie J. Cai, Crystal Qian, Emily Reif, Grady Simon, James Wexler, Michael Terry, Michael Xieyang Liu, Minsuk Kahng, Nada Hussein, Nathan Clement","submitted_at":"2024-12-20T17:34:16Z","abstract_excerpt":"As large language models (LLMs) grow increasingly adept at processing unstructured text data, they offer new opportunities to enhance data curation workflows. This paper explores the evolution of LLM adoption among practitioners at a large technology company, evaluating the impact of LLMs in data curation tasks through participants' perceptions, integration strategies, and reported usage scenarios. Through a series of surveys, interviews, and user studies, we provide a timely snapshot of how organizations are navigating a pivotal moment in LLM evolution. In Q2 2023, we conducted a survey to as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16089","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16089/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16089","created_at":"2026-07-05T09:52:34.786159+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16089v1","created_at":"2026-07-05T09:52:34.786159+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16089","created_at":"2026-07-05T09:52:34.786159+00:00"},{"alias_kind":"pith_short_12","alias_value":"5X3QGFQLN5NJ","created_at":"2026-07-05T09:52:34.786159+00:00"},{"alias_kind":"pith_short_16","alias_value":"5X3QGFQLN5NJJ3I2","created_at":"2026-07-05T09:52:34.786159+00:00"},{"alias_kind":"pith_short_8","alias_value":"5X3QGFQL","created_at":"2026-07-05T09:52:34.786159+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.11848","citing_title":"Compass vs Railway Tracks: Unpacking User Mental Models for Communicating Long-Horizon Work to Humans vs. AI","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V","json":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V.json","graph_json":"https://pith.science/api/pith-number/5X3QGFQLN5NJJ3I2BK7AXXIF6V/graph.json","events_json":"https://pith.science/api/pith-number/5X3QGFQLN5NJJ3I2BK7AXXIF6V/events.json","paper":"https://pith.science/paper/5X3QGFQL"},"agent_actions":{"view_html":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V","download_json":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V.json","view_paper":"https://pith.science/paper/5X3QGFQL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16089&json=true","fetch_graph":"https://pith.science/api/pith-number/5X3QGFQLN5NJJ3I2BK7AXXIF6V/graph.json","fetch_events":"https://pith.science/api/pith-number/5X3QGFQLN5NJJ3I2BK7AXXIF6V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V/action/storage_attestation","attest_author":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V/action/author_attestation","sign_citation":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V/action/citation_signature","submit_replication":"https://pith.science/pith/5X3QGFQLN5NJJ3I2BK7AXXIF6V/action/replication_record"}},"created_at":"2026-07-05T09:52:34.786159+00:00","updated_at":"2026-07-05T09:52:34.786159+00:00"}