{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6PAVMOPZIFSPLLDJTVVK3ULJES","short_pith_number":"pith:6PAVMOPZ","schema_version":"1.0","canonical_sha256":"f3c15639f94164f5ac699d6aadd16924b745e22e7f948d6aa71eadd67011a50f","source":{"kind":"arxiv","id":"2501.03892","version":1},"attestation_state":"computed","paper":{"title":"LEAP: LLM-powered End-to-end Automatic Library for Processing Social Science Queries on Unstructured Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Austin Peters, Chuxuan Hu, Daniel Kang","submitted_at":"2025-01-07T16:00:40Z","abstract_excerpt":"Social scientists are increasingly interested in analyzing the semantic information (e.g., emotion) of unstructured data (e.g., Tweets), where the semantic information is not natively present. Performing this analysis in a cost-efficient manner requires using machine learning (ML) models to extract the semantic information and subsequently analyze the now structured data. However, this process remains challenging for domain experts.\n  To demonstrate the challenges in social science analytics, we collect a dataset, QUIET-ML, of 120 real-world social science queries in natural language and their"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03892","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DB","submitted_at":"2025-01-07T16:00:40Z","cross_cats_sorted":[],"title_canon_sha256":"41b656b768fbb1d31610dfbd87ceb55dfa57ffab614d1e1b0d7804ab90ffbc3c","abstract_canon_sha256":"726b1ec24bbf2224264df954924e8577d2715760adb0903b468178865b337682"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:02.844009Z","signature_b64":"NyB3CxGY8mOxJeeavTMebBHiTV21KJafkKyppUXQbcfugqQnucDUd2hagWNa0kWSmQos1a1g4/cA7gO7Pv2fDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3c15639f94164f5ac699d6aadd16924b745e22e7f948d6aa71eadd67011a50f","last_reissued_at":"2026-07-05T09:58:02.843562Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:02.843562Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LEAP: LLM-powered End-to-end Automatic Library for Processing Social Science Queries on Unstructured Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Austin Peters, Chuxuan Hu, Daniel Kang","submitted_at":"2025-01-07T16:00:40Z","abstract_excerpt":"Social scientists are increasingly interested in analyzing the semantic information (e.g., emotion) of unstructured data (e.g., Tweets), where the semantic information is not natively present. Performing this analysis in a cost-efficient manner requires using machine learning (ML) models to extract the semantic information and subsequently analyze the now structured data. However, this process remains challenging for domain experts.\n  To demonstrate the challenges in social science analytics, we collect a dataset, QUIET-ML, of 120 real-world social science queries in natural language and their"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03892","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03892/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03892","created_at":"2026-07-05T09:58:02.843625+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03892v1","created_at":"2026-07-05T09:58:02.843625+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03892","created_at":"2026-07-05T09:58:02.843625+00:00"},{"alias_kind":"pith_short_12","alias_value":"6PAVMOPZIFSP","created_at":"2026-07-05T09:58:02.843625+00:00"},{"alias_kind":"pith_short_16","alias_value":"6PAVMOPZIFSPLLDJ","created_at":"2026-07-05T09:58:02.843625+00:00"},{"alias_kind":"pith_short_8","alias_value":"6PAVMOPZ","created_at":"2026-07-05T09:58:02.843625+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.12610","citing_title":"ScaleDoc: Scaling LLM-based Predicates over Large Document Collections","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES","json":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES.json","graph_json":"https://pith.science/api/pith-number/6PAVMOPZIFSPLLDJTVVK3ULJES/graph.json","events_json":"https://pith.science/api/pith-number/6PAVMOPZIFSPLLDJTVVK3ULJES/events.json","paper":"https://pith.science/paper/6PAVMOPZ"},"agent_actions":{"view_html":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES","download_json":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES.json","view_paper":"https://pith.science/paper/6PAVMOPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03892&json=true","fetch_graph":"https://pith.science/api/pith-number/6PAVMOPZIFSPLLDJTVVK3ULJES/graph.json","fetch_events":"https://pith.science/api/pith-number/6PAVMOPZIFSPLLDJTVVK3ULJES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES/action/storage_attestation","attest_author":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES/action/author_attestation","sign_citation":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES/action/citation_signature","submit_replication":"https://pith.science/pith/6PAVMOPZIFSPLLDJTVVK3ULJES/action/replication_record"}},"created_at":"2026-07-05T09:58:02.843625+00:00","updated_at":"2026-07-05T09:58:02.843625+00:00"}