{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JXEUCXGSRC7UDXIFE7OKK7ZO2U","short_pith_number":"pith:JXEUCXGS","schema_version":"1.0","canonical_sha256":"4dc9415cd288bf41dd0527dca57f2ed5362ce0c7cba491fd3c2e59d096ba0769","source":{"kind":"arxiv","id":"2503.00725","version":1},"attestation_state":"computed","paper":{"title":"Causal Inference on Outcomes Learned from Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","stat.ME"],"primary_cat":"econ.EM","authors_text":"Amar Venugopal, Iman Modarressi, Jann Spiess","submitted_at":"2025-03-02T04:36:27Z","abstract_excerpt":"We propose a machine-learning tool that yields causal inference on text in randomized trials. Based on a simple econometric framework in which text may capture outcomes of interest, our procedure addresses three questions: First, is the text affected by the treatment? Second, which outcomes is the effect on? And third, how complete is our description of causal effects? To answer all three questions, our approach uses large language models (LLMs) that suggest systematic differences across two groups of text documents and then provides valid inference based on costly validation. Specifically, we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00725","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"econ.EM","submitted_at":"2025-03-02T04:36:27Z","cross_cats_sorted":["cs.CL","cs.LG","stat.ME"],"title_canon_sha256":"aae79bbe9029f24d8fbb6eff523a68ecca3594a5b2a17e3aad694ad4e275a8da","abstract_canon_sha256":"74cd29a5adffd4d8bf6bc3afdb33dc25fde5efd06a4d67c37f8f7349410f91f4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:28.181792Z","signature_b64":"dbK53E3Gt7zI/4JapK27Jt2urMQ4J0CGel7Pb01PyIqxNGKiU8kyezgb4UCGlVftoWeNJ5bzfwEnD6caPVWFDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4dc9415cd288bf41dd0527dca57f2ed5362ce0c7cba491fd3c2e59d096ba0769","last_reissued_at":"2026-07-05T10:22:28.181225Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:28.181225Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Causal Inference on Outcomes Learned from Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","stat.ME"],"primary_cat":"econ.EM","authors_text":"Amar Venugopal, Iman Modarressi, Jann Spiess","submitted_at":"2025-03-02T04:36:27Z","abstract_excerpt":"We propose a machine-learning tool that yields causal inference on text in randomized trials. Based on a simple econometric framework in which text may capture outcomes of interest, our procedure addresses three questions: First, is the text affected by the treatment? Second, which outcomes is the effect on? And third, how complete is our description of causal effects? To answer all three questions, our approach uses large language models (LLMs) that suggest systematic differences across two groups of text documents and then provides valid inference based on costly validation. Specifically, we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00725","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00725/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00725","created_at":"2026-07-05T10:22:28.181286+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00725v1","created_at":"2026-07-05T10:22:28.181286+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00725","created_at":"2026-07-05T10:22:28.181286+00:00"},{"alias_kind":"pith_short_12","alias_value":"JXEUCXGSRC7U","created_at":"2026-07-05T10:22:28.181286+00:00"},{"alias_kind":"pith_short_16","alias_value":"JXEUCXGSRC7UDXIF","created_at":"2026-07-05T10:22:28.181286+00:00"},{"alias_kind":"pith_short_8","alias_value":"JXEUCXGS","created_at":"2026-07-05T10:22:28.181286+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03029","citing_title":"Conditional Hypothesis Generation for LLM-Based Text Analysis with Researcher-Specified Covariates","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01680","citing_title":"Making Interpretable Discoveries from Unstructured Data: A High-Dimensional Multiple Hypothesis Testing Approach","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U","json":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U.json","graph_json":"https://pith.science/api/pith-number/JXEUCXGSRC7UDXIFE7OKK7ZO2U/graph.json","events_json":"https://pith.science/api/pith-number/JXEUCXGSRC7UDXIFE7OKK7ZO2U/events.json","paper":"https://pith.science/paper/JXEUCXGS"},"agent_actions":{"view_html":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U","download_json":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U.json","view_paper":"https://pith.science/paper/JXEUCXGS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00725&json=true","fetch_graph":"https://pith.science/api/pith-number/JXEUCXGSRC7UDXIFE7OKK7ZO2U/graph.json","fetch_events":"https://pith.science/api/pith-number/JXEUCXGSRC7UDXIFE7OKK7ZO2U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U/action/storage_attestation","attest_author":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U/action/author_attestation","sign_citation":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U/action/citation_signature","submit_replication":"https://pith.science/pith/JXEUCXGSRC7UDXIFE7OKK7ZO2U/action/replication_record"}},"created_at":"2026-07-05T10:22:28.181286+00:00","updated_at":"2026-07-05T10:22:28.181286+00:00"}