{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PRJDSKCCS2XHMYVCRQQ2MFDQUZ","short_pith_number":"pith:PRJDSKCC","schema_version":"1.0","canonical_sha256":"7c5239284296ae7662a28c21a61470a64729f9be2c52ce546069bdca486d4b90","source":{"kind":"arxiv","id":"2504.07096","version":2},"attestation_state":"computed","paper":{"title":"OLMoTrace: Tracing Language Model Outputs Back to Trillions of Training Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aaron Sarnat, Ali Farhadi, Arnavi Chheda-Kothary, Bailey Kuehl, Byron Bischoff, Carissa Schoenick, Cassidy Trier, David Albright, Dirk Groeneveld, Eric Marsh, Evie Cheng, Hannaneh Hajishirzi, Huy Tran, Jenna James, Jesse Dodge, Jiacheng Liu, Jon Borchardt, Karen Farley, Luca Soldaini, Michael Schmitz, Noah A. Smith, Pang Wei Koh, Rock Yuren Pang, Sewon Min, Sophie Lebrecht, Sruthi Sreeram, Taira Anderson, Taylor Blanton, Yanai Elazar, Yejin Choi, YenSung Chen","submitted_at":"2025-04-09T17:59:35Z","abstract_excerpt":"We present OLMoTrace, the first system that traces the outputs of language models back to their full, multi-trillion-token training data in real time. OLMoTrace finds and shows verbatim matches between segments of language model output and documents in the training text corpora. Powered by an extended version of infini-gram (Liu et al., 2024), our system returns tracing results within a few seconds. OLMoTrace can help users understand the behavior of language models through the lens of their training data. We showcase how it can be used to explore fact checking, hallucination, and the creativi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07096","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-09T17:59:35Z","cross_cats_sorted":[],"title_canon_sha256":"530cd6676333d4975119523409ebd8fe4d0998a98d8a25476284a29205aee78b","abstract_canon_sha256":"d808d0c06a8d5b100736a6f31f783bc77ea61b7183cc38245b8c811399ef7f8f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:22.788294Z","signature_b64":"2ZIUhL8YefczwriL50l2hPrKx7QK+nB70wCWfssXnSQYOY1VRui6/LFCpElkMINvIJ5CEQJMEBF5nAR/U1uzAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c5239284296ae7662a28c21a61470a64729f9be2c52ce546069bdca486d4b90","last_reissued_at":"2026-07-05T11:33:22.787695Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:22.787695Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OLMoTrace: Tracing Language Model Outputs Back to Trillions of Training Tokens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aaron Sarnat, Ali Farhadi, Arnavi Chheda-Kothary, Bailey Kuehl, Byron Bischoff, Carissa Schoenick, Cassidy Trier, David Albright, Dirk Groeneveld, Eric Marsh, Evie Cheng, Hannaneh Hajishirzi, Huy Tran, Jenna James, Jesse Dodge, Jiacheng Liu, Jon Borchardt, Karen Farley, Luca Soldaini, Michael Schmitz, Noah A. Smith, Pang Wei Koh, Rock Yuren Pang, Sewon Min, Sophie Lebrecht, Sruthi Sreeram, Taira Anderson, Taylor Blanton, Yanai Elazar, Yejin Choi, YenSung Chen","submitted_at":"2025-04-09T17:59:35Z","abstract_excerpt":"We present OLMoTrace, the first system that traces the outputs of language models back to their full, multi-trillion-token training data in real time. OLMoTrace finds and shows verbatim matches between segments of language model output and documents in the training text corpora. Powered by an extended version of infini-gram (Liu et al., 2024), our system returns tracing results within a few seconds. OLMoTrace can help users understand the behavior of language models through the lens of their training data. We showcase how it can be used to explore fact checking, hallucination, and the creativi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07096","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07096/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07096","created_at":"2026-07-05T11:33:22.787765+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07096v2","created_at":"2026-07-05T11:33:22.787765+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07096","created_at":"2026-07-05T11:33:22.787765+00:00"},{"alias_kind":"pith_short_12","alias_value":"PRJDSKCCS2XH","created_at":"2026-07-05T11:33:22.787765+00:00"},{"alias_kind":"pith_short_16","alias_value":"PRJDSKCCS2XHMYVC","created_at":"2026-07-05T11:33:22.787765+00:00"},{"alias_kind":"pith_short_8","alias_value":"PRJDSKCC","created_at":"2026-07-05T11:33:22.787765+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18285","citing_title":"RELIANCE: Curating and Evaluating Reproductive Health Information on Social Media","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06286","citing_title":"LLMs Can Leak Training Data But Do They Want To? A Propensity-Aware Evaluation of Memorization in LLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28510","citing_title":"Efficient and Scalable Provenance Tracking for LLM-Generated Code Snippets","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05687","citing_title":"DataDignity: Training Data Attribution for Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05341","citing_title":"Feature Starvation as Geometric Instability in Sparse Autoencoders","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ","json":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ.json","graph_json":"https://pith.science/api/pith-number/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/graph.json","events_json":"https://pith.science/api/pith-number/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/events.json","paper":"https://pith.science/paper/PRJDSKCC"},"agent_actions":{"view_html":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ","download_json":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ.json","view_paper":"https://pith.science/paper/PRJDSKCC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07096&json=true","fetch_graph":"https://pith.science/api/pith-number/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/graph.json","fetch_events":"https://pith.science/api/pith-number/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/action/storage_attestation","attest_author":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/action/author_attestation","sign_citation":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/action/citation_signature","submit_replication":"https://pith.science/pith/PRJDSKCCS2XHMYVCRQQ2MFDQUZ/action/replication_record"}},"created_at":"2026-07-05T11:33:22.787765+00:00","updated_at":"2026-07-05T11:33:22.787765+00:00"}