{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3LX4Q7U4HOK2RMA3DKFNFVVO6B","short_pith_number":"pith:3LX4Q7U4","schema_version":"1.0","canonical_sha256":"daefc87e9c3b95a8b01b1a8ad2d6aef050f38866de9b00b53ae7d229739a0665","source":{"kind":"arxiv","id":"2304.04576","version":1},"attestation_state":"computed","paper":{"title":"Learnings from Data Integration for Augmented Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alon Halevy, Jane Dwivedi-Yu","submitted_at":"2023-04-10T13:28:35Z","abstract_excerpt":"One of the limitations of large language models is that they do not have access to up-to-date, proprietary or personal data. As a result, there are multiple efforts to extend language models with techniques for accessing external data. In that sense, LLMs share the vision of data integration systems whose goal is to provide seamless access to a large collection of heterogeneous data sources. While the details and the techniques of LLMs differ greatly from those of data integration, this paper shows that some of the lessons learned from research on data integration can elucidate the research pa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.04576","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-10T13:28:35Z","cross_cats_sorted":[],"title_canon_sha256":"4e735deee764322dddc7785704518032518234330d9151822a3c2959cf7044ac","abstract_canon_sha256":"bc55a15e461f41338879e8bf33fae5a6822a4782286c36306da8f45fee6b7b61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:59:27.621856Z","signature_b64":"Tx/pbKS9uAi6wxaWcFBCmRzHilo1aRkkbgx0TyGC6uKWYmuMd+m7mnRB8XF856bast/RPlytncaxouJMIWTkAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"daefc87e9c3b95a8b01b1a8ad2d6aef050f38866de9b00b53ae7d229739a0665","last_reissued_at":"2026-07-05T05:59:27.621374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:59:27.621374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learnings from Data Integration for Augmented Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alon Halevy, Jane Dwivedi-Yu","submitted_at":"2023-04-10T13:28:35Z","abstract_excerpt":"One of the limitations of large language models is that they do not have access to up-to-date, proprietary or personal data. As a result, there are multiple efforts to extend language models with techniques for accessing external data. In that sense, LLMs share the vision of data integration systems whose goal is to provide seamless access to a large collection of heterogeneous data sources. While the details and the techniques of LLMs differ greatly from those of data integration, this paper shows that some of the lessons learned from research on data integration can elucidate the research pa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.04576","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.04576/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.04576","created_at":"2026-07-05T05:59:27.621441+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.04576v1","created_at":"2026-07-05T05:59:27.621441+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.04576","created_at":"2026-07-05T05:59:27.621441+00:00"},{"alias_kind":"pith_short_12","alias_value":"3LX4Q7U4HOK2","created_at":"2026-07-05T05:59:27.621441+00:00"},{"alias_kind":"pith_short_16","alias_value":"3LX4Q7U4HOK2RMA3","created_at":"2026-07-05T05:59:27.621441+00:00"},{"alias_kind":"pith_short_8","alias_value":"3LX4Q7U4","created_at":"2026-07-05T05:59:27.621441+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.06269","citing_title":"REMI: A Novel Causal Schema Memory Architecture for Personalized Lifestyle Recommendation Agents","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B","json":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B.json","graph_json":"https://pith.science/api/pith-number/3LX4Q7U4HOK2RMA3DKFNFVVO6B/graph.json","events_json":"https://pith.science/api/pith-number/3LX4Q7U4HOK2RMA3DKFNFVVO6B/events.json","paper":"https://pith.science/paper/3LX4Q7U4"},"agent_actions":{"view_html":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B","download_json":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B.json","view_paper":"https://pith.science/paper/3LX4Q7U4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.04576&json=true","fetch_graph":"https://pith.science/api/pith-number/3LX4Q7U4HOK2RMA3DKFNFVVO6B/graph.json","fetch_events":"https://pith.science/api/pith-number/3LX4Q7U4HOK2RMA3DKFNFVVO6B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B/action/storage_attestation","attest_author":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B/action/author_attestation","sign_citation":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B/action/citation_signature","submit_replication":"https://pith.science/pith/3LX4Q7U4HOK2RMA3DKFNFVVO6B/action/replication_record"}},"created_at":"2026-07-05T05:59:27.621441+00:00","updated_at":"2026-07-05T05:59:27.621441+00:00"}