{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SQ77XD4AQ5LQ2AOT5ARKHTPSYY","short_pith_number":"pith:SQ77XD4A","schema_version":"1.0","canonical_sha256":"943ffb8f8087570d01d3e822a3cdf2c60424b9962cc0f74652b6ae0747d5d570","source":{"kind":"arxiv","id":"2405.14696","version":2},"attestation_state":"computed","paper":{"title":"A Declarative System for Optimizing AI Workloads","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DB"],"primary_cat":"cs.CL","authors_text":"Chunwei Liu, Gerardo Vitagliano, Lei Cao, Matthew Russo, Michael Cafarella, Michael Franklin, Peter Baille Chen, Samuel Madden, Tim Kraska, Zui Chen","submitted_at":"2024-05-23T15:31:18Z","abstract_excerpt":"A long-standing goal of data management systems has been to build systems which can compute quantitative insights over large corpora of unstructured data in a cost-effective manner. Until recently, it was difficult and expensive to extract facts from company documents, data from scientific papers, or metrics from image and video corpora. Today's models can accomplish these tasks with high accuracy. However, a programmer who wants to answer a substantive AI-powered query must orchestrate large numbers of models, prompts, and data operations. For even a single query, the programmer has to make a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14696","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-23T15:31:18Z","cross_cats_sorted":["cs.AI","cs.DB"],"title_canon_sha256":"a9a8e9d8db5bb60e8707df2aae5206312146539e8103b2b30bcb8ef0a744290d","abstract_canon_sha256":"506b001c5024ceecbdf937dbd42902523a5e944386f31abec1e3fb1c2913e77d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:43.257912Z","signature_b64":"p4EMJmNKTjp2OkG+AmspK1sXIZ6L2k0YmK4feZgRocYsizmyoAjmzG4/AkiEuAyZSvTlButcz42ExI3Sg0lbDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"943ffb8f8087570d01d3e822a3cdf2c60424b9962cc0f74652b6ae0747d5d570","last_reissued_at":"2026-07-05T08:24:43.257438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:43.257438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Declarative System for Optimizing AI Workloads","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DB"],"primary_cat":"cs.CL","authors_text":"Chunwei Liu, Gerardo Vitagliano, Lei Cao, Matthew Russo, Michael Cafarella, Michael Franklin, Peter Baille Chen, Samuel Madden, Tim Kraska, Zui Chen","submitted_at":"2024-05-23T15:31:18Z","abstract_excerpt":"A long-standing goal of data management systems has been to build systems which can compute quantitative insights over large corpora of unstructured data in a cost-effective manner. Until recently, it was difficult and expensive to extract facts from company documents, data from scientific papers, or metrics from image and video corpora. Today's models can accomplish these tasks with high accuracy. However, a programmer who wants to answer a substantive AI-powered query must orchestrate large numbers of models, prompts, and data operations. For even a single query, the programmer has to make a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14696","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14696","created_at":"2026-07-05T08:24:43.257494+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14696v2","created_at":"2026-07-05T08:24:43.257494+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14696","created_at":"2026-07-05T08:24:43.257494+00:00"},{"alias_kind":"pith_short_12","alias_value":"SQ77XD4AQ5LQ","created_at":"2026-07-05T08:24:43.257494+00:00"},{"alias_kind":"pith_short_16","alias_value":"SQ77XD4AQ5LQ2AOT","created_at":"2026-07-05T08:24:43.257494+00:00"},{"alias_kind":"pith_short_8","alias_value":"SQ77XD4A","created_at":"2026-07-05T08:24:43.257494+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08177","citing_title":"ASMR: Agentic Schema Generation for Ship Maintenance Report Writing","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20690","citing_title":"Declarative Data Services: Structured Agentic Discovery for Composing Data Systems","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29532","citing_title":"SemJoin: Semantic Join Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2503.04338","citing_title":"In-depth Analysis of Graph-based RAG in a Unified Framework","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12610","citing_title":"ScaleDoc: Scaling LLM-based Predicates over Large Document Collections","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20690","citing_title":"Declarative Data Services: Structured Agentic Discovery for Composing Data Systems","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2506.04565","citing_title":"From Standalone LLMs to Integrated Intelligence: A Survey of Compound Al Systems","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23477","citing_title":"SEMA-SQL: Beyond Traditional Relational Querying with Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01707","citing_title":"Memory in the LLM Era: Modular Architectures and Strategies in a Unified Framework","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02655","citing_title":"Semantic Data Processing with Holistic Data Understanding","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23477","citing_title":"SEMA-SQL: Beyond Traditional Relational Querying with Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15233","citing_title":"Blue Data Intelligence Layer: Streaming Data and Agents for Multi-source Multi-modal Data-Centric Applications","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY","json":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY.json","graph_json":"https://pith.science/api/pith-number/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/graph.json","events_json":"https://pith.science/api/pith-number/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/events.json","paper":"https://pith.science/paper/SQ77XD4A"},"agent_actions":{"view_html":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY","download_json":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY.json","view_paper":"https://pith.science/paper/SQ77XD4A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14696&json=true","fetch_graph":"https://pith.science/api/pith-number/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/graph.json","fetch_events":"https://pith.science/api/pith-number/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/action/storage_attestation","attest_author":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/action/author_attestation","sign_citation":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/action/citation_signature","submit_replication":"https://pith.science/pith/SQ77XD4AQ5LQ2AOT5ARKHTPSYY/action/replication_record"}},"created_at":"2026-07-05T08:24:43.257494+00:00","updated_at":"2026-07-05T08:24:43.257494+00:00"}