{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3WVIZI3ZDSWN2JHG4EDTGSC7GL","short_pith_number":"pith:3WVIZI3Z","schema_version":"1.0","canonical_sha256":"ddaa8ca3791cacdd24e6e10733485f32fe373d283183b618e82fe671b41c0c56","source":{"kind":"arxiv","id":"2403.02419","version":2},"attestation_state":"computed","paper":{"title":"Are More LLM Calls All You Need? Towards Scaling Laws of Compound Inference Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Boris Hanin, Ion Stoica, James Zou, Jared Quincy Davis, Lingjiao Chen, Matei Zaharia, Peter Bailis","submitted_at":"2024-03-04T19:12:48Z","abstract_excerpt":"Many recent state-of-the-art results in language tasks were achieved using compound systems that perform multiple Language Model (LM) calls and aggregate their responses. However, there is little understanding of how the number of LM calls - e.g., when asking the LM to answer each question multiple times and taking a majority vote - affects such a compound system's performance. In this paper, we initiate the study of scaling properties of compound inference systems. We analyze, theoretically and empirically, how the number of LM calls affects the performance of Vote and Filter-Vote, two of the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.02419","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-04T19:12:48Z","cross_cats_sorted":["cs.AI","cs.CL","cs.SY","eess.SY"],"title_canon_sha256":"c3d21d021564ce52d87bce4cc39bdb5ddb6a181ca94530e0772cf0a0b690f91c","abstract_canon_sha256":"36d58911efc451b91f45947f507a10d65f9e099c664baa4895b1afde2baa66e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:25.260610Z","signature_b64":"s5VXSx4FYNOFT5F4K4hT/ppWUsE3aS2+wtxCMIryeG6Nprm0TE4Oxu4g8xAV+e1cJ4h6DO2d/yszUs8afJRrAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddaa8ca3791cacdd24e6e10733485f32fe373d283183b618e82fe671b41c0c56","last_reissued_at":"2026-07-05T08:27:25.260204Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:25.260204Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are More LLM Calls All You Need? Towards Scaling Laws of Compound Inference Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Boris Hanin, Ion Stoica, James Zou, Jared Quincy Davis, Lingjiao Chen, Matei Zaharia, Peter Bailis","submitted_at":"2024-03-04T19:12:48Z","abstract_excerpt":"Many recent state-of-the-art results in language tasks were achieved using compound systems that perform multiple Language Model (LM) calls and aggregate their responses. However, there is little understanding of how the number of LM calls - e.g., when asking the LM to answer each question multiple times and taking a majority vote - affects such a compound system's performance. In this paper, we initiate the study of scaling properties of compound inference systems. We analyze, theoretically and empirically, how the number of LM calls affects the performance of Vote and Filter-Vote, two of the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.02419","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.02419/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.02419","created_at":"2026-07-05T08:27:25.260269+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.02419v2","created_at":"2026-07-05T08:27:25.260269+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.02419","created_at":"2026-07-05T08:27:25.260269+00:00"},{"alias_kind":"pith_short_12","alias_value":"3WVIZI3ZDSWN","created_at":"2026-07-05T08:27:25.260269+00:00"},{"alias_kind":"pith_short_16","alias_value":"3WVIZI3ZDSWN2JHG","created_at":"2026-07-05T08:27:25.260269+00:00"},{"alias_kind":"pith_short_8","alias_value":"3WVIZI3Z","created_at":"2026-07-05T08:27:25.260269+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28661","citing_title":"When More Sampling Hurts: The Modal Ceiling and Correlation Ceiling of Test-Time Scaling","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":161,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14200","citing_title":"A Scalable Multi-LLM Collaboration System with Retrieval-based Selection and Exploration-Exploitation-Driven Enhancement","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2403.12031","citing_title":"RouterBench: A Benchmark for Multi-LLM Routing System","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00883","citing_title":"Knowledge without Wisdom: Measuring Misalignment between LLMs and Intended Impact","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03295","citing_title":"Scaling Teams or Scaling Time? Memory Enabled Lifelong Learning in LLM Multi-Agent Systems","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2503.13657","citing_title":"Why Do Multi-Agent LLM Systems Fail?","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2407.21787","citing_title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24110","citing_title":"Latency and Cost of Multi-Agent Intelligent Tutoring at Scale","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15233","citing_title":"Blue Data Intelligence Layer: Streaming Data and Agents for Multi-source Multi-modal Data-Centric Applications","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL","json":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL.json","graph_json":"https://pith.science/api/pith-number/3WVIZI3ZDSWN2JHG4EDTGSC7GL/graph.json","events_json":"https://pith.science/api/pith-number/3WVIZI3ZDSWN2JHG4EDTGSC7GL/events.json","paper":"https://pith.science/paper/3WVIZI3Z"},"agent_actions":{"view_html":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL","download_json":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL.json","view_paper":"https://pith.science/paper/3WVIZI3Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.02419&json=true","fetch_graph":"https://pith.science/api/pith-number/3WVIZI3ZDSWN2JHG4EDTGSC7GL/graph.json","fetch_events":"https://pith.science/api/pith-number/3WVIZI3ZDSWN2JHG4EDTGSC7GL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL/action/storage_attestation","attest_author":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL/action/author_attestation","sign_citation":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL/action/citation_signature","submit_replication":"https://pith.science/pith/3WVIZI3ZDSWN2JHG4EDTGSC7GL/action/replication_record"}},"created_at":"2026-07-05T08:27:25.260269+00:00","updated_at":"2026-07-05T08:27:25.260269+00:00"}