{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5X3OYM6I5ISU3KVNC22RL75DY6","short_pith_number":"pith:5X3OYM6I","schema_version":"1.0","canonical_sha256":"edf6ec33c8ea254daaad16b515ffa3c7a184e94af85c0377cd82ba927815fc67","source":{"kind":"arxiv","id":"2310.09949","version":4},"attestation_state":"computed","paper":{"title":"Chameleon: a Heterogeneous and Disaggregated Accelerator System for Retrieval-Augmented Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.AR","cs.CL"],"primary_cat":"cs.LG","authors_text":"Gustavo Alonso, Marco Zeller, Roger Waleffe, Torsten Hoefler, Wenqi Jiang","submitted_at":"2023-10-15T20:57:25Z","abstract_excerpt":"A Retrieval-Augmented Language Model (RALM) combines a large language model (LLM) with a vector database to retrieve context-specific knowledge during text generation. This strategy facilitates impressive generation quality even with smaller models, thus reducing computational demands by orders of magnitude. To serve RALMs efficiently and flexibly, we propose Chameleon, a heterogeneous accelerator system integrating both LLM and vector search accelerators in a disaggregated architecture. The heterogeneity ensures efficient serving for both inference and retrieval, while the disaggregation allo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.09949","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-15T20:57:25Z","cross_cats_sorted":["cs.AI","cs.AR","cs.CL"],"title_canon_sha256":"b895bdf8a4b8c9b1b19a68bea9bf8e5493990493d3f098d2d97730c4bd28c148","abstract_canon_sha256":"b5c9612b5b2fe578c70dc53855fbea19036b0f6e2bddc76668b8f82a115c62e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:31.282252Z","signature_b64":"sXKL/M7C/fFpK568CdH5kGkRKi+Z0KJ4WUAwfz/znYlh1Vo7wxsmok1dFFvh+QLASsQ70rbNM+/+k/RA4jmHCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"edf6ec33c8ea254daaad16b515ffa3c7a184e94af85c0377cd82ba927815fc67","last_reissued_at":"2026-07-05T10:38:31.281850Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:31.281850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chameleon: a Heterogeneous and Disaggregated Accelerator System for Retrieval-Augmented Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.AR","cs.CL"],"primary_cat":"cs.LG","authors_text":"Gustavo Alonso, Marco Zeller, Roger Waleffe, Torsten Hoefler, Wenqi Jiang","submitted_at":"2023-10-15T20:57:25Z","abstract_excerpt":"A Retrieval-Augmented Language Model (RALM) combines a large language model (LLM) with a vector database to retrieve context-specific knowledge during text generation. This strategy facilitates impressive generation quality even with smaller models, thus reducing computational demands by orders of magnitude. To serve RALMs efficiently and flexibly, we propose Chameleon, a heterogeneous accelerator system integrating both LLM and vector search accelerators in a disaggregated architecture. The heterogeneity ensures efficient serving for both inference and retrieval, while the disaggregation allo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.09949","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.09949/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.09949","created_at":"2026-07-05T10:38:31.281907+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.09949v4","created_at":"2026-07-05T10:38:31.281907+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.09949","created_at":"2026-07-05T10:38:31.281907+00:00"},{"alias_kind":"pith_short_12","alias_value":"5X3OYM6I5ISU","created_at":"2026-07-05T10:38:31.281907+00:00"},{"alias_kind":"pith_short_16","alias_value":"5X3OYM6I5ISU3KVN","created_at":"2026-07-05T10:38:31.281907+00:00"},{"alias_kind":"pith_short_8","alias_value":"5X3OYM6I","created_at":"2026-07-05T10:38:31.281907+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.10981","citing_title":"A Survey on Retrieval-Augmented Text Generation for Large Language Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20948","citing_title":"Memory Grafting: Scaling Language Model Pre-training via Offline Conditional Memory","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10406","citing_title":"How to Build a Quantum Supercomputer: Scaling from Hundreds to Millions of Qubits","ref_index":165,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6","json":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6.json","graph_json":"https://pith.science/api/pith-number/5X3OYM6I5ISU3KVNC22RL75DY6/graph.json","events_json":"https://pith.science/api/pith-number/5X3OYM6I5ISU3KVNC22RL75DY6/events.json","paper":"https://pith.science/paper/5X3OYM6I"},"agent_actions":{"view_html":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6","download_json":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6.json","view_paper":"https://pith.science/paper/5X3OYM6I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.09949&json=true","fetch_graph":"https://pith.science/api/pith-number/5X3OYM6I5ISU3KVNC22RL75DY6/graph.json","fetch_events":"https://pith.science/api/pith-number/5X3OYM6I5ISU3KVNC22RL75DY6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6/action/storage_attestation","attest_author":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6/action/author_attestation","sign_citation":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6/action/citation_signature","submit_replication":"https://pith.science/pith/5X3OYM6I5ISU3KVNC22RL75DY6/action/replication_record"}},"created_at":"2026-07-05T10:38:31.281907+00:00","updated_at":"2026-07-05T10:38:31.281907+00:00"}