{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M72DD5BXSISPDD35I7ANGOELIM","short_pith_number":"pith:M72DD5BX","schema_version":"1.0","canonical_sha256":"67f431f4379224f18f7d47c0d3388b433ba54534f49fa724b6e62e75090f9ab5","source":{"kind":"arxiv","id":"2406.08413","version":1},"attestation_state":"computed","paper":{"title":"Memory Is All You Need: An Overview of Compute-in-Memory Architectures for Accelerating Large Language Model Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AR","authors_text":"Christopher Wolters, Toyotaro Suzumura, Ulf Schlichtmann, Xiaoxuan Yang","submitted_at":"2024-06-12T16:57:58Z","abstract_excerpt":"Large language models (LLMs) have recently transformed natural language processing, enabling machines to generate human-like text and engage in meaningful conversations. This development necessitates speed, efficiency, and accessibility in LLM inference as the computational and memory requirements of these systems grow exponentially. Meanwhile, advancements in computing and memory capabilities are lagging behind, exacerbated by the discontinuation of Moore's law. With LLMs exceeding the capacity of single GPUs, they require complex, expert-level configurations for parallel processing. Memory a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08413","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AR","submitted_at":"2024-06-12T16:57:58Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"52836bfb6ce70f9537298b4d928bf46cb1fd09b6e2b61f127ccdb3af2cbc5ad4","abstract_canon_sha256":"e0e42693c19e98b434dc4f4306f7131453b2e19e634fcc2fdb61f38e7b2d690b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:59.371223Z","signature_b64":"0Sm+qWtvBwwU4tfuW5Q9ZehuIDuEq4sCnidub02CwaxHHEKRcQ8TQ6rFMhSQbjKeuFiV3Hk4uZkb0qdw1p7bAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"67f431f4379224f18f7d47c0d3388b433ba54534f49fa724b6e62e75090f9ab5","last_reissued_at":"2026-07-05T08:30:59.370736Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:59.370736Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memory Is All You Need: An Overview of Compute-in-Memory Architectures for Accelerating Large Language Model Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AR","authors_text":"Christopher Wolters, Toyotaro Suzumura, Ulf Schlichtmann, Xiaoxuan Yang","submitted_at":"2024-06-12T16:57:58Z","abstract_excerpt":"Large language models (LLMs) have recently transformed natural language processing, enabling machines to generate human-like text and engage in meaningful conversations. This development necessitates speed, efficiency, and accessibility in LLM inference as the computational and memory requirements of these systems grow exponentially. Meanwhile, advancements in computing and memory capabilities are lagging behind, exacerbated by the discontinuation of Moore's law. With LLMs exceeding the capacity of single GPUs, they require complex, expert-level configurations for parallel processing. Memory a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08413","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08413/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08413","created_at":"2026-07-05T08:30:59.370794+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08413v1","created_at":"2026-07-05T08:30:59.370794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08413","created_at":"2026-07-05T08:30:59.370794+00:00"},{"alias_kind":"pith_short_12","alias_value":"M72DD5BXSISP","created_at":"2026-07-05T08:30:59.370794+00:00"},{"alias_kind":"pith_short_16","alias_value":"M72DD5BXSISPDD35","created_at":"2026-07-05T08:30:59.370794+00:00"},{"alias_kind":"pith_short_8","alias_value":"M72DD5BX","created_at":"2026-07-05T08:30:59.370794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08196","citing_title":"A First-Principles Theory of Slow Thinking and Active Perception","ref_index":172,"is_internal_anchor":true},{"citing_arxiv_id":"2606.00288","citing_title":"Model-Native Computing Architecture: Envisioning Future System Architecture Through the Lens of Computer Architecture","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26555","citing_title":"Chips in the Flatland : 2D Semiconductors for Future Computing Electronic","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08055","citing_title":"From Tokens to Layers: Redefining Stall-Free Scheduling for MoE Serving with Layered Prefill","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09590","citing_title":"DeepReviewer 2.0: A Traceable Agentic System for Auditable Scientific Peer Review","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26074","citing_title":"DAK: Direct-Access-Enabled GPU Memory Offloading with Optimal Efficiency for LLM Inference","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08637","citing_title":"Increased endurance of nonvolatile photonics enabled by nanostructured phase-change materials","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00544","citing_title":"An Unsupervised Machine Learning-based Framework for Wafer Scale Variability Analysis and Performance Prediction of Ferroelectric Hf0.5Zr0.5O2 Thin Film Capacitors","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM","json":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM.json","graph_json":"https://pith.science/api/pith-number/M72DD5BXSISPDD35I7ANGOELIM/graph.json","events_json":"https://pith.science/api/pith-number/M72DD5BXSISPDD35I7ANGOELIM/events.json","paper":"https://pith.science/paper/M72DD5BX"},"agent_actions":{"view_html":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM","download_json":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM.json","view_paper":"https://pith.science/paper/M72DD5BX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08413&json=true","fetch_graph":"https://pith.science/api/pith-number/M72DD5BXSISPDD35I7ANGOELIM/graph.json","fetch_events":"https://pith.science/api/pith-number/M72DD5BXSISPDD35I7ANGOELIM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM/action/storage_attestation","attest_author":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM/action/author_attestation","sign_citation":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM/action/citation_signature","submit_replication":"https://pith.science/pith/M72DD5BXSISPDD35I7ANGOELIM/action/replication_record"}},"created_at":"2026-07-05T08:30:59.370794+00:00","updated_at":"2026-07-05T08:30:59.370794+00:00"}