{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XCZOCFWYJCC5BGN6AA2U2C7SAJ","short_pith_number":"pith:XCZOCFWY","schema_version":"1.0","canonical_sha256":"b8b2e116d84885d099be00354d0bf2024e685fca68e699e9d2dd23269c201f93","source":{"kind":"arxiv","id":"2304.12512","version":1},"attestation_state":"computed","paper":{"title":"Semantic Compression With Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Douglas C. Schmidt, Henry Gilbert, Jesse Spencer-Smith, Jules White, Michael Sandborn","submitted_at":"2023-04-25T01:47:05Z","abstract_excerpt":"The rise of large language models (LLMs) is revolutionizing information retrieval, question answering, summarization, and code generation tasks. However, in addition to confidently presenting factually inaccurate information at times (known as \"hallucinations\"), LLMs are also inherently limited by the number of input and output tokens that can be processed at once, making them potentially less effective on tasks that require processing a large set or continuous stream of information. A common approach to reducing the size of data is through lossless or lossy compression. Yet, in some cases it "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.12512","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-04-25T01:47:05Z","cross_cats_sorted":[],"title_canon_sha256":"21e3aa2d29fcb3028725bf80f3f86ae6a2d653479c2e671d445b11d42eb03bb0","abstract_canon_sha256":"65ff0676c5e4548e9e6aad87e0b5fca40e09d2eb4d46aec5bf8300e7c5cfd00b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:04:17.728085Z","signature_b64":"78x0wQ5i0kYmZ0K9rMzznAIeAYb9ihcku9Sw4gYjjf7aiZxyErBUoxsJeZjris7kNuuPNRcte/6iONl6lfRxCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8b2e116d84885d099be00354d0bf2024e685fca68e699e9d2dd23269c201f93","last_reissued_at":"2026-07-05T06:04:17.727705Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:04:17.727705Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Semantic Compression With Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Douglas C. Schmidt, Henry Gilbert, Jesse Spencer-Smith, Jules White, Michael Sandborn","submitted_at":"2023-04-25T01:47:05Z","abstract_excerpt":"The rise of large language models (LLMs) is revolutionizing information retrieval, question answering, summarization, and code generation tasks. However, in addition to confidently presenting factually inaccurate information at times (known as \"hallucinations\"), LLMs are also inherently limited by the number of input and output tokens that can be processed at once, making them potentially less effective on tasks that require processing a large set or continuous stream of information. A common approach to reducing the size of data is through lossless or lossy compression. Yet, in some cases it "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.12512","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.12512/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.12512","created_at":"2026-07-05T06:04:17.727758+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.12512v1","created_at":"2026-07-05T06:04:17.727758+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.12512","created_at":"2026-07-05T06:04:17.727758+00:00"},{"alias_kind":"pith_short_12","alias_value":"XCZOCFWYJCC5","created_at":"2026-07-05T06:04:17.727758+00:00"},{"alias_kind":"pith_short_16","alias_value":"XCZOCFWYJCC5BGN6","created_at":"2026-07-05T06:04:17.727758+00:00"},{"alias_kind":"pith_short_8","alias_value":"XCZOCFWY","created_at":"2026-07-05T06:04:17.727758+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24541","citing_title":"SemanticZip: A Pilot Framework for Lossy Text Compression with LLMs as Semantic Decompressors","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ","json":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ.json","graph_json":"https://pith.science/api/pith-number/XCZOCFWYJCC5BGN6AA2U2C7SAJ/graph.json","events_json":"https://pith.science/api/pith-number/XCZOCFWYJCC5BGN6AA2U2C7SAJ/events.json","paper":"https://pith.science/paper/XCZOCFWY"},"agent_actions":{"view_html":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ","download_json":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ.json","view_paper":"https://pith.science/paper/XCZOCFWY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.12512&json=true","fetch_graph":"https://pith.science/api/pith-number/XCZOCFWYJCC5BGN6AA2U2C7SAJ/graph.json","fetch_events":"https://pith.science/api/pith-number/XCZOCFWYJCC5BGN6AA2U2C7SAJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ/action/storage_attestation","attest_author":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ/action/author_attestation","sign_citation":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ/action/citation_signature","submit_replication":"https://pith.science/pith/XCZOCFWYJCC5BGN6AA2U2C7SAJ/action/replication_record"}},"created_at":"2026-07-05T06:04:17.727758+00:00","updated_at":"2026-07-05T06:04:17.727758+00:00"}