{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3B5WROBFTTZZHB4NHJRYXT6JCZ","short_pith_number":"pith:3B5WROBF","schema_version":"1.0","canonical_sha256":"d87b68b8259cf393878d3a638bcfc9167b16eab9e706fe0ded0a33c203d24503","source":{"kind":"arxiv","id":"2304.01106","version":1},"attestation_state":"computed","paper":{"title":"Crossword: A Semantic Approach to Data Compression via Masking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.CL","authors_text":"Kaiming Shen, Liyao Xiang, Mingxiao Li, Rui Jin, Shuguang Cui","submitted_at":"2023-04-03T16:04:06Z","abstract_excerpt":"The traditional methods for data compression are typically based on the symbol-level statistics, with the information source modeled as a long sequence of i.i.d. random variables or a stochastic process, thus establishing the fundamental limit as entropy for lossless compression and as mutual information for lossy compression. However, the source (including text, music, and speech) in the real world is often statistically ill-defined because of its close connection to human perception, and thus the model-driven approach can be quite suboptimal. This study places careful emphasis on English tex"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.01106","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-03T16:04:06Z","cross_cats_sorted":["cs.IT","math.IT"],"title_canon_sha256":"6dad0dc37d71007401618ec00153db3de4ccb70606be805a70c02a18c15b0a2b","abstract_canon_sha256":"51e38c602ea8e9d500225baef2fd1fd0ce2f1f1be736daad731c4f56a0541c84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:57:26.593761Z","signature_b64":"+UPCvvMgZPaHwhE1WTElYmSosp6D9uXUJpYhwQoPqBg9yXpKueqhlX5p4TOt7a+kqlbDIG1zz2TOOLauwpmCCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d87b68b8259cf393878d3a638bcfc9167b16eab9e706fe0ded0a33c203d24503","last_reissued_at":"2026-07-05T05:57:26.593337Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:57:26.593337Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Crossword: A Semantic Approach to Data Compression via Masking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.CL","authors_text":"Kaiming Shen, Liyao Xiang, Mingxiao Li, Rui Jin, Shuguang Cui","submitted_at":"2023-04-03T16:04:06Z","abstract_excerpt":"The traditional methods for data compression are typically based on the symbol-level statistics, with the information source modeled as a long sequence of i.i.d. random variables or a stochastic process, thus establishing the fundamental limit as entropy for lossless compression and as mutual information for lossy compression. However, the source (including text, music, and speech) in the real world is often statistically ill-defined because of its close connection to human perception, and thus the model-driven approach can be quite suboptimal. This study places careful emphasis on English tex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.01106","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.01106/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.01106","created_at":"2026-07-05T05:57:26.593403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.01106v1","created_at":"2026-07-05T05:57:26.593403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.01106","created_at":"2026-07-05T05:57:26.593403+00:00"},{"alias_kind":"pith_short_12","alias_value":"3B5WROBFTTZZ","created_at":"2026-07-05T05:57:26.593403+00:00"},{"alias_kind":"pith_short_16","alias_value":"3B5WROBFTTZZHB4N","created_at":"2026-07-05T05:57:26.593403+00:00"},{"alias_kind":"pith_short_8","alias_value":"3B5WROBF","created_at":"2026-07-05T05:57:26.593403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.15250","citing_title":"An Enhanced Text Compression Approach Using Transformer-based Language Models","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ","json":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ.json","graph_json":"https://pith.science/api/pith-number/3B5WROBFTTZZHB4NHJRYXT6JCZ/graph.json","events_json":"https://pith.science/api/pith-number/3B5WROBFTTZZHB4NHJRYXT6JCZ/events.json","paper":"https://pith.science/paper/3B5WROBF"},"agent_actions":{"view_html":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ","download_json":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ.json","view_paper":"https://pith.science/paper/3B5WROBF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.01106&json=true","fetch_graph":"https://pith.science/api/pith-number/3B5WROBFTTZZHB4NHJRYXT6JCZ/graph.json","fetch_events":"https://pith.science/api/pith-number/3B5WROBFTTZZHB4NHJRYXT6JCZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ/action/storage_attestation","attest_author":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ/action/author_attestation","sign_citation":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ/action/citation_signature","submit_replication":"https://pith.science/pith/3B5WROBFTTZZHB4NHJRYXT6JCZ/action/replication_record"}},"created_at":"2026-07-05T05:57:26.593403+00:00","updated_at":"2026-07-05T05:57:26.593403+00:00"}