{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DZFY33WWU26E4CXZ36CZI2QQA4","short_pith_number":"pith:DZFY33WW","schema_version":"1.0","canonical_sha256":"1e4b8deed6a6bc4e0af9df85946a10070c883ee9640f3f319d8afbba01662738","source":{"kind":"arxiv","id":"2407.00085","version":2},"attestation_state":"computed","paper":{"title":"Compressing Search with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Jennifer L. Steele, Thomas Mulc","submitted_at":"2024-06-24T17:43:49Z","abstract_excerpt":"Millions of people turn to Google Search each day for information on things as diverse as new cars or flu symptoms. The terms that they enter contain valuable information on their daily intent and activities, but the information in these search terms has been difficult to fully leverage. User-defined categorical filters have been the most common way to shrink the dimensionality of search data to a tractable size for analysis and modeling. In this paper we present a new approach to reducing the dimensionality of search data while retaining much of the information in the individual terms without"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00085","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-06-24T17:43:49Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2567774ebb4adefbc21a88825edd308b68ff5fd7b9844de7dac2f94757a34276","abstract_canon_sha256":"2cdafcf52b09d7f9220543b4614e5e63a0ae94f31526d8cbd4e9f2d0f6c1d9d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:55.707087Z","signature_b64":"K58CL5zuN4yqJYIkl8Os1tJtPGJpejl97rgo5ru6WHQt6ZX5Op6jqdKY/XBTvlv8HAJBDnJUofiwDWs+Kf7vBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e4b8deed6a6bc4e0af9df85946a10070c883ee9640f3f319d8afbba01662738","last_reissued_at":"2026-07-05T10:46:55.706677Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:55.706677Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compressing Search with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Jennifer L. Steele, Thomas Mulc","submitted_at":"2024-06-24T17:43:49Z","abstract_excerpt":"Millions of people turn to Google Search each day for information on things as diverse as new cars or flu symptoms. The terms that they enter contain valuable information on their daily intent and activities, but the information in these search terms has been difficult to fully leverage. User-defined categorical filters have been the most common way to shrink the dimensionality of search data to a tractable size for analysis and modeling. In this paper we present a new approach to reducing the dimensionality of search data while retaining much of the information in the individual terms without"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00085","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00085/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00085","created_at":"2026-07-05T10:46:55.706744+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00085v2","created_at":"2026-07-05T10:46:55.706744+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00085","created_at":"2026-07-05T10:46:55.706744+00:00"},{"alias_kind":"pith_short_12","alias_value":"DZFY33WWU26E","created_at":"2026-07-05T10:46:55.706744+00:00"},{"alias_kind":"pith_short_16","alias_value":"DZFY33WWU26E4CXZ","created_at":"2026-07-05T10:46:55.706744+00:00"},{"alias_kind":"pith_short_8","alias_value":"DZFY33WW","created_at":"2026-07-05T10:46:55.706744+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06139","citing_title":"LCIRC: A Recurrent Compression Approach for Efficient Long-form Context and Query Dependent Modeling in LLMs","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4","json":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4.json","graph_json":"https://pith.science/api/pith-number/DZFY33WWU26E4CXZ36CZI2QQA4/graph.json","events_json":"https://pith.science/api/pith-number/DZFY33WWU26E4CXZ36CZI2QQA4/events.json","paper":"https://pith.science/paper/DZFY33WW"},"agent_actions":{"view_html":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4","download_json":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4.json","view_paper":"https://pith.science/paper/DZFY33WW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00085&json=true","fetch_graph":"https://pith.science/api/pith-number/DZFY33WWU26E4CXZ36CZI2QQA4/graph.json","fetch_events":"https://pith.science/api/pith-number/DZFY33WWU26E4CXZ36CZI2QQA4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4/action/storage_attestation","attest_author":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4/action/author_attestation","sign_citation":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4/action/citation_signature","submit_replication":"https://pith.science/pith/DZFY33WWU26E4CXZ36CZI2QQA4/action/replication_record"}},"created_at":"2026-07-05T10:46:55.706744+00:00","updated_at":"2026-07-05T10:46:55.706744+00:00"}