{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YM5VRIV5R6SCOKFL7HHCBOE2QM","short_pith_number":"pith:YM5VRIV5","schema_version":"1.0","canonical_sha256":"c33b58a2bd8fa42728abf9ce20b89a832b4be06e7b1b7a5c6f5c69ff59f2f784","source":{"kind":"arxiv","id":"2502.02199","version":1},"attestation_state":"computed","paper":{"title":"When Dimensionality Hurts: The Role of LLM Embedding Compression for Noisy Regression Tasks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CE","cs.LG","q-fin.CP"],"primary_cat":"cs.CL","authors_text":"Felix Drinkall, Janet B. Pierrehumbert, Stefan Zohren","submitted_at":"2025-02-04T10:23:11Z","abstract_excerpt":"Large language models (LLMs) have shown remarkable success in language modelling due to scaling laws found in model size and the hidden dimension of the model's text representation. Yet, we demonstrate that compressed representations of text can yield better performance in LLM-based regression tasks. In this paper, we compare the relative performance of embedding compression in three different signal-to-noise contexts: financial return prediction, writing quality assessment and review scoring. Our results show that compressing embeddings, in a minimally supervised manner using an autoencoder's"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02199","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-04T10:23:11Z","cross_cats_sorted":["cs.CE","cs.LG","q-fin.CP"],"title_canon_sha256":"db397694ae56daa11a65a578c4a3e89b5c8bfdcc8d8f4375718ed1b1c0cb48e1","abstract_canon_sha256":"e493b99141181bf1d935cf5e38605c5ef9f9cbfaa6ec7c795765f1a92eb31ac9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:25.330712Z","signature_b64":"Hs+z8AOmf5C3CcLYpEFyJljNCXbaMS6JRXloFU52pFvjkUDu6ilR6d0ti7syJaHw6iv9zswsEmaain72dtrfAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c33b58a2bd8fa42728abf9ce20b89a832b4be06e7b1b7a5c6f5c69ff59f2f784","last_reissued_at":"2026-07-05T10:09:25.330232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:25.330232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Dimensionality Hurts: The Role of LLM Embedding Compression for Noisy Regression Tasks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CE","cs.LG","q-fin.CP"],"primary_cat":"cs.CL","authors_text":"Felix Drinkall, Janet B. Pierrehumbert, Stefan Zohren","submitted_at":"2025-02-04T10:23:11Z","abstract_excerpt":"Large language models (LLMs) have shown remarkable success in language modelling due to scaling laws found in model size and the hidden dimension of the model's text representation. Yet, we demonstrate that compressed representations of text can yield better performance in LLM-based regression tasks. In this paper, we compare the relative performance of embedding compression in three different signal-to-noise contexts: financial return prediction, writing quality assessment and review scoring. Our results show that compressing embeddings, in a minimally supervised manner using an autoencoder's"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02199","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02199/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02199","created_at":"2026-07-05T10:09:25.330288+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02199v1","created_at":"2026-07-05T10:09:25.330288+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02199","created_at":"2026-07-05T10:09:25.330288+00:00"},{"alias_kind":"pith_short_12","alias_value":"YM5VRIV5R6SC","created_at":"2026-07-05T10:09:25.330288+00:00"},{"alias_kind":"pith_short_16","alias_value":"YM5VRIV5R6SCOKFL","created_at":"2026-07-05T10:09:25.330288+00:00"},{"alias_kind":"pith_short_8","alias_value":"YM5VRIV5","created_at":"2026-07-05T10:09:25.330288+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM","json":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM.json","graph_json":"https://pith.science/api/pith-number/YM5VRIV5R6SCOKFL7HHCBOE2QM/graph.json","events_json":"https://pith.science/api/pith-number/YM5VRIV5R6SCOKFL7HHCBOE2QM/events.json","paper":"https://pith.science/paper/YM5VRIV5"},"agent_actions":{"view_html":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM","download_json":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM.json","view_paper":"https://pith.science/paper/YM5VRIV5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02199&json=true","fetch_graph":"https://pith.science/api/pith-number/YM5VRIV5R6SCOKFL7HHCBOE2QM/graph.json","fetch_events":"https://pith.science/api/pith-number/YM5VRIV5R6SCOKFL7HHCBOE2QM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM/action/storage_attestation","attest_author":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM/action/author_attestation","sign_citation":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM/action/citation_signature","submit_replication":"https://pith.science/pith/YM5VRIV5R6SCOKFL7HHCBOE2QM/action/replication_record"}},"created_at":"2026-07-05T10:09:25.330288+00:00","updated_at":"2026-07-05T10:09:25.330288+00:00"}