{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PLOGSUNYUZ6MDQEGNZOP4FGKNH","short_pith_number":"pith:PLOGSUNY","schema_version":"1.0","canonical_sha256":"7adc6951b8a67cc1c0866e5cfe14ca69fa23915752b8438a8e6f92093b64b494","source":{"kind":"arxiv","id":"2410.07395","version":1},"attestation_state":"computed","paper":{"title":"LLM Embeddings Improve Test-time Adaptation to Tabular $Y|X$-Shifts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Henry Lam, Hongseok Namkoong, Jiashuo Liu, Yibo Zeng","submitted_at":"2024-10-09T19:46:30Z","abstract_excerpt":"For tabular datasets, the change in the relationship between the label and covariates ($Y|X$-shifts) is common due to missing variables (a.k.a. confounders). Since it is impossible to generalize to a completely new and unknown domain, we study models that are easy to adapt to the target domain even with few labeled examples. We focus on building more informative representations of tabular data that can mitigate $Y|X$-shifts, and propose to leverage the prior world knowledge in LLMs by serializing (write down) the tabular data to encode it. We find LLM embeddings alone provide inconsistent impr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07395","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-09T19:46:30Z","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"title_canon_sha256":"b3ff3fe7c535bc621a9f70bc8bb912fc59176b11ac25de19ea221cf55f1df4e1","abstract_canon_sha256":"ea2c4dd2e0107f74f224d7347b660b6ea070194f49eb48d7bd0a555303221288"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:35.729751Z","signature_b64":"CuxKN2s36tZjn0nkwPah4Ze3E9kmh6RDcSUf+qj7+WD/Udt1d9hSi0w8J0OvuGN668pXl0MEdcVgNVlp8xTqAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7adc6951b8a67cc1c0866e5cfe14ca69fa23915752b8438a8e6f92093b64b494","last_reissued_at":"2026-07-05T09:18:35.729262Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:35.729262Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM Embeddings Improve Test-time Adaptation to Tabular $Y|X$-Shifts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Henry Lam, Hongseok Namkoong, Jiashuo Liu, Yibo Zeng","submitted_at":"2024-10-09T19:46:30Z","abstract_excerpt":"For tabular datasets, the change in the relationship between the label and covariates ($Y|X$-shifts) is common due to missing variables (a.k.a. confounders). Since it is impossible to generalize to a completely new and unknown domain, we study models that are easy to adapt to the target domain even with few labeled examples. We focus on building more informative representations of tabular data that can mitigate $Y|X$-shifts, and propose to leverage the prior world knowledge in LLMs by serializing (write down) the tabular data to encode it. We find LLM embeddings alone provide inconsistent impr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07395","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07395/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07395","created_at":"2026-07-05T09:18:35.729320+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07395v1","created_at":"2026-07-05T09:18:35.729320+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07395","created_at":"2026-07-05T09:18:35.729320+00:00"},{"alias_kind":"pith_short_12","alias_value":"PLOGSUNYUZ6M","created_at":"2026-07-05T09:18:35.729320+00:00"},{"alias_kind":"pith_short_16","alias_value":"PLOGSUNYUZ6MDQEG","created_at":"2026-07-05T09:18:35.729320+00:00"},{"alias_kind":"pith_short_8","alias_value":"PLOGSUNY","created_at":"2026-07-05T09:18:35.729320+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.08000","citing_title":"AntiDote: Bi-level Adversarial Training for Tamper-Resistant LLMs","ref_index":96,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH","json":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH.json","graph_json":"https://pith.science/api/pith-number/PLOGSUNYUZ6MDQEGNZOP4FGKNH/graph.json","events_json":"https://pith.science/api/pith-number/PLOGSUNYUZ6MDQEGNZOP4FGKNH/events.json","paper":"https://pith.science/paper/PLOGSUNY"},"agent_actions":{"view_html":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH","download_json":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH.json","view_paper":"https://pith.science/paper/PLOGSUNY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07395&json=true","fetch_graph":"https://pith.science/api/pith-number/PLOGSUNYUZ6MDQEGNZOP4FGKNH/graph.json","fetch_events":"https://pith.science/api/pith-number/PLOGSUNYUZ6MDQEGNZOP4FGKNH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH/action/storage_attestation","attest_author":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH/action/author_attestation","sign_citation":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH/action/citation_signature","submit_replication":"https://pith.science/pith/PLOGSUNYUZ6MDQEGNZOP4FGKNH/action/replication_record"}},"created_at":"2026-07-05T09:18:35.729320+00:00","updated_at":"2026-07-05T09:18:35.729320+00:00"}