{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4W6NLENT4AMJEABSJFC3XN76XA","short_pith_number":"pith:4W6NLENT","schema_version":"1.0","canonical_sha256":"e5bcd591b3e0189200324945bbb7feb838414fd546c5b381a84741def87d2f9e","source":{"kind":"arxiv","id":"2502.03147","version":1},"attestation_state":"computed","paper":{"title":"Scalable In-Context Learning on Tabular Data via Retrieval-Augmented Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiang Bian, Shun Zheng, Xumeng Wen, Yiming Sun, Zhen Xu","submitted_at":"2025-02-05T13:16:41Z","abstract_excerpt":"Recent studies have shown that large language models (LLMs), when customized with post-training on tabular data, can acquire general tabular in-context learning (TabICL) capabilities. These models are able to transfer effectively across diverse data schemas and different task domains. However, existing LLM-based TabICL approaches are constrained to few-shot scenarios due to the sequence length limitations of LLMs, as tabular instances represented in plain text consume substantial tokens. To address this limitation and enable scalable TabICL for any data size, we propose retrieval-augmented LLM"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03147","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-05T13:16:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8961e45a60fba6f2d0d9b847682db9491d9dc94db8325fe6401cab48875af3be","abstract_canon_sha256":"a07592e1e6a84d18f4f62d84307f99dc62bc9a4e11f67f29f87345ef798c7050"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:00.445377Z","signature_b64":"lZA1L/9fkL+z4f8P61XS5nc20djMBvbB9QNPbezHCoIS9dyac1kN2gYFwijv01T3JsY8QntBH2J2NJO2aYxMBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5bcd591b3e0189200324945bbb7feb838414fd546c5b381a84741def87d2f9e","last_reissued_at":"2026-07-05T10:10:00.445000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:00.445000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable In-Context Learning on Tabular Data via Retrieval-Augmented Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiang Bian, Shun Zheng, Xumeng Wen, Yiming Sun, Zhen Xu","submitted_at":"2025-02-05T13:16:41Z","abstract_excerpt":"Recent studies have shown that large language models (LLMs), when customized with post-training on tabular data, can acquire general tabular in-context learning (TabICL) capabilities. These models are able to transfer effectively across diverse data schemas and different task domains. However, existing LLM-based TabICL approaches are constrained to few-shot scenarios due to the sequence length limitations of LLMs, as tabular instances represented in plain text consume substantial tokens. To address this limitation and enable scalable TabICL for any data size, we propose retrieval-augmented LLM"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03147","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03147/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03147","created_at":"2026-07-05T10:10:00.445055+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03147v1","created_at":"2026-07-05T10:10:00.445055+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03147","created_at":"2026-07-05T10:10:00.445055+00:00"},{"alias_kind":"pith_short_12","alias_value":"4W6NLENT4AMJ","created_at":"2026-07-05T10:10:00.445055+00:00"},{"alias_kind":"pith_short_16","alias_value":"4W6NLENT4AMJEABS","created_at":"2026-07-05T10:10:00.445055+00:00"},{"alias_kind":"pith_short_8","alias_value":"4W6NLENT","created_at":"2026-07-05T10:10:00.445055+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29280","citing_title":"Deterministic Decisions for High-Stakes AI. A Zero-Egress Pipeline with the Deployability of RAG and the Accuracy of Machine Learning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31272","citing_title":"Algorithmic Recourse of In-Context Learning for Tabular Data","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06806","citing_title":"MachineLearningLM: Scaling Many-shot In-context Learning via Continued Pretraining","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA","json":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA.json","graph_json":"https://pith.science/api/pith-number/4W6NLENT4AMJEABSJFC3XN76XA/graph.json","events_json":"https://pith.science/api/pith-number/4W6NLENT4AMJEABSJFC3XN76XA/events.json","paper":"https://pith.science/paper/4W6NLENT"},"agent_actions":{"view_html":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA","download_json":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA.json","view_paper":"https://pith.science/paper/4W6NLENT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03147&json=true","fetch_graph":"https://pith.science/api/pith-number/4W6NLENT4AMJEABSJFC3XN76XA/graph.json","fetch_events":"https://pith.science/api/pith-number/4W6NLENT4AMJEABSJFC3XN76XA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA/action/storage_attestation","attest_author":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA/action/author_attestation","sign_citation":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA/action/citation_signature","submit_replication":"https://pith.science/pith/4W6NLENT4AMJEABSJFC3XN76XA/action/replication_record"}},"created_at":"2026-07-05T10:10:00.445055+00:00","updated_at":"2026-07-05T10:10:00.445055+00:00"}