{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:W7WQ6TS2CMH4PHFWT36IG7AF6W","short_pith_number":"pith:W7WQ6TS2","schema_version":"1.0","canonical_sha256":"b7ed0f4e5a130fc79cb69efc837c05f5a7bf882dd1bfc0a0ef62100899b7d8f7","source":{"kind":"arxiv","id":"2312.09634","version":1},"attestation_state":"computed","paper":{"title":"Vectorizing string entries for data processing on tables: when are larger language models better?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"CNRS, Edouard Oyallon (MLIA, Ga\\\"el Varoquaux (SODA), ISIR, ISIR), L\\'eo Grinsztajn (SODA, MLIA, Myung Jun Kim (SODA), SU)","submitted_at":"2023-12-15T09:23:56Z","abstract_excerpt":"There are increasingly efficient data processing pipelines that work on vectors of numbers, for instance most machine learning models, or vector databases for fast similarity search. These require converting the data to numbers. While this conversion is easy for simple numerical and categorical entries, databases are strife with text entries, such as names or descriptions. In the age of large language models, what's the best strategies to vectorize tables entries, baring in mind that larger models entail more operational complexity? We study the benefits of language models in 14 analytical tas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.09634","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2023-12-15T09:23:56Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3b7c5e93ff68dd2ee57422dcb3b7bfeca97aa6d296b6b88940ee3768acb4597e","abstract_canon_sha256":"3ddc715cd1f7a9719818db39131e78d013c2f3cf23d9b8eb04d3d77bbdce203c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:24:28.488022Z","signature_b64":"lXV8JUmr2YZwYypQwnTBbNOVapeTIZRMJxf/8iEQhfLdHBboM3mjlsDHn/4HP7+I9aV4BMNxMSkNLBmNyDH1DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7ed0f4e5a130fc79cb69efc837c05f5a7bf882dd1bfc0a0ef62100899b7d8f7","last_reissued_at":"2026-07-05T07:24:28.487586Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:24:28.487586Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vectorizing string entries for data processing on tables: when are larger language models better?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"CNRS, Edouard Oyallon (MLIA, Ga\\\"el Varoquaux (SODA), ISIR, ISIR), L\\'eo Grinsztajn (SODA, MLIA, Myung Jun Kim (SODA), SU)","submitted_at":"2023-12-15T09:23:56Z","abstract_excerpt":"There are increasingly efficient data processing pipelines that work on vectors of numbers, for instance most machine learning models, or vector databases for fast similarity search. These require converting the data to numbers. While this conversion is easy for simple numerical and categorical entries, databases are strife with text entries, such as names or descriptions. In the age of large language models, what's the best strategies to vectorize tables entries, baring in mind that larger models entail more operational complexity? We study the benefits of language models in 14 analytical tas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09634","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09634/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.09634","created_at":"2026-07-05T07:24:28.487639+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.09634v1","created_at":"2026-07-05T07:24:28.487639+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09634","created_at":"2026-07-05T07:24:28.487639+00:00"},{"alias_kind":"pith_short_12","alias_value":"W7WQ6TS2CMH4","created_at":"2026-07-05T07:24:28.487639+00:00"},{"alias_kind":"pith_short_16","alias_value":"W7WQ6TS2CMH4PHFW","created_at":"2026-07-05T07:24:28.487639+00:00"},{"alias_kind":"pith_short_8","alias_value":"W7WQ6TS2","created_at":"2026-07-05T07:24:28.487639+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13986","citing_title":"TabPFN-3: Technical Report","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30410","citing_title":"Beyond IID: How General Are Tabular Foundation Models, Really?","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13986","citing_title":"TabPFN-3: Technical Report","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12292","citing_title":"STRABLE: Benchmarking Tabular Machine Learning with Strings","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10616","citing_title":"MulTaBench: Benchmarking Multimodal Tabular Learning with Text and Image","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W","json":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W.json","graph_json":"https://pith.science/api/pith-number/W7WQ6TS2CMH4PHFWT36IG7AF6W/graph.json","events_json":"https://pith.science/api/pith-number/W7WQ6TS2CMH4PHFWT36IG7AF6W/events.json","paper":"https://pith.science/paper/W7WQ6TS2"},"agent_actions":{"view_html":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W","download_json":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W.json","view_paper":"https://pith.science/paper/W7WQ6TS2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.09634&json=true","fetch_graph":"https://pith.science/api/pith-number/W7WQ6TS2CMH4PHFWT36IG7AF6W/graph.json","fetch_events":"https://pith.science/api/pith-number/W7WQ6TS2CMH4PHFWT36IG7AF6W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W/action/storage_attestation","attest_author":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W/action/author_attestation","sign_citation":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W/action/citation_signature","submit_replication":"https://pith.science/pith/W7WQ6TS2CMH4PHFWT36IG7AF6W/action/replication_record"}},"created_at":"2026-07-05T07:24:28.487639+00:00","updated_at":"2026-07-05T07:24:28.487639+00:00"}