{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6BLHTB7DONB3DUEAEQR7KX7J7X","short_pith_number":"pith:6BLHTB7D","schema_version":"1.0","canonical_sha256":"f0567987e37343b1d0802423f55fe9fdea1042e6f44881fda5340480d5683462","source":{"kind":"arxiv","id":"2407.02694","version":2},"attestation_state":"computed","paper":{"title":"LLM-Select: Feature Selection with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Daniel P. Jeong, Pradeep Ravikumar, Zachary C. Lipton","submitted_at":"2024-07-02T22:23:40Z","abstract_excerpt":"In this paper, we demonstrate a surprising capability of large language models (LLMs): given only input feature names and a description of a prediction task, they are capable of selecting the most predictive features, with performance rivaling the standard tools of data science. Remarkably, these models exhibit this capacity across various query mechanisms. For example, we zero-shot prompt an LLM to output a numerical importance score for a feature (e.g., \"blood pressure\") in predicting an outcome of interest (e.g., \"heart failure\"), with no additional context. In particular, we find that the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.02694","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-02T22:23:40Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"90a0e3871a3b3e8b2aae728519256b9751cf90ea2537b9ded839e0e01e083eee","abstract_canon_sha256":"506e7bb3201b4f236f8e9e7b37ff27ef437932bc5042307f5027031fae648e1c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:42.762621Z","signature_b64":"2gnmwdrT0g42SX3cbXR6g1A2r2lnQ4xMQHa8OjGaQbQICtLxoHINwj6nb8ewZEcGzL1VFrZeDInC6w6NETlkCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f0567987e37343b1d0802423f55fe9fdea1042e6f44881fda5340480d5683462","last_reissued_at":"2026-07-05T10:50:42.762148Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:42.762148Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM-Select: Feature Selection with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Daniel P. Jeong, Pradeep Ravikumar, Zachary C. Lipton","submitted_at":"2024-07-02T22:23:40Z","abstract_excerpt":"In this paper, we demonstrate a surprising capability of large language models (LLMs): given only input feature names and a description of a prediction task, they are capable of selecting the most predictive features, with performance rivaling the standard tools of data science. Remarkably, these models exhibit this capacity across various query mechanisms. For example, we zero-shot prompt an LLM to output a numerical importance score for a feature (e.g., \"blood pressure\") in predicting an outcome of interest (e.g., \"heart failure\"), with no additional context. In particular, we find that the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.02694","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.02694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.02694","created_at":"2026-07-05T10:50:42.762205+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.02694v2","created_at":"2026-07-05T10:50:42.762205+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.02694","created_at":"2026-07-05T10:50:42.762205+00:00"},{"alias_kind":"pith_short_12","alias_value":"6BLHTB7DONB3","created_at":"2026-07-05T10:50:42.762205+00:00"},{"alias_kind":"pith_short_16","alias_value":"6BLHTB7DONB3DUEA","created_at":"2026-07-05T10:50:42.762205+00:00"},{"alias_kind":"pith_short_8","alias_value":"6BLHTB7D","created_at":"2026-07-05T10:50:42.762205+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.16759","citing_title":"Can Explanations Improve Recommendations? Evidence from Prediction-Informed Explanations","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14334","citing_title":"Mamba-SSM with LLM Reasoning for Feature Selection: Faithfulness-Aware Biomarker Discovery","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20261","citing_title":"Memory-Augmented LLM-based Multi-Agent System for Automated Feature Generation on Tabular Data","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X","json":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X.json","graph_json":"https://pith.science/api/pith-number/6BLHTB7DONB3DUEAEQR7KX7J7X/graph.json","events_json":"https://pith.science/api/pith-number/6BLHTB7DONB3DUEAEQR7KX7J7X/events.json","paper":"https://pith.science/paper/6BLHTB7D"},"agent_actions":{"view_html":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X","download_json":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X.json","view_paper":"https://pith.science/paper/6BLHTB7D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.02694&json=true","fetch_graph":"https://pith.science/api/pith-number/6BLHTB7DONB3DUEAEQR7KX7J7X/graph.json","fetch_events":"https://pith.science/api/pith-number/6BLHTB7DONB3DUEAEQR7KX7J7X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X/action/storage_attestation","attest_author":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X/action/author_attestation","sign_citation":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X/action/citation_signature","submit_replication":"https://pith.science/pith/6BLHTB7DONB3DUEAEQR7KX7J7X/action/replication_record"}},"created_at":"2026-07-05T10:50:42.762205+00:00","updated_at":"2026-07-05T10:50:42.762205+00:00"}