{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RBXYAWIYLSYSYJLD63RU4SBB55","short_pith_number":"pith:RBXYAWIY","schema_version":"1.0","canonical_sha256":"886f8059185cb12c2563f6e34e4821ef5f5bcd09400d07a679767bae9ccc77f3","source":{"kind":"arxiv","id":"2410.08020","version":3},"attestation_state":"computed","paper":{"title":"Efficiently Learning at Test-Time: Active Fine-Tuning of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Ido Hakimi, Jonas H\\\"ubotter, Sascha Bongni","submitted_at":"2024-10-10T15:17:49Z","abstract_excerpt":"Recent efforts in fine-tuning language models often rely on automatic data selection, commonly using Nearest Neighbors retrieval from large datasets. However, we theoretically show that this approach tends to select redundant data, limiting its effectiveness or even hurting performance. To address this, we introduce SIFT, a data selection algorithm designed to reduce uncertainty about the model's response given a prompt, which unifies ideas from retrieval and active learning. Whereas Nearest Neighbor retrieval typically fails in the presence of information duplication, SIFT accounts for inform"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08020","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-10T15:17:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8eb87983fe3cc683d884cfaee0579cc531c4f9418df6cb1fb0ef9734e3a11371","abstract_canon_sha256":"028321bcfe1bf16a1f96b67b315ef732b0ef42801b18709104c60e87523b3b9f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:25.643027Z","signature_b64":"zoXxOmJCBld0KhT1SLKly5QfQE90DKVgStPoPZK2phNSnEsm6CpqjHnh/MwfMfZ0W1o9mC75U5pcpd8GkR1tAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"886f8059185cb12c2563f6e34e4821ef5f5bcd09400d07a679767bae9ccc77f3","last_reissued_at":"2026-07-05T10:11:25.642479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:25.642479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficiently Learning at Test-Time: Active Fine-Tuning of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Ido Hakimi, Jonas H\\\"ubotter, Sascha Bongni","submitted_at":"2024-10-10T15:17:49Z","abstract_excerpt":"Recent efforts in fine-tuning language models often rely on automatic data selection, commonly using Nearest Neighbors retrieval from large datasets. However, we theoretically show that this approach tends to select redundant data, limiting its effectiveness or even hurting performance. To address this, we introduce SIFT, a data selection algorithm designed to reduce uncertainty about the model's response given a prompt, which unifies ideas from retrieval and active learning. Whereas Nearest Neighbor retrieval typically fails in the presence of information duplication, SIFT accounts for inform"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08020","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08020/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08020","created_at":"2026-07-05T10:11:25.642535+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08020v3","created_at":"2026-07-05T10:11:25.642535+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08020","created_at":"2026-07-05T10:11:25.642535+00:00"},{"alias_kind":"pith_short_12","alias_value":"RBXYAWIYLSYS","created_at":"2026-07-05T10:11:25.642535+00:00"},{"alias_kind":"pith_short_16","alias_value":"RBXYAWIYLSYSYJLD","created_at":"2026-07-05T10:11:25.642535+00:00"},{"alias_kind":"pith_short_8","alias_value":"RBXYAWIY","created_at":"2026-07-05T10:11:25.642535+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26614","citing_title":"HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28349","citing_title":"HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18570","citing_title":"Streamlining Analysis and Design of Two-Dimensional Electronic Spectroscopy using Machine Learning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03988","citing_title":"The Signal is in the Steps: Local Scoring for Reasoning Data Selection","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2601.16175","citing_title":"Learning to Discover at Test Time","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10237","citing_title":"The Benefits of Temporal Correlations: SGD Learns k-Juntas from Random Walks Efficiently","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18907","citing_title":"Gradient-Based Program Synthesis with Neurally Interpreted Languages","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18493","citing_title":"Too Correct to Learn: Reinforcement Learning on Saturated Reasoning Data","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55","json":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55.json","graph_json":"https://pith.science/api/pith-number/RBXYAWIYLSYSYJLD63RU4SBB55/graph.json","events_json":"https://pith.science/api/pith-number/RBXYAWIYLSYSYJLD63RU4SBB55/events.json","paper":"https://pith.science/paper/RBXYAWIY"},"agent_actions":{"view_html":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55","download_json":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55.json","view_paper":"https://pith.science/paper/RBXYAWIY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08020&json=true","fetch_graph":"https://pith.science/api/pith-number/RBXYAWIYLSYSYJLD63RU4SBB55/graph.json","fetch_events":"https://pith.science/api/pith-number/RBXYAWIYLSYSYJLD63RU4SBB55/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55/action/storage_attestation","attest_author":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55/action/author_attestation","sign_citation":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55/action/citation_signature","submit_replication":"https://pith.science/pith/RBXYAWIYLSYSYJLD63RU4SBB55/action/replication_record"}},"created_at":"2026-07-05T10:11:25.642535+00:00","updated_at":"2026-07-05T10:11:25.642535+00:00"}