{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TBBYYBPVDHNTVEA4X3YYGDVUAT","short_pith_number":"pith:TBBYYBPV","schema_version":"1.0","canonical_sha256":"98438c05f519db3a901cbef1830eb404caf32f995d833464ebbdf6b4c4b9e789","source":{"kind":"arxiv","id":"2410.16710","version":1},"attestation_state":"computed","paper":{"title":"Influential Language Data Selection via Gradient Trajectory Pursuit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Tao Li, Yang Li, Zhiwei Deng","submitted_at":"2024-10-22T05:32:40Z","abstract_excerpt":"Curating a desirable dataset for training has been the core of building highly capable large language models (Touvron et al., 2023; Achiam et al., 2023; Team et al.,2024). Gradient influence scores (Pruthi et al., 2020; Xia et al., 2024) are shown to be correlated with model performance and are commonly used as the criterion for data selection. However, existing methods are built upon either individual sample rankings or inefficient matching process, leading to suboptimal performance or scaling up issues.In this paper, we propose Gradient Trajectory Pursuit (GTP), an algorithm that performs pu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.16710","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-10-22T05:32:40Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"ad6fad52491d66196aa30d398be238c2aad326045c6c8acdd551431bdd6f1da6","abstract_canon_sha256":"a9f6a3e861d3875b065f5a7abdf074476ec11dc2f8cecc6ebc67a0fdbd51e4d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:05.463645Z","signature_b64":"qHTa5Xd9eJYIaMHOS7kRFpFB4dNzTXzaA+MCqWfbz8WbK00AwnQjR/Kacwmkk2l8yvx9GySxwfOszug6ghVrDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98438c05f519db3a901cbef1830eb404caf32f995d833464ebbdf6b4c4b9e789","last_reissued_at":"2026-07-05T09:24:05.463213Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:05.463213Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Influential Language Data Selection via Gradient Trajectory Pursuit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Tao Li, Yang Li, Zhiwei Deng","submitted_at":"2024-10-22T05:32:40Z","abstract_excerpt":"Curating a desirable dataset for training has been the core of building highly capable large language models (Touvron et al., 2023; Achiam et al., 2023; Team et al.,2024). Gradient influence scores (Pruthi et al., 2020; Xia et al., 2024) are shown to be correlated with model performance and are commonly used as the criterion for data selection. However, existing methods are built upon either individual sample rankings or inefficient matching process, leading to suboptimal performance or scaling up issues.In this paper, we propose Gradient Trajectory Pursuit (GTP), an algorithm that performs pu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.16710","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.16710/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.16710","created_at":"2026-07-05T09:24:05.463281+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.16710v1","created_at":"2026-07-05T09:24:05.463281+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.16710","created_at":"2026-07-05T09:24:05.463281+00:00"},{"alias_kind":"pith_short_12","alias_value":"TBBYYBPVDHNT","created_at":"2026-07-05T09:24:05.463281+00:00"},{"alias_kind":"pith_short_16","alias_value":"TBBYYBPVDHNTVEA4","created_at":"2026-07-05T09:24:05.463281+00:00"},{"alias_kind":"pith_short_8","alias_value":"TBBYYBPV","created_at":"2026-07-05T09:24:05.463281+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09404","citing_title":"Let the Target Select for Itself: Data Selection via Target-Aligned Paths","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11810","citing_title":"GRACE: A Dynamic Coreset Selection Framework for Large Language Model Optimization","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT","json":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT.json","graph_json":"https://pith.science/api/pith-number/TBBYYBPVDHNTVEA4X3YYGDVUAT/graph.json","events_json":"https://pith.science/api/pith-number/TBBYYBPVDHNTVEA4X3YYGDVUAT/events.json","paper":"https://pith.science/paper/TBBYYBPV"},"agent_actions":{"view_html":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT","download_json":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT.json","view_paper":"https://pith.science/paper/TBBYYBPV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.16710&json=true","fetch_graph":"https://pith.science/api/pith-number/TBBYYBPVDHNTVEA4X3YYGDVUAT/graph.json","fetch_events":"https://pith.science/api/pith-number/TBBYYBPVDHNTVEA4X3YYGDVUAT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT/action/storage_attestation","attest_author":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT/action/author_attestation","sign_citation":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT/action/citation_signature","submit_replication":"https://pith.science/pith/TBBYYBPVDHNTVEA4X3YYGDVUAT/action/replication_record"}},"created_at":"2026-07-05T09:24:05.463281+00:00","updated_at":"2026-07-05T09:24:05.463281+00:00"}