{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SG7MSVS44533M6TVREQ3L6X7OD","short_pith_number":"pith:SG7MSVS4","schema_version":"1.0","canonical_sha256":"91bec9565ce777b67a758921b5faff70de0e1f639b290ca40cc33ab59f7ea591","source":{"kind":"arxiv","id":"2212.04612","version":3},"attestation_state":"computed","paper":{"title":"Training Data Influence Analysis and Estimation: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Daniel Lowd, Zayd Hammoudeh","submitted_at":"2022-12-09T00:32:46Z","abstract_excerpt":"Good models require good training data. For overparameterized deep models, the causal relationship between training data and model predictions is increasingly opaque and poorly understood. Influence analysis partially demystifies training's underlying interactions by quantifying the amount each training instance alters the final model. Measuring the training data's influence exactly can be provably hard in the worst case; this has led to the development and use of influence estimators, which only approximate the true influence. This paper provides the first comprehensive survey of training dat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.04612","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-12-09T00:32:46Z","cross_cats_sorted":[],"title_canon_sha256":"01f0cfc9cf7be7a6387f86e5a926cf799d172bc262be19e061e7247167e35fdc","abstract_canon_sha256":"91efdb5b79392ff7950ad13f26ae91a3edd973bdf5086b36fb80b7582014a513"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:19.486992Z","signature_b64":"x+AVKPVYjr61QVRskx2vbtMHp0dbLg+u+lgpZkCE2i2VcRWJHlWhiIXJu/Vr4TtXPOUYa5SpVog7hxrwQ4yWBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"91bec9565ce777b67a758921b5faff70de0e1f639b290ca40cc33ab59f7ea591","last_reissued_at":"2026-07-05T08:02:19.486555Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:19.486555Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Data Influence Analysis and Estimation: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Daniel Lowd, Zayd Hammoudeh","submitted_at":"2022-12-09T00:32:46Z","abstract_excerpt":"Good models require good training data. For overparameterized deep models, the causal relationship between training data and model predictions is increasingly opaque and poorly understood. Influence analysis partially demystifies training's underlying interactions by quantifying the amount each training instance alters the final model. Measuring the training data's influence exactly can be provably hard in the worst case; this has led to the development and use of influence estimators, which only approximate the true influence. This paper provides the first comprehensive survey of training dat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.04612","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.04612/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.04612","created_at":"2026-07-05T08:02:19.486621+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.04612v3","created_at":"2026-07-05T08:02:19.486621+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.04612","created_at":"2026-07-05T08:02:19.486621+00:00"},{"alias_kind":"pith_short_12","alias_value":"SG7MSVS44533","created_at":"2026-07-05T08:02:19.486621+00:00"},{"alias_kind":"pith_short_16","alias_value":"SG7MSVS44533M6TV","created_at":"2026-07-05T08:02:19.486621+00:00"},{"alias_kind":"pith_short_8","alias_value":"SG7MSVS4","created_at":"2026-07-05T08:02:19.486621+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.14648","citing_title":"Understanding Data Influence with Differential Approximation","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD","json":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD.json","graph_json":"https://pith.science/api/pith-number/SG7MSVS44533M6TVREQ3L6X7OD/graph.json","events_json":"https://pith.science/api/pith-number/SG7MSVS44533M6TVREQ3L6X7OD/events.json","paper":"https://pith.science/paper/SG7MSVS4"},"agent_actions":{"view_html":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD","download_json":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD.json","view_paper":"https://pith.science/paper/SG7MSVS4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.04612&json=true","fetch_graph":"https://pith.science/api/pith-number/SG7MSVS44533M6TVREQ3L6X7OD/graph.json","fetch_events":"https://pith.science/api/pith-number/SG7MSVS44533M6TVREQ3L6X7OD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD/action/storage_attestation","attest_author":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD/action/author_attestation","sign_citation":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD/action/citation_signature","submit_replication":"https://pith.science/pith/SG7MSVS44533M6TVREQ3L6X7OD/action/replication_record"}},"created_at":"2026-07-05T08:02:19.486621+00:00","updated_at":"2026-07-05T08:02:19.486621+00:00"}