{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:EN727XKQ5U2STZ5BC7AWJFBIXJ","short_pith_number":"pith:EN727XKQ","schema_version":"1.0","canonical_sha256":"237fafdd50ed3529e7a117c1649428ba66979ce228e05f126fb83efe25fdf9e4","source":{"kind":"arxiv","id":"1904.02668","version":4},"attestation_state":"computed","paper":{"title":"Inoculation by Fine-Tuning: A Method for Analyzing Challenge Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Nelson F. Liu, Noah A. Smith, Roy Schwartz","submitted_at":"2019-04-04T17:04:30Z","abstract_excerpt":"Several datasets have recently been constructed to expose brittleness in models trained on existing benchmarks. While model performance on these challenge datasets is significantly lower compared to the original benchmark, it is unclear what particular weaknesses they reveal. For example, a challenge dataset may be difficult because it targets phenomena that current models cannot capture, or because it simply exploits blind spots in a model's specific training set. We introduce inoculation by fine-tuning, a new analysis method for studying challenge datasets by exposing models (the metaphorica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.02668","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-04-04T17:04:30Z","cross_cats_sorted":[],"title_canon_sha256":"e9d56c93e308e2dfc71da63aaf608e6ba65a7cfe85f2ca1f9c93eb615d2cba59","abstract_canon_sha256":"03c87ad226bdfcefd332c735fe18eaaede989a9c536364edc7bff0103d099874"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:47:39.267094Z","signature_b64":"5uNxlFqLguxI8/rY8SVIPcyKWI60hHEGgbVy3ivsGsZtXcZNTwizJyRLPyh5laG0hIPwlinNTm6VODIm+0VBCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"237fafdd50ed3529e7a117c1649428ba66979ce228e05f126fb83efe25fdf9e4","last_reissued_at":"2026-05-17T23:47:39.266689Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:47:39.266689Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inoculation by Fine-Tuning: A Method for Analyzing Challenge Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Nelson F. Liu, Noah A. Smith, Roy Schwartz","submitted_at":"2019-04-04T17:04:30Z","abstract_excerpt":"Several datasets have recently been constructed to expose brittleness in models trained on existing benchmarks. While model performance on these challenge datasets is significantly lower compared to the original benchmark, it is unclear what particular weaknesses they reveal. For example, a challenge dataset may be difficult because it targets phenomena that current models cannot capture, or because it simply exploits blind spots in a model's specific training set. We introduce inoculation by fine-tuning, a new analysis method for studying challenge datasets by exposing models (the metaphorica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.02668","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.02668","created_at":"2026-05-17T23:47:39.266749+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.02668v4","created_at":"2026-05-17T23:47:39.266749+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.02668","created_at":"2026-05-17T23:47:39.266749+00:00"},{"alias_kind":"pith_short_12","alias_value":"EN727XKQ5U2S","created_at":"2026-05-18T12:33:15.570797+00:00"},{"alias_kind":"pith_short_16","alias_value":"EN727XKQ5U2STZ5B","created_at":"2026-05-18T12:33:15.570797+00:00"},{"alias_kind":"pith_short_8","alias_value":"EN727XKQ","created_at":"2026-05-18T12:33:15.570797+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"1905.00537","citing_title":"SuperGLUE: A Stickier Benchmark for General-Purpose Language Understanding Systems","ref_index":118,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ","json":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ.json","graph_json":"https://pith.science/api/pith-number/EN727XKQ5U2STZ5BC7AWJFBIXJ/graph.json","events_json":"https://pith.science/api/pith-number/EN727XKQ5U2STZ5BC7AWJFBIXJ/events.json","paper":"https://pith.science/paper/EN727XKQ"},"agent_actions":{"view_html":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ","download_json":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ.json","view_paper":"https://pith.science/paper/EN727XKQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.02668&json=true","fetch_graph":"https://pith.science/api/pith-number/EN727XKQ5U2STZ5BC7AWJFBIXJ/graph.json","fetch_events":"https://pith.science/api/pith-number/EN727XKQ5U2STZ5BC7AWJFBIXJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ/action/storage_attestation","attest_author":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ/action/author_attestation","sign_citation":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ/action/citation_signature","submit_replication":"https://pith.science/pith/EN727XKQ5U2STZ5BC7AWJFBIXJ/action/replication_record"}},"created_at":"2026-05-17T23:47:39.266749+00:00","updated_at":"2026-05-17T23:47:39.266749+00:00"}