{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ZCYB25GWFSR6WORL5L3C67IOD6","short_pith_number":"pith:ZCYB25GW","schema_version":"1.0","canonical_sha256":"c8b01d74d62ca3eb3a2beaf62f7d0e1fa2899b7a402b9cb43a12a930137dd28c","source":{"kind":"arxiv","id":"2106.14282","version":3},"attestation_state":"computed","paper":{"title":"A Closer Look at How Fine-tuning Changes BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Vivek Srikumar, Yichu Zhou","submitted_at":"2021-06-27T17:01:43Z","abstract_excerpt":"Given the prevalence of pre-trained contextualized representations in today's NLP, there have been many efforts to understand what information they contain, and why they seem to be universally successful. The most common approach to use these representations involves fine-tuning them for an end task. Yet, how fine-tuning changes the underlying embedding space is less studied. In this work, we study the English BERT family and use two probing techniques to analyze how fine-tuning changes the space. We hypothesize that fine-tuning affects classification performance by increasing the distances be"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.14282","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-06-27T17:01:43Z","cross_cats_sorted":[],"title_canon_sha256":"c023e8e00c6f5786beb3b1824ef4fab4fdb756635f978d80311023f64ec9f967","abstract_canon_sha256":"e6fe26ac63b5f0ebb516d74998fa6726d2068c1a9986d0d410c522db7f9ed1fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:32.131296Z","signature_b64":"iEnE9OOAvYWgJ8d5Qp8aoFYyRe2PO40ZAvsOUjDJQluAynb02kzGAw+GRktQ4HOVhJH5tJs6PPXTYe/qIsdvDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c8b01d74d62ca3eb3a2beaf62f7d0e1fa2899b7a402b9cb43a12a930137dd28c","last_reissued_at":"2026-07-05T04:05:32.130829Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:32.130829Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Closer Look at How Fine-tuning Changes BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Vivek Srikumar, Yichu Zhou","submitted_at":"2021-06-27T17:01:43Z","abstract_excerpt":"Given the prevalence of pre-trained contextualized representations in today's NLP, there have been many efforts to understand what information they contain, and why they seem to be universally successful. The most common approach to use these representations involves fine-tuning them for an end task. Yet, how fine-tuning changes the underlying embedding space is less studied. In this work, we study the English BERT family and use two probing techniques to analyze how fine-tuning changes the space. We hypothesize that fine-tuning affects classification performance by increasing the distances be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.14282","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.14282/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.14282","created_at":"2026-07-05T04:05:32.130893+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.14282v3","created_at":"2026-07-05T04:05:32.130893+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.14282","created_at":"2026-07-05T04:05:32.130893+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZCYB25GWFSR6","created_at":"2026-07-05T04:05:32.130893+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZCYB25GWFSR6WORL","created_at":"2026-07-05T04:05:32.130893+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZCYB25GW","created_at":"2026-07-05T04:05:32.130893+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.18306","citing_title":"SALMAN: Stability Analysis of Language Models Through the Maps Between Graph-based Manifolds","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6","json":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6.json","graph_json":"https://pith.science/api/pith-number/ZCYB25GWFSR6WORL5L3C67IOD6/graph.json","events_json":"https://pith.science/api/pith-number/ZCYB25GWFSR6WORL5L3C67IOD6/events.json","paper":"https://pith.science/paper/ZCYB25GW"},"agent_actions":{"view_html":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6","download_json":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6.json","view_paper":"https://pith.science/paper/ZCYB25GW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.14282&json=true","fetch_graph":"https://pith.science/api/pith-number/ZCYB25GWFSR6WORL5L3C67IOD6/graph.json","fetch_events":"https://pith.science/api/pith-number/ZCYB25GWFSR6WORL5L3C67IOD6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6/action/storage_attestation","attest_author":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6/action/author_attestation","sign_citation":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6/action/citation_signature","submit_replication":"https://pith.science/pith/ZCYB25GWFSR6WORL5L3C67IOD6/action/replication_record"}},"created_at":"2026-07-05T04:05:32.130893+00:00","updated_at":"2026-07-05T04:05:32.130893+00:00"}