{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:CBXRWXSSNYWZ27MB53NBFDHOFZ","short_pith_number":"pith:CBXRWXSS","schema_version":"1.0","canonical_sha256":"106f1b5e526e2d9d7d81eeda128cee2e65732c91f566c9a8c88a02484e81fbc2","source":{"kind":"arxiv","id":"2104.08671","version":3},"attestation_state":"computed","paper":{"title":"When Does Pretraining Help? Assessing Self-Supervised Learning for Law and the CaseHOLD Dataset","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Brandon R. Anderson, Daniel E. Ho, Lucia Zheng, Neel Guha, Peter Henderson","submitted_at":"2021-04-18T00:57:16Z","abstract_excerpt":"While self-supervised learning has made rapid advances in natural language processing, it remains unclear when researchers should engage in resource-intensive domain-specific pretraining (domain pretraining). The law, puzzlingly, has yielded few documented instances of substantial gains to domain pretraining in spite of the fact that legal language is widely seen to be unique. We hypothesize that these existing results stem from the fact that existing legal NLP tasks are too easy and fail to meet conditions for when domain pretraining can help. To address this, we first present CaseHOLD (Case "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.08671","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-04-18T00:57:16Z","cross_cats_sorted":[],"title_canon_sha256":"cee0bf76231ba554bb25080679c1067d6dfe89df38cc8616bf5888d907e53974","abstract_canon_sha256":"9d0997123eda2d1d4c8a2848f08c8765b92398481645ada668fd9760bf53f240"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:55:13.964209Z","signature_b64":"hjkwb5fGnuBMHNG8P3heRhtNYm2BivOmYySgB+1t2RcCAVZa8k65Cqlmw/geef8Av+NujCeW3n21lTwHBSumDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"106f1b5e526e2d9d7d81eeda128cee2e65732c91f566c9a8c88a02484e81fbc2","last_reissued_at":"2026-07-05T02:55:13.963591Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:55:13.963591Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Does Pretraining Help? Assessing Self-Supervised Learning for Law and the CaseHOLD Dataset","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Brandon R. Anderson, Daniel E. Ho, Lucia Zheng, Neel Guha, Peter Henderson","submitted_at":"2021-04-18T00:57:16Z","abstract_excerpt":"While self-supervised learning has made rapid advances in natural language processing, it remains unclear when researchers should engage in resource-intensive domain-specific pretraining (domain pretraining). The law, puzzlingly, has yielded few documented instances of substantial gains to domain pretraining in spite of the fact that legal language is widely seen to be unique. We hypothesize that these existing results stem from the fact that existing legal NLP tasks are too easy and fail to meet conditions for when domain pretraining can help. To address this, we first present CaseHOLD (Case "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.08671","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.08671/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.08671","created_at":"2026-07-05T02:55:13.963660+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.08671v3","created_at":"2026-07-05T02:55:13.963660+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.08671","created_at":"2026-07-05T02:55:13.963660+00:00"},{"alias_kind":"pith_short_12","alias_value":"CBXRWXSSNYWZ","created_at":"2026-07-05T02:55:13.963660+00:00"},{"alias_kind":"pith_short_16","alias_value":"CBXRWXSSNYWZ27MB","created_at":"2026-07-05T02:55:13.963660+00:00"},{"alias_kind":"pith_short_8","alias_value":"CBXRWXSS","created_at":"2026-07-05T02:55:13.963660+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28538","citing_title":"Legal Domain Adaptation of Modern BERT Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17691","citing_title":"Validate Your Authority: Benchmarking LLMs on Multi-Label Precedent Treatment Classification","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22933","citing_title":"How Can AI Augment Access to Justice? Public Defenders' Perspectives on AI Adoption","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19895","citing_title":"Learning When Not to Decide: A Framework for Overcoming Factual Presumptuousness in AI Adjudication","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ","json":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ.json","graph_json":"https://pith.science/api/pith-number/CBXRWXSSNYWZ27MB53NBFDHOFZ/graph.json","events_json":"https://pith.science/api/pith-number/CBXRWXSSNYWZ27MB53NBFDHOFZ/events.json","paper":"https://pith.science/paper/CBXRWXSS"},"agent_actions":{"view_html":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ","download_json":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ.json","view_paper":"https://pith.science/paper/CBXRWXSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.08671&json=true","fetch_graph":"https://pith.science/api/pith-number/CBXRWXSSNYWZ27MB53NBFDHOFZ/graph.json","fetch_events":"https://pith.science/api/pith-number/CBXRWXSSNYWZ27MB53NBFDHOFZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ/action/storage_attestation","attest_author":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ/action/author_attestation","sign_citation":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ/action/citation_signature","submit_replication":"https://pith.science/pith/CBXRWXSSNYWZ27MB53NBFDHOFZ/action/replication_record"}},"created_at":"2026-07-05T02:55:13.963660+00:00","updated_at":"2026-07-05T02:55:13.963660+00:00"}