{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ICWBBQLIW4GZKTO6YULSHN7MMW","short_pith_number":"pith:ICWBBQLI","schema_version":"1.0","canonical_sha256":"40ac10c168b70d954ddec51723b7ec65ad1e2e95f7c09c288a6497d1fbeebe34","source":{"kind":"arxiv","id":"2305.07507","version":2},"attestation_state":"computed","paper":{"title":"LeXFiles and LegalLAMA: Facilitating English Multinational Legal Language Model Development","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anders S{\\o}gaard, Catalina Goanta, Daniel Martin Katz, Ilias Chalkidis, Nicolas Garneau","submitted_at":"2023-05-12T14:21:38Z","abstract_excerpt":"In this work, we conduct a detailed analysis on the performance of legal-oriented pre-trained language models (PLMs). We examine the interplay between their original objective, acquired knowledge, and legal language understanding capacities which we define as the upstream, probing, and downstream performance, respectively. We consider not only the models' size but also the pre-training corpora used as important dimensions in our study. To this end, we release a multinational English legal corpus (LeXFiles) and a legal knowledge probing benchmark (LegalLAMA) to facilitate training and detailed "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.07507","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-12T14:21:38Z","cross_cats_sorted":[],"title_canon_sha256":"5a2e1a1703fa373cf746a806736dc0bd940aba898568e21e82e46bd45dd128a3","abstract_canon_sha256":"3c6f25ad38be71906774ac5a4bb7046c8001d785bd1ec7eeabaebebbcfd23c15"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:50.289429Z","signature_b64":"OVIpTaRLrwlg7EmaoLxJ/71liGJaCPEW6PrYasqFdVTIIdtdKWH8PDqjEbAKqsOggB18oNZZrLQlwUmrd0vaBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40ac10c168b70d954ddec51723b7ec65ad1e2e95f7c09c288a6497d1fbeebe34","last_reissued_at":"2026-07-05T06:12:50.288870Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:50.288870Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LeXFiles and LegalLAMA: Facilitating English Multinational Legal Language Model Development","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anders S{\\o}gaard, Catalina Goanta, Daniel Martin Katz, Ilias Chalkidis, Nicolas Garneau","submitted_at":"2023-05-12T14:21:38Z","abstract_excerpt":"In this work, we conduct a detailed analysis on the performance of legal-oriented pre-trained language models (PLMs). We examine the interplay between their original objective, acquired knowledge, and legal language understanding capacities which we define as the upstream, probing, and downstream performance, respectively. We consider not only the models' size but also the pre-training corpora used as important dimensions in our study. To this end, we release a multinational English legal corpus (LeXFiles) and a legal knowledge probing benchmark (LegalLAMA) to facilitate training and detailed "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.07507","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.07507/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.07507","created_at":"2026-07-05T06:12:50.288944+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.07507v2","created_at":"2026-07-05T06:12:50.288944+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.07507","created_at":"2026-07-05T06:12:50.288944+00:00"},{"alias_kind":"pith_short_12","alias_value":"ICWBBQLIW4GZ","created_at":"2026-07-05T06:12:50.288944+00:00"},{"alias_kind":"pith_short_16","alias_value":"ICWBBQLIW4GZKTO6","created_at":"2026-07-05T06:12:50.288944+00:00"},{"alias_kind":"pith_short_8","alias_value":"ICWBBQLI","created_at":"2026-07-05T06:12:50.288944+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.00495","citing_title":"FLoE: Fisher-Based Layer Selection for Efficient Sparse Adaptation of Low-Rank Experts","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW","json":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW.json","graph_json":"https://pith.science/api/pith-number/ICWBBQLIW4GZKTO6YULSHN7MMW/graph.json","events_json":"https://pith.science/api/pith-number/ICWBBQLIW4GZKTO6YULSHN7MMW/events.json","paper":"https://pith.science/paper/ICWBBQLI"},"agent_actions":{"view_html":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW","download_json":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW.json","view_paper":"https://pith.science/paper/ICWBBQLI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.07507&json=true","fetch_graph":"https://pith.science/api/pith-number/ICWBBQLIW4GZKTO6YULSHN7MMW/graph.json","fetch_events":"https://pith.science/api/pith-number/ICWBBQLIW4GZKTO6YULSHN7MMW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW/action/storage_attestation","attest_author":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW/action/author_attestation","sign_citation":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW/action/citation_signature","submit_replication":"https://pith.science/pith/ICWBBQLIW4GZKTO6YULSHN7MMW/action/replication_record"}},"created_at":"2026-07-05T06:12:50.288944+00:00","updated_at":"2026-07-05T06:12:50.288944+00:00"}