{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2C3RJRXACF3JH6YRAF2ZDA3CX7","short_pith_number":"pith:2C3RJRXA","schema_version":"1.0","canonical_sha256":"d0b714c6e0117693fb110175918362bfe4eff1e0e52e8701f3968a9d03af3465","source":{"kind":"arxiv","id":"2505.02172","version":3},"attestation_state":"computed","paper":{"title":"Identifying Legal Holdings with LLMs: A Systematic Study of Performance, Scale, and Memorization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuck Arvin","submitted_at":"2025-05-04T16:24:12Z","abstract_excerpt":"As large language models (LLMs) continue to advance in capabilities, it is essential to assess how they perform on established benchmarks. In this study, we present a suite of experiments to assess the performance of modern LLMs (ranging from 3B to 90B+ parameters) on CaseHOLD, a legal benchmark dataset for identifying case holdings. Our experiments demonstrate scaling effects - performance on this task improves with model size, with more capable models like GPT4o and AmazonNovaPro achieving macro F1 scores of 0.744 and 0.720 respectively. These scores are competitive with the best published r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.02172","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-04T16:24:12Z","cross_cats_sorted":[],"title_canon_sha256":"9a1f96b38b4813644d20e33c7e351233cbeb469d1f7fa3ca4d1c7b2831daa70a","abstract_canon_sha256":"876532b8698cc61f7ff1e15d78d119f4586ed188f59ec40ea9f8e5924876cf88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:12.625261Z","signature_b64":"uhgu8TDJfMguJlqoFm5GK9aw5LEecNSQsBEyxNGfffjgL0aK+MXmeAx0hQIqP+w1D9cbRk/Y+bwfvUH7bu68CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d0b714c6e0117693fb110175918362bfe4eff1e0e52e8701f3968a9d03af3465","last_reissued_at":"2026-07-05T11:09:12.624758Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:12.624758Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Identifying Legal Holdings with LLMs: A Systematic Study of Performance, Scale, and Memorization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuck Arvin","submitted_at":"2025-05-04T16:24:12Z","abstract_excerpt":"As large language models (LLMs) continue to advance in capabilities, it is essential to assess how they perform on established benchmarks. In this study, we present a suite of experiments to assess the performance of modern LLMs (ranging from 3B to 90B+ parameters) on CaseHOLD, a legal benchmark dataset for identifying case holdings. Our experiments demonstrate scaling effects - performance on this task improves with model size, with more capable models like GPT4o and AmazonNovaPro achieving macro F1 scores of 0.744 and 0.720 respectively. These scores are competitive with the best published r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.02172","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.02172/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.02172","created_at":"2026-07-05T11:09:12.624816+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.02172v3","created_at":"2026-07-05T11:09:12.624816+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.02172","created_at":"2026-07-05T11:09:12.624816+00:00"},{"alias_kind":"pith_short_12","alias_value":"2C3RJRXACF3J","created_at":"2026-07-05T11:09:12.624816+00:00"},{"alias_kind":"pith_short_16","alias_value":"2C3RJRXACF3JH6YR","created_at":"2026-07-05T11:09:12.624816+00:00"},{"alias_kind":"pith_short_8","alias_value":"2C3RJRXA","created_at":"2026-07-05T11:09:12.624816+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17674","citing_title":"Towards Intelligent Legal Document Analysis: CNN-Driven Classification of Case Law Texts","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7","json":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7.json","graph_json":"https://pith.science/api/pith-number/2C3RJRXACF3JH6YRAF2ZDA3CX7/graph.json","events_json":"https://pith.science/api/pith-number/2C3RJRXACF3JH6YRAF2ZDA3CX7/events.json","paper":"https://pith.science/paper/2C3RJRXA"},"agent_actions":{"view_html":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7","download_json":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7.json","view_paper":"https://pith.science/paper/2C3RJRXA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.02172&json=true","fetch_graph":"https://pith.science/api/pith-number/2C3RJRXACF3JH6YRAF2ZDA3CX7/graph.json","fetch_events":"https://pith.science/api/pith-number/2C3RJRXACF3JH6YRAF2ZDA3CX7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7/action/storage_attestation","attest_author":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7/action/author_attestation","sign_citation":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7/action/citation_signature","submit_replication":"https://pith.science/pith/2C3RJRXACF3JH6YRAF2ZDA3CX7/action/replication_record"}},"created_at":"2026-07-05T11:09:12.624816+00:00","updated_at":"2026-07-05T11:09:12.624816+00:00"}