{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BFWUVYHYN4IUCVHHFDBYCAD2IR","short_pith_number":"pith:BFWUVYHY","schema_version":"1.0","canonical_sha256":"096d4ae0f86f114154e728c381007a444501a43b79bd21dcb0d062310b6009c3","source":{"kind":"arxiv","id":"2310.11761","version":1},"attestation_state":"computed","paper":{"title":"A Comprehensive Evaluation of Large Language Models on Legal Judgment Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ruihao Shui, Tat-Seng Chua, Xiang Wang, Yixin Cao","submitted_at":"2023-10-18T07:38:04Z","abstract_excerpt":"Large language models (LLMs) have demonstrated great potential for domain-specific applications, such as the law domain. However, recent disputes over GPT-4's law evaluation raise questions concerning their performance in real-world legal tasks. To systematically investigate their competency in the law, we design practical baseline solutions based on LLMs and test on the task of legal judgment prediction. In our solutions, LLMs can work alone to answer open questions or coordinate with an information retrieval (IR) system to learn from similar cases or solve simplified multi-choice questions. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.11761","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-18T07:38:04Z","cross_cats_sorted":[],"title_canon_sha256":"4a8506911d661ba5e60893125c8c40ef20e0a9c4aa0ab4e881335d3cd9f35a00","abstract_canon_sha256":"a9df0e2f3a92148cd5b097278ba1e5cf000edd4d22bd613bfbacda1659cf636b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:11.850865Z","signature_b64":"ZZATanlB5flSgQF/rXlds3SvD4eZ+1GyTfhtR7aMsBeuH+Vjl0vVY17UQeWxNot4lPKrs4XuMKEJAvvYxBSSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"096d4ae0f86f114154e728c381007a444501a43b79bd21dcb0d062310b6009c3","last_reissued_at":"2026-07-05T07:02:11.850436Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:11.850436Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Evaluation of Large Language Models on Legal Judgment Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ruihao Shui, Tat-Seng Chua, Xiang Wang, Yixin Cao","submitted_at":"2023-10-18T07:38:04Z","abstract_excerpt":"Large language models (LLMs) have demonstrated great potential for domain-specific applications, such as the law domain. However, recent disputes over GPT-4's law evaluation raise questions concerning their performance in real-world legal tasks. To systematically investigate their competency in the law, we design practical baseline solutions based on LLMs and test on the task of legal judgment prediction. In our solutions, LLMs can work alone to answer open questions or coordinate with an information retrieval (IR) system to learn from similar cases or solve simplified multi-choice questions. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.11761","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.11761/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.11761","created_at":"2026-07-05T07:02:11.850493+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.11761v1","created_at":"2026-07-05T07:02:11.850493+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.11761","created_at":"2026-07-05T07:02:11.850493+00:00"},{"alias_kind":"pith_short_12","alias_value":"BFWUVYHYN4IU","created_at":"2026-07-05T07:02:11.850493+00:00"},{"alias_kind":"pith_short_16","alias_value":"BFWUVYHYN4IUCVHH","created_at":"2026-07-05T07:02:11.850493+00:00"},{"alias_kind":"pith_short_8","alias_value":"BFWUVYHY","created_at":"2026-07-05T07:02:11.850493+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07748","citing_title":"When Large Language Models Meet Law: Dual-Lens Taxonomy, Technical Advances, and Ethical Governance","ref_index":163,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR","json":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR.json","graph_json":"https://pith.science/api/pith-number/BFWUVYHYN4IUCVHHFDBYCAD2IR/graph.json","events_json":"https://pith.science/api/pith-number/BFWUVYHYN4IUCVHHFDBYCAD2IR/events.json","paper":"https://pith.science/paper/BFWUVYHY"},"agent_actions":{"view_html":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR","download_json":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR.json","view_paper":"https://pith.science/paper/BFWUVYHY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.11761&json=true","fetch_graph":"https://pith.science/api/pith-number/BFWUVYHYN4IUCVHHFDBYCAD2IR/graph.json","fetch_events":"https://pith.science/api/pith-number/BFWUVYHYN4IUCVHHFDBYCAD2IR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR/action/storage_attestation","attest_author":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR/action/author_attestation","sign_citation":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR/action/citation_signature","submit_replication":"https://pith.science/pith/BFWUVYHYN4IUCVHHFDBYCAD2IR/action/replication_record"}},"created_at":"2026-07-05T07:02:11.850493+00:00","updated_at":"2026-07-05T07:02:11.850493+00:00"}