{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B2BHYCRVWVETPH7ZD2FGEFQQX2","short_pith_number":"pith:B2BHYCRV","schema_version":"1.0","canonical_sha256":"0e827c0a35b549379ff91e8a621610be93b7154cd118be32dff980c2d16dabc7","source":{"kind":"arxiv","id":"2506.00694","version":2},"attestation_state":"computed","paper":{"title":"Measuring Faithfulness and Abstention: An Automated Pipeline for Evaluating LLM-Generated 3-ply Case-Based Legal Arguments","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jaromir Savelka, Kevin D. Ashley, Li Zhang, Morgan Gray","submitted_at":"2025-05-31T19:56:40Z","abstract_excerpt":"Large Language Models (LLMs) demonstrate potential in complex legal tasks like argument generation, yet their reliability remains a concern. Building upon pilot work assessing LLM generation of 3-ply legal arguments using human evaluation, this paper introduces an automated pipeline to evaluate LLM performance on this task, specifically focusing on faithfulness (absence of hallucination), factor utilization, and appropriate abstention. We define hallucination as the generation of factors not present in the input case materials and abstention as the model's ability to refrain from generating ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00694","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-31T19:56:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"dd13e609bdb95cbd6d381b39645b8e8c67c328e0d2a0dbedb5d350c287292f14","abstract_canon_sha256":"e2b5c0765152a3fb1fc883ea23b855db2db6ecb5fa5c7e93eb2ad12697bd696f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:29.874197Z","signature_b64":"BBMIVpBqiRwIMJVv527+CPTijemtmxLS+8mjyAPk4XbMe6skuhHSXItfTovNpHo/FQ9WFGXcvTq1GQKB0AYPDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e827c0a35b549379ff91e8a621610be93b7154cd118be32dff980c2d16dabc7","last_reissued_at":"2026-07-05T11:14:29.873763Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:29.873763Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring Faithfulness and Abstention: An Automated Pipeline for Evaluating LLM-Generated 3-ply Case-Based Legal Arguments","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jaromir Savelka, Kevin D. Ashley, Li Zhang, Morgan Gray","submitted_at":"2025-05-31T19:56:40Z","abstract_excerpt":"Large Language Models (LLMs) demonstrate potential in complex legal tasks like argument generation, yet their reliability remains a concern. Building upon pilot work assessing LLM generation of 3-ply legal arguments using human evaluation, this paper introduces an automated pipeline to evaluate LLM performance on this task, specifically focusing on faithfulness (absence of hallucination), factor utilization, and appropriate abstention. We define hallucination as the generation of factors not present in the input case materials and abstention as the model's ability to refrain from generating ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00694","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00694","created_at":"2026-07-05T11:14:29.873823+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00694v2","created_at":"2026-07-05T11:14:29.873823+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00694","created_at":"2026-07-05T11:14:29.873823+00:00"},{"alias_kind":"pith_short_12","alias_value":"B2BHYCRVWVET","created_at":"2026-07-05T11:14:29.873823+00:00"},{"alias_kind":"pith_short_16","alias_value":"B2BHYCRVWVETPH7Z","created_at":"2026-07-05T11:14:29.873823+00:00"},{"alias_kind":"pith_short_8","alias_value":"B2BHYCRV","created_at":"2026-07-05T11:14:29.873823+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2","json":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2.json","graph_json":"https://pith.science/api/pith-number/B2BHYCRVWVETPH7ZD2FGEFQQX2/graph.json","events_json":"https://pith.science/api/pith-number/B2BHYCRVWVETPH7ZD2FGEFQQX2/events.json","paper":"https://pith.science/paper/B2BHYCRV"},"agent_actions":{"view_html":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2","download_json":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2.json","view_paper":"https://pith.science/paper/B2BHYCRV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00694&json=true","fetch_graph":"https://pith.science/api/pith-number/B2BHYCRVWVETPH7ZD2FGEFQQX2/graph.json","fetch_events":"https://pith.science/api/pith-number/B2BHYCRVWVETPH7ZD2FGEFQQX2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2/action/storage_attestation","attest_author":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2/action/author_attestation","sign_citation":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2/action/citation_signature","submit_replication":"https://pith.science/pith/B2BHYCRVWVETPH7ZD2FGEFQQX2/action/replication_record"}},"created_at":"2026-07-05T11:14:29.873823+00:00","updated_at":"2026-07-05T11:14:29.873823+00:00"}