{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OG32KAPF46GXVZO5DSK6DWQE5V","short_pith_number":"pith:OG32KAPF","schema_version":"1.0","canonical_sha256":"71b7a501e5e78d7ae5dd1c95e1da04ed6fff9b76f878da813a6da716db2efc82","source":{"kind":"arxiv","id":"2410.09083","version":2},"attestation_state":"computed","paper":{"title":"Evaluating the Correctness of Inference Patterns Used by LLMs for Judgment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Dongrui Liu, Kun Kuang, Lu Chen, Qihan Ren, Quanshi Zhang, Shuai Zhao, Yixing Li, Yuxuan Huang, Zilong Zheng","submitted_at":"2024-10-06T08:33:39Z","abstract_excerpt":"This paper presents a method to analyze the inference patterns used by Large Language Models (LLMs) for judgment in a case study on legal LLMs, so as to identify potential incorrect representations of the LLM, according to human domain knowledge. Unlike traditional evaluations on language generation results, we propose to evaluate the correctness of the detailed inference patterns of an LLM behind its seemingly correct outputs. To this end, we quantify the interactions between input phrases used by the LLM as primitive inference patterns, because recent theoretical achievements have proven sev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09083","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-10-06T08:33:39Z","cross_cats_sorted":["cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"2236624a6bff566b45deabdb76e7a9dab09b9e272c1841491f6cc05e68de3279","abstract_canon_sha256":"4babb34bceab4314959b706e00864a1311f2ef2734894bf0786dc6b6613bbb8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:09.359758Z","signature_b64":"oRZ204AVTEuYufcLCZ+5S76Rn3Mfpp7xA0UXdy9cNBiJmM9VDuXbmD8mHoxBEeiyYb66DvXz8fuS+zIFyyMwBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"71b7a501e5e78d7ae5dd1c95e1da04ed6fff9b76f878da813a6da716db2efc82","last_reissued_at":"2026-07-05T11:06:09.359187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:09.359187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating the Correctness of Inference Patterns Used by LLMs for Judgment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Dongrui Liu, Kun Kuang, Lu Chen, Qihan Ren, Quanshi Zhang, Shuai Zhao, Yixing Li, Yuxuan Huang, Zilong Zheng","submitted_at":"2024-10-06T08:33:39Z","abstract_excerpt":"This paper presents a method to analyze the inference patterns used by Large Language Models (LLMs) for judgment in a case study on legal LLMs, so as to identify potential incorrect representations of the LLM, according to human domain knowledge. Unlike traditional evaluations on language generation results, we propose to evaluate the correctness of the detailed inference patterns of an LLM behind its seemingly correct outputs. To this end, we quantify the interactions between input phrases used by the LLM as primitive inference patterns, because recent theoretical achievements have proven sev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09083","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09083/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09083","created_at":"2026-07-05T11:06:09.359249+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09083v2","created_at":"2026-07-05T11:06:09.359249+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09083","created_at":"2026-07-05T11:06:09.359249+00:00"},{"alias_kind":"pith_short_12","alias_value":"OG32KAPF46GX","created_at":"2026-07-05T11:06:09.359249+00:00"},{"alias_kind":"pith_short_16","alias_value":"OG32KAPF46GXVZO5","created_at":"2026-07-05T11:06:09.359249+00:00"},{"alias_kind":"pith_short_8","alias_value":"OG32KAPF","created_at":"2026-07-05T11:06:09.359249+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V","json":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V.json","graph_json":"https://pith.science/api/pith-number/OG32KAPF46GXVZO5DSK6DWQE5V/graph.json","events_json":"https://pith.science/api/pith-number/OG32KAPF46GXVZO5DSK6DWQE5V/events.json","paper":"https://pith.science/paper/OG32KAPF"},"agent_actions":{"view_html":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V","download_json":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V.json","view_paper":"https://pith.science/paper/OG32KAPF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09083&json=true","fetch_graph":"https://pith.science/api/pith-number/OG32KAPF46GXVZO5DSK6DWQE5V/graph.json","fetch_events":"https://pith.science/api/pith-number/OG32KAPF46GXVZO5DSK6DWQE5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V/action/storage_attestation","attest_author":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V/action/author_attestation","sign_citation":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V/action/citation_signature","submit_replication":"https://pith.science/pith/OG32KAPF46GXVZO5DSK6DWQE5V/action/replication_record"}},"created_at":"2026-07-05T11:06:09.359249+00:00","updated_at":"2026-07-05T11:06:09.359249+00:00"}