{"as_of":"2026-08-01T13:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:46a8c25203a1b2dc701a1178bb8030ea46a88606d3f5ce8a3099336df0de40a1","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-01T06:32:01.292127+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T04:04:13.822723Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2510.26083","last_updated":"2026-04-08T13:55:00Z","snapshot_observed_at":"2026-07-06T22:34:28.484520Z","submitted_at":"2025-10-30T02:41:54Z","title":"Nirvana: A Specialized Generalist Model With Task-Aware Memory Mechanism","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T03:05:06.069642Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2510.26083"},"observation_digest":"sha256:2c1c4d12fa45b2d6cd4bd9339a94eddc4bd66adf8518298e80ffd420d403cbb3","observation_id":"19fa69a8-2765-403f-8f13-90135d012582","resolution":{"observed_at":"2026-05-18T03:05:48.097934Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2511.05501","last_updated":"2026-04-28T16:06:15Z","snapshot_observed_at":"2026-07-06T22:35:13.699594Z","submitted_at":"2025-09-30T21:36:23Z","title":"Towards Real-World Validity in Generative AI Benchmarks: Understanding and Designing Domain-Centered Evaluations for Journalism Practitioners","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T10:59:16.139525Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2511.05501"},"observation_digest":"sha256:16c4beba5cab9755a4fb2e69d4cec3a48dc303995b81e7ac45914804388f33c2","observation_id":"11d79624-30eb-429a-9918-0e82de232510","resolution":{"observed_at":"2026-05-18T11:01:17.029897Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2603.06610","last_updated":"2026-05-22T08:27:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-19T09:46:24Z","title":"CapTrack: Multifaceted Evaluation of Forgetting in LLM Post-Training","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T06:40:51.046965Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2603.06610"},"observation_digest":"sha256:fe0056ae299ef7f34a534078a412d332a5a236750868207c3f0f0a0e65ca5afe","observation_id":"2fb5a109-91ee-4c48-9816-733b21e74e4e","resolution":{"observed_at":"2026-05-25T06:45:25.582582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.10718","last_updated":"2026-04-12T16:28:51Z","snapshot_observed_at":"2026-07-31T09:28:04.079811Z","submitted_at":"2026-04-12T16:28:51Z","title":"SciPredict: Can LLMs Predict the Outcomes of Scientific Experiments in Natural Sciences?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T15:55:34.768853Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.10718"},"observation_digest":"sha256:a40155a6ff6095832a80fe55ddeaa68bc8de6d3baa47bad8a925415c105cab7f","observation_id":"0770fa1b-4680-4353-8f8d-7a5cd87c9fbc","resolution":{"observed_at":"2026-05-11T09:36:03.821732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.18878","last_updated":"2026-04-20T22:00:02Z","snapshot_observed_at":"2026-07-06T23:05:39.724237Z","submitted_at":"2026-04-20T22:00:02Z","title":"LegalBench-BR: A Benchmark for Evaluating Large Language Models on Brazilian Legal Decision Classification","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T04:27:23.053818Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.18878"},"observation_digest":"sha256:398f5239ef171d47365d39e63da24b68f34984c570a7db8a2690603736ed82ba","observation_id":"e0c78989-4ba3-4b36-a240-7692ee3934dc","resolution":{"observed_at":"2026-05-11T11:56:26.755806Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.19820","last_updated":"2026-04-19T07:09:42Z","snapshot_observed_at":"2026-07-06T23:06:21.774788Z","submitted_at":"2026-04-19T07:09:42Z","title":"KnowPilot: Your Knowledge-Driven Copilot for Domain Tasks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T06:15:44.360621Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.19820"},"observation_digest":"sha256:8129e6c1653a688d4b5849e6a72657d7a1cdb68c985d60a70a95b094625b55a5","observation_id":"e4704eac-14b1-4ed5-9f66-7916bd338bd6","resolution":{"observed_at":"2026-05-10T06:16:20.746297Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.19895","last_updated":"2026-04-21T18:17:08Z","snapshot_observed_at":"2026-07-06T23:06:26.596990Z","submitted_at":"2026-04-21T18:17:08Z","title":"Learning When Not to Decide: A Framework for Overcoming Factual Presumptuousness in AI Adjudication","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T02:08:24.770003Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.19895"},"observation_digest":"sha256:b886f6d1e5cdc55f26d01f12890827ba7a5071eb45b52455d278aa98476c51e9","observation_id":"398924d6-1683-4640-a4fd-9614f707dddd","resolution":{"observed_at":"2026-05-10T02:11:57.255230Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.20726","last_updated":"2026-04-23T08:13:50Z","snapshot_observed_at":"2026-07-06T23:07:25.185196Z","submitted_at":"2026-04-22T16:12:36Z","title":"Exploiting LLM-as-a-Judge Disposition on Free Text Legal QA via Prompt Optimization","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T01:10:56.318664Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.20726"},"observation_digest":"sha256:526effa465a57bd3a75ea7a2db6e659f994de7acc5997391e489497673cf02f0","observation_id":"a7818266-5c70-4bb6-b044-e164685aefa9","resolution":{"observed_at":"2026-05-11T13:41:10.502069Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.23511","last_updated":"2026-04-26T03:13:47Z","snapshot_observed_at":"2026-07-06T23:09:43.659053Z","submitted_at":"2026-04-26T03:13:47Z","title":"Breaking the Secret: Economic Interventions for Combating Collusion in Embodied Multi-Agent Systems","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-08T06:13:17.633146Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.23511"},"observation_digest":"sha256:158414c719596f6b36a860a1241df512aa3b5cc337e34e8e3d059cfab3bfb9b9","observation_id":"cf967cc5-c82b-4272-8813-f631e1ac7063","resolution":{"observed_at":"2026-05-11T21:16:16.378379Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.23730","last_updated":"2026-04-26T14:15:43Z","snapshot_observed_at":"2026-07-31T16:26:34.698889Z","submitted_at":"2026-04-26T14:15:43Z","title":"Expert Evaluation of LLM's Open-Ended Legal Reasoning on the Japanese Bar Exam Writing Task","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T05:59:54.438189Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.23730"},"observation_digest":"sha256:193e806a2a637094022260b79bfa0d283afda991b9fedf611b4970eb3d747961","observation_id":"1fd275bc-c7b0-44e8-8671-6e4c428bb6cd","resolution":{"observed_at":"2026-05-11T21:16:36.010805Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2604.24902","last_updated":"2026-04-27T18:34:08Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T18:34:08Z","title":"Safety Drift After Fine-Tuning: Evidence from High-Stakes Domains","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-07T17:53:57.169962Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2604.24902"},"observation_digest":"sha256:d2b86df1fa7a233a4c72c58a1f765e80767e2104cb2591682d732f789bb8a741","observation_id":"ebcbdeb7-2d0d-4972-9bbf-2ed349ec7aad","resolution":{"observed_at":"2026-05-11T23:11:19.337354Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.07096","last_updated":"2026-06-04T16:41:18Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T01:24:06Z","title":"Query-efficient model evaluation using cached responses","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-30T23:28:47.530333Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.07096"},"observation_digest":"sha256:9b42931436e2a3ef38e2e37acedbb36cbff99eca63a4c8ddb1d8bdd5ca3e3b89","observation_id":"11fe66d3-1f22-46f5-8cfc-2863a2f81a9c","resolution":{"observed_at":"2026-06-30T23:35:07.682388Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.09611","last_updated":"2026-05-10T15:48:38Z","snapshot_observed_at":"2026-07-06T23:21:39.966502Z","submitted_at":"2026-05-10T15:48:38Z","title":"Byte-Exact Deduplication in Retrieval-Augmented Generation: A Three-Regime Empirical Analysis Across Public Benchmarks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T04:18:11.836537Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.09611"},"observation_digest":"sha256:a9bbe1560235a6a63800196ed6c9ca8173ae757eb547e4b4243f65eb39eb319f","observation_id":"2d676823-a8f4-4cee-8652-9637cce9eb42","resolution":{"observed_at":"2026-05-12T06:26:24.208127Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.24454","last_updated":"2026-05-23T08:03:31Z","snapshot_observed_at":"2026-07-06T23:34:33.493637Z","submitted_at":"2026-05-23T08:03:31Z","title":"Decompose-and-Refine: Structured Legal Question Answering with Parametric Retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T13:23:16.122654Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.24454"},"observation_digest":"sha256:1b6d823bed0eb36129b3662d43dc75422fcbc9ec4425d2974092503be6701279","observation_id":"f34661f7-0ff0-484e-953f-b59557e57208","resolution":{"observed_at":"2026-06-30T13:24:39.984446Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.25474","last_updated":"2026-05-25T06:26:46Z","snapshot_observed_at":"2026-07-06T23:35:26.687529Z","submitted_at":"2026-05-25T06:26:46Z","title":"TypedCSIP: Typed Counterfactual Pretraining for Chinese Legislative Conflict Classification","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T22:06:28.247167Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.25474"},"observation_digest":"sha256:b8cf2263214164db16f235bd6ce34e43881913f879db3fb9a20703b252d19cce","observation_id":"0a20cf91-1c6c-49fb-80ab-640f03d4acf5","resolution":{"observed_at":"2026-06-29T22:14:00.218650Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.29170","last_updated":"2026-05-27T23:12:20Z","snapshot_observed_at":"2026-07-06T23:38:37.178378Z","submitted_at":"2026-05-27T23:12:20Z","title":"UA-Legal-Bench: A Benchmark for Evaluating Large Language Models on Ukrainian Legal Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T12:15:12.570661Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.29170"},"observation_digest":"sha256:e8aed3a080850a2fe6c5f47e95a3cd73b7913e7aeed9c0c201b3238eadfaab7b","observation_id":"762cbd88-088c-4f89-a677-07649586ee0f","resolution":{"observed_at":"2026-06-29T12:23:24.527606Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2605.29738","last_updated":"2026-05-28T10:31:37Z","snapshot_observed_at":"2026-07-31T22:45:33.205335Z","submitted_at":"2026-05-28T10:31:37Z","title":"Multi-Legal-Bench: Evaluating LLMs on Legal Reasoning Across Jurisdictions, Languages, and Legal Traditions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T07:24:08.269519Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2605.29738"},"observation_digest":"sha256:96119de5757fdd2d6e7a18c8f3e2ad30c4a6a27fc48fd1a1d7b4b692375d1580","observation_id":"89084092-98aa-40f3-ae4c-8eaade03f412","resolution":{"observed_at":"2026-06-29T08:33:15.935573Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.00898","last_updated":"2026-05-30T21:22:47Z","snapshot_observed_at":"2026-07-06T23:41:34.459700Z","submitted_at":"2026-05-30T21:22:47Z","title":"Citation Grounding: Detecting and Reducing LLM Citation Hallucinations via Legal Citation Graphs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T18:37:22.299236Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.00898"},"observation_digest":"sha256:632d26c16cb8e2b7f11e20bc47fbc97848d7683b6b34ea3f47c8751505c86e32","observation_id":"0e17b22c-bbd3-4541-94e6-7a937abef3c6","resolution":{"observed_at":"2026-06-28T20:32:37.522670Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.08036","last_updated":"2026-06-06T07:56:40Z","snapshot_observed_at":"2026-07-06T23:47:33.680573Z","submitted_at":"2026-06-06T07:56:40Z","title":"GIScholarBench: Benchmarking LLM Overconfidence in GIS Research","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T19:17:26.825374Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.08036"},"observation_digest":"sha256:b1bf560dd409c9ac9b0c052b6be372c438a4a9e1f26738bf301c11e029700e52","observation_id":"c02298e6-473d-41c6-a231-e1f3028cde94","resolution":{"observed_at":"2026-07-02T22:07:25.908133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.10457","last_updated":"2026-06-09T06:05:29Z","snapshot_observed_at":"2026-07-06T23:49:37.345711Z","submitted_at":"2026-06-09T06:05:29Z","title":"Trace2Policy: From Expert Behavior Traces to Self-Evolving Decision Agents","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-27T13:19:10.714343Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.10457"},"observation_digest":"sha256:aea0cf30075c61e3806435964fdf6b8d2a37335c9f1dabb8094692b2008ca2b5","observation_id":"11bcab69-c3de-42e6-b065-a10de6e042fb","resolution":{"observed_at":"2026-07-03T05:27:39.663246Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.18021","last_updated":"2026-06-16T15:02:37Z","snapshot_observed_at":"2026-07-06T23:53:30.953084Z","submitted_at":"2026-06-16T15:02:37Z","title":"LegalHalluLens: Typed Hallucination Auditing and Calibrated Multi-Agent Debate for Trustworthy Legal AI","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-27T00:34:53.489694Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.18021"},"observation_digest":"sha256:fac64a39409735ef032acc87bad077a69ce4f3c91b0738b78b3b9f9741169676","observation_id":"39b03420-626a-435f-bb7b-9451693948dd","resolution":{"observed_at":"2026-07-03T21:28:58.612291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.18158","last_updated":"2026-06-16T16:57:12Z","snapshot_observed_at":"2026-07-06T23:53:40.698123Z","submitted_at":"2026-06-16T16:57:12Z","title":"The Measurement Gap in the Automation of EU Law: Benchmarking Doctrinal Legal Reasoning under the EU AI Act","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-26T22:16:25.919667Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.18158"},"observation_digest":"sha256:d4c4a4d61665ea53af4938a97d27008da928a4b3e818e0fc749edcca13edda9c","observation_id":"c454b269-996a-4f29-b290-fd070686884d","resolution":{"observed_at":"2026-07-03T23:29:02.495106Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.21121","last_updated":"2026-06-19T05:47:44Z","snapshot_observed_at":"2026-07-06T23:56:12.374739Z","submitted_at":"2026-06-19T05:47:44Z","title":"Answer Engineering: Local Trajectory Editing for Protocol-Constrained Decision Making in Large Language Models","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-06-26T14:13:13.678245Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.21121"},"observation_digest":"sha256:5a8cf8b308f26b351e732e1de2bc7e1f6d2ff9e603d7918622b684aa9fa709da","observation_id":"5d0442ac-aa2b-455d-935b-99fef5a8c367","resolution":{"observed_at":"2026-07-04T06:49:37.576858Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.22778","last_updated":"2026-06-22T02:42:06Z","snapshot_observed_at":"2026-07-06T23:57:38.036151Z","submitted_at":"2026-06-22T02:42:06Z","title":"HAKARI-Bench: A Lightweight Benchmark for Comparing Retrieval Architectures and Efficiency Settings under Unified Conditions","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-06-26T07:22:34.547816Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.22778"},"observation_digest":"sha256:f1c0b21f4b4d95d9223ec7cdc789d6b0c6fc39db08b764a02484705bca0ca112","observation_id":"9675f5fb-93ac-48c5-bd49-3d74a374a686","resolution":{"observed_at":"2026-07-04T11:59:51.263031Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.23716","last_updated":"2026-06-16T14:19:04Z","snapshot_observed_at":"2026-07-06T23:58:25.400508Z","submitted_at":"2026-06-16T14:19:04Z","title":"Legal Reasoning Is Not Lawyering: Rethinking Legal Benchmarks for Pro Se Access to Justice","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-06-26T22:26:54.239505Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.23716"},"observation_digest":"sha256:7c4fe3e53fac45e0b7e64787f90a238b847e45cb8291ab55c4edccc1cd9414a8","observation_id":"77a56eeb-1020-40e3-b02d-1e0bc4c61849","resolution":{"observed_at":"2026-07-03T23:19:03.985515Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2308.11462","doi":"10.48550/arxiv.2308.11462","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URLhttps://doi.org/10.48550/arxiv.2308.11462","venue":null,"work_id":"39567bec-304b-4b6e-9c07-4269cd9d1af6","year":2023},"citing_paper":{"arxiv_id":"2606.26346","last_updated":"2026-06-24T19:38:21Z","snapshot_observed_at":"2026-08-01T10:31:00.655586Z","submitted_at":"2026-06-24T19:38:21Z","title":"How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T01:34:10.103638Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2606.26346"},"observation_digest":"sha256:b76676868616ec32de96fcadf0b747715199c31e32580f69ef2a11dc0c70303f","observation_id":"58c35bc4-9e48-487d-bae7-3e457539c130","resolution":{"observed_at":"2026-07-04T15:29:56.715847Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-14T10:38:03.549728Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10588","last_updated":"2026-07-12T05:57:22Z","snapshot_observed_at":"2026-07-16T23:18:51.643051Z","submitted_at":"2026-07-12T05:57:22Z","title":"Constraint-Aware Hierarchical Search for Regulation-Driven Fine-Grained Classification","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-07-14T10:38:03.549728Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.10588"},"observation_digest":"sha256:bb5cd9527dd762c82baf55b9a27c4a1d763c1452ed6f9134ee43170d6ddde3f2","observation_id":"b2379fd8-9d26-4202-822d-e816f773e61c","resolution":{"observed_at":"2026-07-14T10:38:03.549728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-01T04:04:13.822723Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22954","last_updated":"2026-07-24T23:43:35Z","snapshot_observed_at":"2026-08-01T04:04:12.778417Z","submitted_at":"2026-07-24T23:43:35Z","title":"Toward Automated Detection of Documentation Inconsistencies in Electronic Health Records","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T04:04:13.822723Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.22954"},"observation_digest":"sha256:168cbf3d0dbb7c0ba0a2bede0d7e3f49697b09e82dd88c841dcf3ce6d0d96b45","observation_id":"0b3da14e-438a-45a5-b651-7b16132da10e","resolution":{"observed_at":"2026-08-01T04:04:13.822723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-07-30T23:41:20.109043Z","title":"Ho, Christopher Ré, Adam Chilton, Alex Chohlas-Wood, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.23386","last_updated":"2026-07-25T22:40:02Z","snapshot_observed_at":"2026-07-30T23:54:09.847287Z","submitted_at":"2026-07-25T22:40:02Z","title":"Confidently Wrong: Exception Chain Collapse in Frontier LLM Rule Evaluation","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-07-30T23:41:20.109043Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.23386"},"observation_digest":"sha256:df419ee5ec964e5d5cabc2abe69d7b11e6d6860fb9485ace044747ca2922c707","observation_id":"10731c04-647d-4086-9b51-aa77a094d1c9","resolution":{"observed_at":"2026-07-30T23:41:20.109043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11462","snapshot_observed_at":"2026-08-01T01:32:21.858955Z","title":"2023 , bdsk-url-1 =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.27783","last_updated":"2026-07-30T07:14:50Z","snapshot_observed_at":"2026-08-01T12:45:39.908410Z","submitted_at":"2026-07-30T07:14:50Z","title":"Reasoning Consensus: Structural Ensembling of LLM Reasoning via Weighted DAG Aggregation","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-01T01:32:21.858955Z"},"links":{"cited_paper":"/paper/2308.11462","citing_paper":"/paper/2607.27783"},"observation_digest":"sha256:3b3317558f863347767e3b3be7b9dab593c1088879d49abc358bf18548d993e5","observation_id":"1557386e-86b7-4762-b185-713397cfc408","resolution":{"observed_at":"2026-08-01T01:32:21.858955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2308.11462/citation-record","integrity":"/paper/2308.11462/integrity","json":"/paper/2308.11462/citation-record.json","paper":"/paper/2308.11462"},"outbound":[],"paper":{"arxiv_id":"2308.11462","last_updated":"2023-08-20T22:08:03Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:09:05.106099Z","submitted_at":"2023-08-20T22:08:03Z","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"thesis":"As of 1 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2308.11462."}