{"as_of":"2026-08-09T08:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:237a77474ec29daa251d06011c1cb4744dc9030e2e25089a973a6a22cf7ab354","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T17:26:38.394751Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.10963/citation-record","integrity":"/paper/2509.10963/integrity","json":"/paper/2509.10963/citation-record.json","paper":"/paper/2509.10963"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.274545Z","title":"A., and Sheikh, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.274545Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:eca538fb92e89eeb2b38d7e51c6e0c090f8e94fc7a531256a7c3fbd2872199f4","observation_id":"c83ea906-353f-44e0-b624-c9fe36a54cbd","resolution":{"observed_at":"2026-08-04T17:26:38.274545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-04T17:26:38.281877Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.281877Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:fc25cd6f2b30d6b237df26cc3aa57656a0989fde90e7d4f96b0199cc21f33b81","observation_id":"6f44d130-7c88-43dc-91fd-b7eabc4621cb","resolution":{"observed_at":"2026-08-04T17:26:38.281877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.08784","last_updated":"2026-07-24T02:23:43Z","snapshot_observed_at":"2026-08-07T15:45:39.857890Z","submitted_at":"2025-05-13T17:58:16Z","title":"PCS-UQ: Uncertainty Quantification via the Predictability-Computability-Stability Framework","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.08784","snapshot_observed_at":"2026-08-04T17:26:38.286778Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.286778Z"},"links":{"cited_paper":"/paper/2505.08784","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:a0807ba2af66a86138ac1bef5d0ff001b771d9165e31340d62cf815905e1458c","observation_id":"92eb883e-14b0-40b7-ba1f-cf10bb00367e","resolution":{"observed_at":"2026-08-04T17:26:38.286778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.291673Z","title":null,"venue":null,"work_id":null,"year":1980},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.291673Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:3d30d3ee971d5883fb116c45e18d715680957ed4540b305f64f08a2e757f327e","observation_id":"262ac843-df59-4488-b0a3-78a6a830d550","resolution":{"observed_at":"2026-08-04T17:26:38.291673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.295941Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.295941Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:7d825fbccb376a8188f7d23a248eeb9aa9746e880bec58bbd862e3abd58b9ce3","observation_id":"5ff41350-95a1-4dd6-a4b6-e4441380da6f","resolution":{"observed_at":"2026-08-04T17:26:38.295941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08939","last_updated":"2024-05-28T04:32:09Z","snapshot_observed_at":"2026-08-06T03:42:39.770101Z","submitted_at":"2024-02-14T04:50:18Z","title":"Premise Order Matters in Reasoning with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08939","snapshot_observed_at":"2026-08-04T17:26:38.300315Z","title":"A., Wang, X., and Zhou, D","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.300315Z"},"links":{"cited_paper":"/paper/2402.08939","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:6d4563ec58f31e24c2d090bac93b2d47dfca8acbbef126ddb9e95116fcc9643f","observation_id":"ee658f95-f988-4f5e-81c7-455f8ada20ba","resolution":{"observed_at":"2026-08-04T17:26:38.300315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.305563Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.305563Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:1550e649c0b8184c9b4a8e4b33020ff735854fe35f55f7ae3a0011cc3bef5162","observation_id":"b4326488-5829-44d2-8156-94dab54f8307","resolution":{"observed_at":"2026-08-04T17:26:38.305563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12066","last_updated":"2024-06-19T03:59:41Z","snapshot_observed_at":"2026-08-09T05:32:22.143870Z","submitted_at":"2024-06-17T20:09:24Z","title":"Language Models are Surprisingly Fragile to Drug Names in Biomedical Benchmarks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12066","snapshot_observed_at":"2026-08-04T17:26:38.309880Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.309880Z"},"links":{"cited_paper":"/paper/2406.12066","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:d5559eaf9f56a8f43060ecbfa2f02a82894e95c1817a061d28220510fc406874","observation_id":"17d0b253-d3ad-4e5a-b560-e53b2b545bf9","resolution":{"observed_at":"2026-08-04T17:26:38.309880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-04T17:26:38.314279Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.314279Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:f4b12eb4917fee3c941e47b7b247314de5d4ba7d2cb44fa134f4908bbc6ddf44","observation_id":"60088fa0-5e18-4d49-aeaf-b07d58503228","resolution":{"observed_at":"2026-08-04T17:26:38.314279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.319011Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.319011Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:ad551378e72810a3518e178a69685febca5f4afcd022d94a9c52abd932d187de","observation_id":"38076c9e-4f16-45cd-a402-be49e2cfe0ee","resolution":{"observed_at":"2026-08-04T17:26:38.319011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08913","last_updated":"2023-09-16T07:36:07Z","snapshot_observed_at":"2026-07-06T16:19:22.122087Z","submitted_at":"2023-09-16T07:36:07Z","title":"A Statistical Turing Test for Generative Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08913","snapshot_observed_at":"2026-08-04T17:26:38.323143Z","title":"E., and Yang, W","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.323143Z"},"links":{"cited_paper":"/paper/2309.08913","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:422666d776e618120a83309855ce4b9d6ea87e1d3daa8289e025cefba8020278","observation_id":"f019bd5b-25b9-4d3a-af87-9a9adb3d788a","resolution":{"observed_at":"2026-08-04T17:26:38.323143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08825","last_updated":"2024-03-08T02:49:12Z","snapshot_observed_at":"2026-08-07T23:37:32.488201Z","submitted_at":"2023-10-13T02:41:55Z","title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08825","snapshot_observed_at":"2026-08-04T17:26:38.328095Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.328095Z"},"links":{"cited_paper":"/paper/2310.08825","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:dc0e1999da42d457528b9da92b0c7972cc10041c7544a2b8030b323edd98ff87","observation_id":"a54931fd-ea48-473d-a5de-b2c9e84bb2d0","resolution":{"observed_at":"2026-08-04T17:26:38.328095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.332606Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.332606Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:aa834a02d8f7343bda796087cd1b5ee7da6d54f88ac1632b86c430ad4b3272e4","observation_id":"4e6bcd10-9aa7-49db-b159-50e9b757ee75","resolution":{"observed_at":"2026-08-04T17:26:38.332606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.337094Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.337094Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:8bdbcb61be7f4744a8b88142708758730333cb6f5f67a66bf11eb25e775490ee","observation_id":"0936f829-08fc-4804-b0d8-bb73ab1b9fba","resolution":{"observed_at":"2026-08-04T17:26:38.337094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.341371Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.341371Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:1407d70b50072d1ffd51b54f541a34019bcd5ac18a86afaaa642e7dffd6ff1ef","observation_id":"7f977a3c-2a57-46c3-823a-b8f29db191db","resolution":{"observed_at":"2026-08-04T17:26:38.341371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00640","last_updated":"2024-11-01T14:57:16Z","snapshot_observed_at":"2026-08-07T10:39:44.274910Z","submitted_at":"2024-11-01T14:57:16Z","title":"Adding Error Bars to Evals: A Statistical Approach to Language Model Evaluations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00640","snapshot_observed_at":"2026-08-04T17:26:38.345872Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.345872Z"},"links":{"cited_paper":"/paper/2411.00640","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:8e8c9451ce98961a24de12485b8fd1381fea5f841bbdba83664a90a8db8e7b5b","observation_id":"a649924f-008f-4c42-9079-f22fba09622d","resolution":{"observed_at":"2026-08-04T17:26:38.345872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06573","last_updated":"2024-09-01T19:38:02Z","snapshot_observed_at":"2026-07-06T18:28:20.634381Z","submitted_at":"2024-06-03T18:15:56Z","title":"MedFuzz: Exploring the Robustness of Large Language Models in Medical Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06573","snapshot_observed_at":"2026-08-04T17:26:38.350247Z","title":"O., Matton, K., Helm, H., Zhang, S., Bajwa, J., Priebe, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.350247Z"},"links":{"cited_paper":"/paper/2406.06573","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:7fcf0e314faf71def4872f60e6790a22f6aa1e67110d8f0f731f332c4cdfa655","observation_id":"d2966f35-8df5-4396-8403-1ab3f0feb77c","resolution":{"observed_at":"2026-08-04T17:26:38.350247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16452","last_updated":"2023-11-28T03:16:12Z","snapshot_observed_at":"2026-08-07T08:26:36.845011Z","submitted_at":"2023-11-28T03:16:12Z","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16452","snapshot_observed_at":"2026-08-04T17:26:38.354877Z","title":"T., Zhang, S., Carignan, D., Edgar, R., Fusi, N., King, N., Larson, J., Li, Y., Liu, W., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.354877Z"},"links":{"cited_paper":"/paper/2311.16452","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:c1f34922ad160f4c84187437c952d4e197e53d5e70ccef6ba60f0101e3bbe1c9","observation_id":"797b5eb6-1460-48b0-b778-10461e3117c3","resolution":{"observed_at":"2026-08-04T17:26:38.354877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.359620Z","title":"M., Terano, H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.359620Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:e2769ec70278ca8c923bfffc83dbac3cbb39d7c96b66365cdefc6270bb540ad2","observation_id":"0bba18bf-9cd8-46cb-8a49-697560c21547","resolution":{"observed_at":"2026-08-04T17:26:38.359620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.363860Z","title":"J., Ryan, P","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.363860Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:3f45cb2e87f9bf38beae459ba11d942b522b24d421203a90288d8033cb9b0b8b","observation_id":"b8dbb191-40e5-4b80-948f-30f759369290","resolution":{"observed_at":"2026-08-04T17:26:38.363860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16789","last_updated":"2024-03-09T22:26:06Z","snapshot_observed_at":"2026-08-08T18:07:29.632928Z","submitted_at":"2023-10-25T17:21:23Z","title":"Detecting Pretraining Data from Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16789","snapshot_observed_at":"2026-08-04T17:26:38.368248Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.368248Z"},"links":{"cited_paper":"/paper/2310.16789","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:a75675de9c8dbeffbecdf0b758a6168943eef883380707f5e697ae1e465ebc5b","observation_id":"02110994-59fe-4827-af19-80ee369c0289","resolution":{"observed_at":"2026-08-04T17:26:38.368248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.372460Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.372460Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:8fc55ca161b2e430c5d73d5f848c85473c3dbdb80112513d43bcbe6cf8d1a1c3","observation_id":"46b0ba09-038a-44ab-81fe-2b23fdf69ec1","resolution":{"observed_at":"2026-08-04T17:26:38.372460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.09136","last_updated":"2023-03-16T08:01:22Z","snapshot_observed_at":"2026-08-08T20:48:07.587002Z","submitted_at":"2023-03-16T08:01:22Z","title":"A Short Survey of Viewing Large Language Models in Legal Aspect","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.09136","snapshot_observed_at":"2026-08-04T17:26:38.376545Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.376545Z"},"links":{"cited_paper":"/paper/2303.09136","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:1ea6a2abf91a798d1cd0a8aed1c02d720b660d88dbf3291da6ab5d8fb7daf16e","observation_id":"e96bdd8e-e364-4eb7-95ea-4207c7361227","resolution":{"observed_at":"2026-08-04T17:26:38.376545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-04T17:26:38.381255Z","title":"M., Hauth, A., Millican, K., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.381255Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:a4c8adab66c0105f77714c594808bfd855cf34c470b4040b9a1500df1d785eeb","observation_id":"5e31537c-3a1f-4ce6-902d-dcf2b6ab02da","resolution":{"observed_at":"2026-08-04T17:26:38.381255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07827","last_updated":"2024-02-12T17:34:13Z","snapshot_observed_at":"2026-08-08T14:33:43.216422Z","submitted_at":"2024-02-12T17:34:13Z","title":"Aya Model: An Instruction Finetuned Open-Access Multilingual Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07827","snapshot_observed_at":"2026-08-04T17:26:38.385609Z","title":"J., Ting, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.385609Z"},"links":{"cited_paper":"/paper/2402.07827","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:bfcd3593bb2baa91ba425ae3c9be06e7ff9c8d415d3d09d74740468bbe01f2b3","observation_id":"3e159ec2-9d47-49dd-9ef2-f4cfb13fcec1","resolution":{"observed_at":"2026-08-04T17:26:38.385609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T17:26:38.390712Z","title":"and Barter, R","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.390712Z"},"links":{"citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:620618426fa43458e88bf12b7ccabcfcf318e876d5c6b68cc4b352a43fd8d895","observation_id":"0026d3b8-97b2-4d65-801b-3de412dd33f2","resolution":{"observed_at":"2026-08-04T17:26:38.390712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07339","last_updated":"2024-08-09T06:16:55Z","snapshot_observed_at":"2026-07-06T17:15:22.551637Z","submitted_at":"2024-01-14T18:12:03Z","title":"CodeAgent: Enhancing Code Generation with Tool-Integrated Agent Systems for Real-World Repo-level Coding Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07339","snapshot_observed_at":"2026-08-04T17:26:38.394751Z","title":"rX k=1 X (j) k −X ′ k r −E rX k=1 X (j) k −X ′ k r ! < t # ≥1−2e − rt2 2 =⇒P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T17:26:38.394751Z"},"links":{"cited_paper":"/paper/2401.07339","citing_paper":"/paper/2509.10963"},"observation_digest":"sha256:6fccf18bc0d4774b1b833b77ac5ee696b1abc48cfc186c08878d1da240087d17","observation_id":"227be055-c831-4a01-808e-66ccb3da17de","resolution":{"observed_at":"2026-08-04T17:26:38.394751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.10963","last_updated":"2025-09-13T19:44:42Z","latest_version":1,"primary_category":"math.ST","snapshot_observed_at":"2026-08-08T11:52:22.723584Z","submitted_at":"2025-09-13T19:44:42Z","title":"Testing for LLM response differences: the case of a composite null consisting of semantically irrelevant query perturbations"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":27,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2509.10963."}