{"as_of":"2026-08-09T05:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:148fae97a11b155e21d64baa899c6b380b38e2418ec39a8b0968d935c6dae5d6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":11,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":11,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T14:51:15.504623Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T14:18:22.614907Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2309.17167","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-07-03T14:18:22.614907Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"1b207299-354d-4eb1-ba5a-57ca81ccfc18","year":2023},"citing_paper":{"arxiv_id":"2310.19852","last_updated":"2025-04-04T11:14:49Z","snapshot_observed_at":"2026-08-05T20:02:35.087707Z","submitted_at":"2023-10-30T15:52:15Z","title":"AI Alignment: A Comprehensive Survey","version":6},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T14:28:48.987140Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2310.19852"},"observation_digest":"sha256:5f82a10bb321073e7c9db1706790eefad2cd8ed4d2b93cdf95a419642fbb0431","observation_id":"80abcfa1-0db8-4723-ad31-ad77ce6bf351","resolution":{"observed_at":"2026-05-17T14:28:49.009396Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2309.17167","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-07-03T14:18:22.614907Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"1b207299-354d-4eb1-ba5a-57ca81ccfc18","year":2023},"citing_paper":{"arxiv_id":"2404.11584","last_updated":"2024-04-17T17:32:41Z","snapshot_observed_at":"2026-08-04T23:37:27.678120Z","submitted_at":"2024-04-17T17:32:41Z","title":"The Landscape of Emerging AI Agent Architectures for Reasoning, Planning, and Tool Calling: A Survey","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T23:16:41.679855Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2404.11584"},"observation_digest":"sha256:d0ea8b391a3dff940cd127bd8445d243ad621cc484de0b822775020e05f5df15","observation_id":"d096e8aa-526c-456e-9285-3ca85a935ca0","resolution":{"observed_at":"2026-05-16T23:16:41.701975Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-08-08T14:51:15.504623Z","title":"Z., Yang, D., and Xie, X","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06655","last_updated":"2025-05-12T14:34:05Z","snapshot_observed_at":"2026-08-09T03:53:29.137552Z","submitted_at":"2025-02-10T16:45:18Z","title":"Unbiased Evaluation of Large Language Models from a Causal Perspective","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-08T14:51:15.504623Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2502.06655"},"observation_digest":"sha256:e01e4a567fa924df1de4b6395255c97173d2f96c464472dadd9230f4b1ce6376","observation_id":"4d3f199f-97ef-4ba1-aebd-2a112b542616","resolution":{"observed_at":"2026-08-08T14:51:15.504623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-08-08T12:17:55.420540Z","title":"Hello World","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07577","last_updated":"2025-06-09T17:49:46Z","snapshot_observed_at":"2026-08-08T12:13:06.365821Z","submitted_at":"2025-02-11T14:23:13Z","title":"Automated Capability Discovery via Foundation Model Self-Exploration","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-08T12:17:55.420540Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2502.07577"},"observation_digest":"sha256:3b06b258baa7cb8366eb9e579b26c20ae5da2941538ca2ac4b1ad499f1fc80a9","observation_id":"ffb450d2-df90-4b18-9933-f04b013d0295","resolution":{"observed_at":"2026-08-08T12:17:55.420540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-08-07T05:43:43.859281Z","title":"Dyval: Graph-informed dynamic evaluation of large language models.arXiv preprint arXiv:2309.17167, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07202","last_updated":"2025-06-08T15:52:38Z","snapshot_observed_at":"2026-08-08T08:51:05.416786Z","submitted_at":"2025-06-08T15:52:38Z","title":"Reasoning Multimodal Large Language Model: Data Contamination and Dynamic Evaluation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T05:43:43.859281Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2506.07202"},"observation_digest":"sha256:2326ff508b18b282d129f4e4ff283a7fbfaac982eedfdfe3f3fd64ec5f6edc1c","observation_id":"bcc4f788-8ab4-443c-b4bf-217c5acd20de","resolution":{"observed_at":"2026-08-07T05:43:43.859281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-08-05T15:43:12.862651Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.19570","last_updated":"2025-08-27T05:04:07Z","snapshot_observed_at":"2026-08-09T02:49:31.518578Z","submitted_at":"2025-08-27T05:04:07Z","title":"Generative Models for Synthetic Data: Transforming Data Mining in the GenAI Era","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T15:43:12.862651Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2508.19570"},"observation_digest":"sha256:9445d97a47c6102bdc1b5ea8f9b6add4e4fdd1e1595fb59c73560a799dfff393","observation_id":"733e74a4-d303-4d0a-9215-9bd1fdd42f1d","resolution":{"observed_at":"2026-08-05T15:43:12.862651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2309.17167","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-07-03T14:18:22.614907Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"1b207299-354d-4eb1-ba5a-57ca81ccfc18","year":2023},"citing_paper":{"arxiv_id":"2604.17842","last_updated":"2026-04-20T05:51:50Z","snapshot_observed_at":"2026-08-04T02:34:16.438625Z","submitted_at":"2026-04-20T05:51:50Z","title":"QuickScope: Certifying Hard Questions in Dynamic LLM Benchmarks","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-10T04:27:11.735657Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2604.17842"},"observation_digest":"sha256:307e0926284677a373c4ddd7d3fb81370733ba50a60d70972af2228a1115be55","observation_id":"fd6e6606-853a-45cf-9a72-7e3e07d768a2","resolution":{"observed_at":"2026-05-11T11:56:29.575749Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2309.17167","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-07-03T14:18:22.614907Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"1b207299-354d-4eb1-ba5a-57ca81ccfc18","year":2023},"citing_paper":{"arxiv_id":"2605.08904","last_updated":"2026-05-09T11:51:34Z","snapshot_observed_at":"2026-07-06T23:21:02.177557Z","submitted_at":"2026-05-09T11:51:34Z","title":"OPT-BENCH: Evaluating the Iterative Self-Optimization of LLM Agents in Large-Scale Search Spaces","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-12T02:57:15.521594Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2605.08904"},"observation_digest":"sha256:89a2d39207cc153035af10ef2fd03382e4010a9961cd3764f8b2d10e4c0db4e6","observation_id":"d8b1cad9-4724-4f5e-a61c-630045c9aaf5","resolution":{"observed_at":"2026-05-12T03:01:18.745491Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2309.17167","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-07-03T14:18:22.614907Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"1b207299-354d-4eb1-ba5a-57ca81ccfc18","year":2023},"citing_paper":{"arxiv_id":"2605.15865","last_updated":"2026-05-15T11:33:48Z","snapshot_observed_at":"2026-08-01T08:51:45.063110Z","submitted_at":"2026-05-15T11:33:48Z","title":"From Text to DSL: Evaluating Grammar-Based Model Generation Using Open LLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-20T16:40:08.788179Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2605.15865"},"observation_digest":"sha256:923fd295f2dd75fc349905a274cdc5d055435df1509a8d79acd7deae359ad54e","observation_id":"72ba854a-499a-49dd-8308-cf390beb281c","resolution":{"observed_at":"2026-05-20T16:43:34.739101Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2309.17167","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-07-03T14:18:22.614907Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"1b207299-354d-4eb1-ba5a-57ca81ccfc18","year":2023},"citing_paper":{"arxiv_id":"2605.22564","last_updated":"2026-05-21T14:45:02Z","snapshot_observed_at":"2026-07-06T23:32:54.127514Z","submitted_at":"2026-05-21T14:45:02Z","title":"SynAE: A Framework for Measuring the Quality of Synthetic Data for Tool-Calling Agent Evaluations","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-22T06:25:06.024542Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2605.22564"},"observation_digest":"sha256:7ff85b4e90afa654fc97bba6d1a91000682663122a0f93c4a54e149057681c33","observation_id":"d15d5841-55e1-4ab4-aac9-a2a998bfcdcb","resolution":{"observed_at":"2026-05-22T06:26:09.983958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2309.17167","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.17167","snapshot_observed_at":"2026-07-03T14:18:22.614907Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"1b207299-354d-4eb1-ba5a-57ca81ccfc18","year":2023},"citing_paper":{"arxiv_id":"2607.02141","last_updated":"2026-07-02T13:18:57Z","snapshot_observed_at":"2026-07-07T00:07:37.820438Z","submitted_at":"2026-07-02T13:18:57Z","title":"A$^{2}$utoLPBench: An Auto-Generated, Agent-Friendly LP Benchmark via Inverse-KKT Construction","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-03T14:08:50.462880Z"},"links":{"cited_paper":"/paper/2309.17167","citing_paper":"/paper/2607.02141"},"observation_digest":"sha256:154fd5c03c24fa5dc75ffd6bf8f08dd799bc7bb6991ad50c5087a63e0c9264e7","observation_id":"1e0e4761-68dd-421c-af32-243da1e6ddd5","resolution":{"observed_at":"2026-07-03T14:18:22.616286Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2309.17167/citation-record","integrity":"/paper/2309.17167/integrity","json":"/paper/2309.17167/citation-record.json","paper":"/paper/2309.17167"},"outbound":[],"paper":{"arxiv_id":"2309.17167","last_updated":"2024-03-14T09:52:16Z","latest_version":3,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-08T18:43:29.715488Z","submitted_at":"2023-09-29T12:04:14Z","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 11 inbound Pith citation observations for arXiv:2309.17167."}