{"as_of":"2026-08-09T05:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bbaf2908f3ac004d378dd2421b9e2b36805a1ff0291d78184739b69ea7ad9e80","coverage":[{"denominator":13,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:17:45.470828Z","state":"measured"},{"denominator":13,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":13,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.00172/citation-record","integrity":"/paper/2506.00172/integrity","json":"/paper/2506.00172/citation-record.json","paper":"/paper/2506.00172"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.04872","last_updated":"2025-12-23T02:23:47Z","snapshot_observed_at":"2026-08-04T15:54:46.196160Z","submitted_at":"2024-11-07T17:07:35Z","title":"FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04872","snapshot_observed_at":"2026-08-07T12:17:44.198051Z","title":"Frontiermath: A benchmark for evaluating advanced mathematical reasoning in ai","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.198051Z"},"links":{"cited_paper":"/paper/2411.04872","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:a9de7b577553a1506fd1b9b5d127adb5ca2da929b6824d4c0cf7c75a023b9478","observation_id":"e3c96934-e189-4598-8b84-cb32a3653808","resolution":{"observed_at":"2026-08-07T12:17:44.198051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01926","last_updated":"2024-10-02T18:20:51Z","snapshot_observed_at":"2026-08-09T01:21:18.358004Z","submitted_at":"2024-10-02T18:20:51Z","title":"MARPLE: A Benchmark for Long-Horizon Inference","version":1},"cited_work":{"arxiv_id":"2410.01926","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.01926","snapshot_observed_at":"2026-08-07T12:17:45.802359Z","title":"MARPLE: A Benchmark for Long-Horizon Inference","venue":"cs.LG","work_id":"7fd55a84-fc1f-477b-817f-85b0b4c3d139","year":2024},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.420152Z"},"links":{"cited_paper":"/paper/2410.01926","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:1d794d1b153371e7aeceded7f6492f3d2c908ed45bfde993363336c44f9d2c61","observation_id":"edf39d07-da64-4c83-a760-260036ed7813","resolution":{"observed_at":"2026-08-07T12:17:45.870546Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:17:44.778572Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.778572Z"},"links":{"citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:12d9a6e16c9fafa1b2deb684be70b7612f53ef016f1223ec5009abbce43fdc76","observation_id":"0db5c82e-6129-49d0-b34c-14192f4f7637","resolution":{"observed_at":"2026-08-07T12:17:44.778572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-06T20:36:41.418114Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-07T12:17:44.888628Z","title":"Nat McAleese, Rai Michael Pokorny, Juan Felipe Ceron Uribe, Evgenia Nitishinskaya, Maja Trebacz, and Jan Leike","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.888628Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:f227fb7734176849c7ab42ee897cef1f082ec205da83b85c24c17dad1c2ff073","observation_id":"9d2ff84f-c45c-443a-a488-d72430be49dd","resolution":{"observed_at":"2026-08-07T12:17:44.888628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00215","last_updated":"2024-06-28T19:53:17Z","snapshot_observed_at":"2026-07-06T18:38:46.314431Z","submitted_at":"2024-06-28T19:53:17Z","title":"LLM Critics Help Catch LLM Bugs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00215","snapshot_observed_at":"2026-08-07T12:17:44.997683Z","title":"Daniel McFadden","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.997683Z"},"links":{"cited_paper":"/paper/2407.00215","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:16270125841e5e062450446c788f502ff246b70b07657fb82a40a040ed4c01d0","observation_id":"d664ad6a-ee01-489f-9aac-0f635ddb0cc3","resolution":{"observed_at":"2026-08-07T12:17:44.997683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12115","last_updated":"2025-05-29T23:07:34Z","snapshot_observed_at":"2026-08-07T18:11:26.038748Z","submitted_at":"2025-02-17T18:41:16Z","title":"SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12115","snapshot_observed_at":"2026-08-07T12:17:45.190655Z","title":"Xinshuai Song, Weixing Chen, Yang Liu, Weikai Chen, Guanbin Li, and Liang Lin","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:45.190655Z"},"links":{"cited_paper":"/paper/2502.12115","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:aacfff51e14e47d0e28d861e20eaa9fd56176db012d40a164be8705c172f90a2","observation_id":"2ff36b65-ec3a-4205-b6ec-6d73fc436d05","resolution":{"observed_at":"2026-08-07T12:17:45.190655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.09082","last_updated":"2025-03-19T13:31:16Z","snapshot_observed_at":"2026-08-09T01:20:56.801787Z","submitted_at":"2024-12-12T09:08:13Z","title":"Towards Long-Horizon Vision-Language Navigation: Platform, Benchmark and Method","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.09082","snapshot_observed_at":"2026-08-07T12:17:45.278095Z","title":"15 Preprint","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:45.278095Z"},"links":{"cited_paper":"/paper/2412.09082","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:b4bf368ecf949f2bdf2c010871e1d6093921e6aa7ecd936263123188a0ea33ad","observation_id":"13c666ef-947e-41ce-be2f-5852f06b73e0","resolution":{"observed_at":"2026-08-07T12:17:45.278095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.15793","last_updated":"2024-11-11T20:01:15Z","snapshot_observed_at":"2026-07-06T18:19:29.996982Z","submitted_at":"2024-05-06T17:41:33Z","title":"SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.15793","snapshot_observed_at":"2026-08-07T12:17:45.374967Z","title":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:45.374967Z"},"links":{"cited_paper":"/paper/2405.15793","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:108ce02ba05deb7d8048654c6de11019c09b8f9d5d1bb25cabf0f7ba762d61e5","observation_id":"ab9b37ec-79fd-46c5-afcc-461cd431d942","resolution":{"observed_at":"2026-08-07T12:17:45.374967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12115","last_updated":"2025-05-29T23:07:34Z","snapshot_observed_at":"2026-08-07T18:11:26.038748Z","submitted_at":"2025-02-17T18:41:16Z","title":"SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12115","snapshot_observed_at":"2026-08-07T12:17:45.122364Z","title":"SWE- Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engi- neering? arXiv preprint arXiv:2502.12115,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:45.122364Z"},"links":{"cited_paper":"/paper/2502.12115","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:5831fd35a4208b52c3158e25a561cc05dbd3fb02e04070e76e2d305a245ca9e5","observation_id":"34b13d74-94aa-43d4-ae23-797e0a045e74","resolution":{"observed_at":"2026-08-07T12:17:45.122364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.14858","last_updated":"2022-07-01T02:15:12Z","snapshot_observed_at":"2026-08-05T15:41:22.691461Z","submitted_at":"2022-06-29T18:54:49Z","title":"Solving Quantitative Reasoning Problems with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.14858","snapshot_observed_at":"2026-08-07T12:17:44.675554Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.675554Z"},"links":{"cited_paper":"/paper/2206.14858","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:b12058937e8168383d4fef7fb3453161268eced1f6528b3ef6d3dba400ce94a1","observation_id":"7ff258c3-e82b-42e2-a58b-d3e9d85fff9c","resolution":{"observed_at":"2026-08-07T12:17:44.675554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-07T12:17:45.470828Z","title":"A Agent details The LM agent is provided with a set of tools to interact with the codebase","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:45.470828Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:800de0fa33993ba338fd1ade0016c44ee33a6a11057f8eae3c30890a5ffa879d","observation_id":"0eafdb86-590c-4383-b8d7-23735c733308","resolution":{"observed_at":"2026-08-07T12:17:45.470828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-07T12:17:44.287496Z","title":"Emily Jin, Zhuoyi Huang, Jan-Philipp Fr¨anken, Weiyu Liu, Hannah Cha, Erik Brockbank, Sarah Wu, Ruohan Zhang, Jiajun Wu, and Tobias Gerstenberg","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.287496Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:93dadaac4479b110579e4d4a5185def89a3a7a739c9d46c0744ca6fb439d3852","observation_id":"444f1190-f0e7-4f3b-b90d-ce3c4286dd53","resolution":{"observed_at":"2026-08-07T12:17:44.287496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14499","last_updated":"2026-07-10T21:48:09Z","snapshot_observed_at":"2026-08-07T16:54:09.525403Z","submitted_at":"2025-03-18T17:59:31Z","title":"Measuring AI Ability to Complete Long Software Tasks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14499","snapshot_observed_at":"2026-08-07T12:17:44.566544Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T12:17:44.566544Z"},"links":{"cited_paper":"/paper/2503.14499","citing_paper":"/paper/2506.00172"},"observation_digest":"sha256:37147fcb2c6db8cdbe50e3dde5f7c04ab05283b9e442818c287d4cb054fdbc88","observation_id":"d8fe9f11-f709-4ea8-b24b-29bb36acf662","resolution":{"observed_at":"2026-08-07T12:17:44.566544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.00172","last_updated":"2025-05-30T19:23:51Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T01:21:19.366440Z","submitted_at":"2025-05-30T19:23:51Z","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents"},"reference_resolution":{"displayed":13,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":12,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":13},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 13 of 13 outbound references and 0 inbound Pith citation observations for arXiv:2506.00172."}