{"as_of":"2026-08-15T21:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f93d24e6766b0d1471be47ac51ccb8d900f269e7568e205711ecde48ee325d4e","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T06:42:09.641449Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T00:39:34.778096Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-03T12:16:14.848390Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.12252","snapshot_observed_at":"2026-08-03T10:48:03.586532Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.29252","last_updated":"2026-07-31T10:21:56Z","snapshot_observed_at":"2026-08-07T03:23:45.914339Z","submitted_at":"2026-07-31T10:21:56Z","title":"CalibratedRubric: Task-Adaptive Rubric Banks for Open-Ended LLM Evaluation","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-03T10:48:03.586532Z"},"links":{"cited_paper":"/paper/2607.12252","citing_paper":"/paper/2607.29252"},"observation_digest":"sha256:2259f45407f56a1b1b47551e3d1a0373b48cf83c0dae9404b4f317791f2363b0","observation_id":"40b7bff8-2aa2-4a06-b5be-46f5a3f9164a","resolution":{"observed_at":"2026-08-03T10:48:03.586532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"cited_work":{"arxiv_id":"2607.12252","doi":"10.48550/arxiv.2607.12252","metadata_source":"pith","pith_arxiv_id":"2607.12252","snapshot_observed_at":"2026-08-03T12:16:14.848390Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","venue":"cs.CL","work_id":"506f652c-264a-4a76-98b3-c803213486f4","year":2026},"citing_paper":{"arxiv_id":"2607.29252","last_updated":"2026-07-31T10:21:56Z","snapshot_observed_at":"2026-08-07T03:23:45.914339Z","submitted_at":"2026-07-31T10:21:56Z","title":"CalibratedRubric: Task-Adaptive Rubric Banks for Open-Ended LLM Evaluation","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-03T10:48:03.665912Z"},"links":{"cited_paper":"/paper/2607.12252","citing_paper":"/paper/2607.29252"},"observation_digest":"sha256:10a560ee6659a00a60671c501ade9f9473e479c43899fb71ae1e9786c93e9e19","observation_id":"0fb45fc1-202e-410e-ada6-66e1282fb769","resolution":{"observed_at":"2026-08-03T10:48:30.603281Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.12252","snapshot_observed_at":"2026-08-08T00:39:34.778096Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04077","last_updated":"2026-08-04T18:00:00Z","snapshot_observed_at":"2026-08-15T06:51:08.406461Z","submitted_at":"2026-08-04T18:00:00Z","title":"FinProBench: Evaluating Financial AI Agents with Role-Grounded Rubrics Derived from Professional Deliverables","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-08T00:39:34.778096Z"},"links":{"cited_paper":"/paper/2607.12252","citing_paper":"/paper/2608.04077"},"observation_digest":"sha256:5b781336fc9de86f744d96f4c0af7b242c960515054c4ab95f81ce8421e5fa4c","observation_id":"9b3690fa-dd86-4197-a962-8c27a9431167","resolution":{"observed_at":"2026-08-08T00:39:34.778096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2607.12252/citation-record","integrity":"/paper/2607.12252/integrity","json":"/paper/2607.12252/citation-record.json","paper":"/paper/2607.12252"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:07.919219Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:07.919219Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:352aa18c9d6562658f506b38197658c47eff8d7e049e74b4e65de58404949308","observation_id":"f48fab4e-7d8b-441f-b44a-8bad75e0fe1c","resolution":{"observed_at":"2026-08-02T06:42:07.919219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.17776","last_updated":"2026-07-13T06:52:44Z","snapshot_observed_at":"2026-08-14T06:01:54.475545Z","submitted_at":"2025-12-19T16:46:20Z","title":"DEER: A Benchmark for Evaluating Deep Research Agents on Expert Report Generation","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.17776","snapshot_observed_at":"2026-08-02T06:42:08.127436Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.127436Z"},"links":{"cited_paper":"/paper/2512.17776","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:cb7429e82d394e4841ea67448b0d687a2a8a292763d9c9b67549580cc094beb6","observation_id":"a65a7ba9-db4b-4d98-aec5-eb1734d1a06d","resolution":{"observed_at":"2026-08-02T06:42:08.127436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:08.241173Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.241173Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:9851ce68836fd366b8f10559bdd78481da1a59e45c62295e0d53cfdbc9591971","observation_id":"bceb10e3-d39f-418e-8cbf-87f0774d914a","resolution":{"observed_at":"2026-08-02T06:42:08.241173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19457","last_updated":"2025-05-26T03:23:02Z","snapshot_observed_at":"2026-08-08T14:50:13.472663Z","submitted_at":"2025-05-26T03:23:02Z","title":"BizFinBench: A Business-Driven Real-World Financial Benchmark for Evaluating LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19457","snapshot_observed_at":"2026-08-02T06:42:08.398960Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.398960Z"},"links":{"cited_paper":"/paper/2505.19457","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:a5c247198ec3b2534efec9859a9928156c368ee080014dc05b1859750fc9d4ea","observation_id":"76b952e0-6513-4e6e-bcfc-9704500c9b59","resolution":{"observed_at":"2026-08-02T06:42:08.398960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:08.461514Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.461514Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:3ebe1066246ef5a1442493f570515acea88c79e8f0e4b6c9c73c9b9ca450875e","observation_id":"cd6a4237-faaa-4657-8234-995e5298f0b5","resolution":{"observed_at":"2026-08-02T06:42:08.461514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.16191","last_updated":"2024-09-24T15:38:11Z","snapshot_observed_at":"2026-08-12T22:38:31.274535Z","submitted_at":"2024-09-24T15:38:11Z","title":"HelloBench: Evaluating Long Text Generation Capabilities of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.16191","snapshot_observed_at":"2026-08-02T06:42:08.570674Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.570674Z"},"links":{"cited_paper":"/paper/2409.16191","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:0704b63980f1cd633efd07282ed58cdcd438fe8e7bb11964371ed5f77ad49683","observation_id":"e50a4124-0d90-4ddd-9f75-66bc7224786b","resolution":{"observed_at":"2026-08-02T06:42:08.570674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:08.624206Z","title":"InFindings of the Association for Computational Linguistics: EMNLP 2025, pages 5977–6043","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.624206Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:80a5bd781cbe5ab3091eaff0243cd194bb27e424791b3c9c63ab9a1e3a9d05a7","observation_id":"88db5f24-768f-4ed3-a3c2-e1bbdd7e6cc6","resolution":{"observed_at":"2026-08-02T06:42:08.624206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:08.693996Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.693996Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:d344fbffe5ef82bf1f9752c66afbf040a941ed0bca50b0cbe6e12f0a4c3bacbc","observation_id":"581ae318-8efb-42f9-9210-cc8d88f2ac3e","resolution":{"observed_at":"2026-08-02T06:42:08.693996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-14T08:25:18.333075Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-02T06:42:08.783508Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.783508Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:74ddcd443eea08861f60517c617e6e285dbc51febb4024a432ee4a2392b44a4b","observation_id":"15f66bc1-57d8-4608-886c-8243edea3907","resolution":{"observed_at":"2026-08-02T06:42:08.783508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12516","last_updated":"2025-04-16T22:27:45Z","snapshot_observed_at":"2026-08-13T03:19:29.377723Z","submitted_at":"2025-04-16T22:27:45Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12516","snapshot_observed_at":"2026-08-02T06:42:08.873217Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.873217Z"},"links":{"cited_paper":"/paper/2504.12516","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:56d116887cf89a11a43509f641ece74efc21ac7f11fd5d7edcd7823b5f9a34bc","observation_id":"ae20f553-2ee5-40b2-9c31-b31403181a56","resolution":{"observed_at":"2026-08-02T06:42:08.873217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19103","last_updated":"2025-03-07T11:05:01Z","snapshot_observed_at":"2026-08-07T17:46:34.831368Z","submitted_at":"2025-02-26T12:46:36Z","title":"LongEval: A Comprehensive Analysis of Long-Text Generation Through a Plan-based Paradigm","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19103","snapshot_observed_at":"2026-08-02T06:42:08.925768Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.925768Z"},"links":{"cited_paper":"/paper/2502.19103","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:68cee6bb5c56ed4c97e85433e15775abf1e1ee3ba0188e84ecebcee502ecd1d4","observation_id":"4dd093ed-d0ba-4f34-b129-8b254c0c7724","resolution":{"observed_at":"2026-08-02T06:42:08.925768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12594","last_updated":"2025-06-14T18:19:05Z","snapshot_observed_at":"2026-08-13T01:21:59.385623Z","submitted_at":"2025-06-14T18:19:05Z","title":"A Comprehensive Survey of Deep Research: Systems, Methodologies, and Applications","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.12594","snapshot_observed_at":"2026-08-02T06:42:09.001256Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:09.001256Z"},"links":{"cited_paper":"/paper/2506.12594","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:913191746d1052d69bb09033a025892a844cd3d0afe46bdccf4549000e9d2b6b","observation_id":"1e5994d6-ff87-4c8f-81b5-506f258a1959","resolution":{"observed_at":"2026-08-02T06:42:09.001256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:09.172939Z","title":"arXiv preprint arXiv:2601.05111","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:09.172939Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:51a176fc759513794f63e1703affca680ede20764f374aa5d6fb43acdeb7f7c2","observation_id":"21db6964-ef75-4378-a602-96d607e0b95a","resolution":{"observed_at":"2026-08-02T06:42:09.172939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.17186","last_updated":"2025-07-31T08:14:21Z","snapshot_observed_at":"2026-08-07T21:24:43.546217Z","submitted_at":"2025-07-23T04:19:16Z","title":"FinGAIA: A Chinese Benchmark for AI Agents in Real-World Financial Domain","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.17186","snapshot_observed_at":"2026-08-02T06:42:09.297213Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:09.297213Z"},"links":{"cited_paper":"/paper/2507.17186","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:9139450d82e9c31532b4caa0e01acfef37426bb6583100e38e1219fe0babad79","observation_id":"7d1d83a4-f5b1-497d-b91d-14440922655b","resolution":{"observed_at":"2026-08-02T06:42:09.297213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:09.370074Z","title":"InProceedings of the 2025 Conference on Empirical Methods in Natural Lan- guage Processing, pages 414–431","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:09.370074Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:88c3f1608d068ebb1cf1f6258ce802e71cacfd47e38d9b88379c6322248c489b","observation_id":"5d6a9d45-4a8e-4f57-8576-c6ef12e3ec18","resolution":{"observed_at":"2026-08-02T06:42:09.370074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:09.489574Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:09.489574Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:c965089acf702f557eee9afa9514969d4386da5a5af278a196419acaf6d2034b","observation_id":"646a2981-2a8e-4c1d-b8ba-c341df0c83e2","resolution":{"observed_at":"2026-08-02T06:42:09.489574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10934","last_updated":"2024-10-16T17:54:12Z","snapshot_observed_at":"2026-08-12T22:23:27.859816Z","submitted_at":"2024-10-14T17:57:02Z","title":"Agent-as-a-Judge: Evaluate Agents with Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10934","snapshot_observed_at":"2026-08-02T06:42:09.641449Z","title":"Yes” •If the report does not satisfy the criterion, answer “No","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:09.641449Z"},"links":{"cited_paper":"/paper/2410.10934","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:bc3b7ec1395ac3d42ea337928449ae7bf8f83e7d7a934c5263b4c870a517b161","observation_id":"86b67f9f-38ad-4783-8a98-836a59ac5cd9","resolution":{"observed_at":"2026-08-02T06:42:09.641449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:42:09.070946Z","title":"InProceedings of the 2018 Conference on Empirical Methods in Natural Language Process- ing, pages 2369–2380","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:09.070946Z"},"links":{"citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:c53f05fc9fe665e21015ffae23510582d9b704219cb4759b5585cb2c21ab3c91","observation_id":"ad4b7ce1-d9f0-4115-ac33-cc3cbb4b4932","resolution":{"observed_at":"2026-08-02T06:42:09.070946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.06474","last_updated":"2026-04-07T21:19:26Z","snapshot_observed_at":"2026-08-14T00:15:16.774906Z","submitted_at":"2026-04-07T21:19:26Z","title":"DataSTORM: Deep Research on Large-Scale Databases using Exploratory Data Analysis and Data Storytelling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.06474","snapshot_observed_at":"2026-08-02T06:42:08.297818Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.297818Z"},"links":{"cited_paper":"/paper/2604.06474","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:0255dfa080996c110dd5fce4ac392923b70f86139460fde9f7d3eb14d2f22e87","observation_id":"bdc2b554-74e2-4efd-ab94-b560e9542a8e","resolution":{"observed_at":"2026-08-02T06:42:08.297818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-13T14:09:17.964744Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-02T06:42:08.350415Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.350415Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:f166becd290a896f512ec5d19af6fc0bb9470f227feff0493605a2449b996e96","observation_id":"ebf59298-5f97-40a3-b4a8-ad9f5f064476","resolution":{"observed_at":"2026-08-02T06:42:08.350415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07055","last_updated":"2024-08-13T17:46:12Z","snapshot_observed_at":"2026-08-12T23:04:08.269970Z","submitted_at":"2024-08-13T17:46:12Z","title":"LongWriter: Unleashing 10,000+ Word Generation from Long Context LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07055","snapshot_observed_at":"2026-08-02T06:42:07.646102Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:07.646102Z"},"links":{"cited_paper":"/paper/2408.07055","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:c7cc07231bd8a45dbb7b2e626576f64958942881bca4fec68da03baa61e6969b","observation_id":"5695399d-7498-4a8f-88e3-b65b6faf7ca4","resolution":{"observed_at":"2026-08-02T06:42:07.646102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11763","last_updated":"2025-06-13T13:17:32Z","snapshot_observed_at":"2026-08-14T22:03:22.542485Z","submitted_at":"2025-06-13T13:17:32Z","title":"DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11763","snapshot_observed_at":"2026-08-02T06:42:07.800790Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:07.800790Z"},"links":{"cited_paper":"/paper/2506.11763","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:05cdb4799a973d70c0547e08651982c1b53103b132be8f012fc1a0875a7568ae","observation_id":"c68b6e74-dfea-4c26-8820-e258fedee60d","resolution":{"observed_at":"2026-08-02T06:42:07.800790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.06401","last_updated":"2026-07-13T06:29:58Z","snapshot_observed_at":"2026-08-15T02:52:34.524674Z","submitted_at":"2026-01-10T02:51:53Z","title":"BizFinBench.v2: Towards Reliable LLMs in Finance via Real-User Data and Offline/Online Bilingual Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.06401","snapshot_observed_at":"2026-08-02T06:42:07.998224Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:07.998224Z"},"links":{"cited_paper":"/paper/2601.06401","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:946f50324b84743191627b499a9d59c834d42e32368e98fcfe1f3ee53a6fcd7d","observation_id":"bcdbd8bc-9523-4a0c-a55d-61fb0dff0982","resolution":{"observed_at":"2026-08-02T06:42:07.998224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-15T06:29:59.077676Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 3 inbound Pith citation observations for arXiv:2607.12252."}