{"as_of":"2026-08-23T14:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a86268f870209793323d93a96cf58d812c18d5e690e5b350b1be04b771eadb3b","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T05:17:18.477910Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":13,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T23:22:10.051070Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2605.07335","last_updated":"2026-05-08T06:40:24Z","snapshot_observed_at":"2026-08-17T13:49:03.131484Z","submitted_at":"2026-05-08T06:40:24Z","title":"CellScientist: Dual-Space Hierarchical Orchestration for Closed-Loop Refinement of Virtual Cell Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T01:07:26.231896Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2605.07335"},"observation_digest":"sha256:9448b0cb96960cc0bc90283d13eabbb57b94c2abc2882e2f8e803be71c7392d8","observation_id":"70392bc8-573d-456f-aaaf-417787b1c0ca","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-08-14T16:53:14.701034Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":1},"reference_index":105,"source":"pdf_text","source_observed_at":"2026-05-12T01:13:35.990078Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:e14e09ec93664d5d8ea998f48a443f46a270435ef8acb1ae3553cba6646e318e","observation_id":"d4e56307-b23b-4f2c-92cb-1fe798064293","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-08-14T16:53:14.701034Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":2},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-06-30T23:12:57.154537Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:a6380e92c3f6fe0e1b21b8ea5b242203f05aa82d95b8df5ba1e1ec2a9d61dad1","observation_id":"05dda5d5-c53b-4cf4-933c-187747fa5d41","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2605.23204","last_updated":"2026-05-22T03:40:30Z","snapshot_observed_at":"2026-07-06T23:33:29.550551Z","submitted_at":"2026-05-22T03:40:30Z","title":"AutoResearch AI: Towards AI-Powered Research Automation for Scientific Discovery","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-25T04:46:43.679185Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2605.23204"},"observation_digest":"sha256:9cbd5d6df9941d607e86ad7d5ef7a2234ae17468a9a5cb3c5320137ff54753d8","observation_id":"4d64b97b-8781-4b8d-acf9-52addbcdd093","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2605.26340","last_updated":"2026-05-25T21:30:27Z","snapshot_observed_at":"2026-08-14T19:28:55.309579Z","submitted_at":"2026-05-25T21:30:27Z","title":"ScientistOne: Towards Human-Level Autonomous Research via Chain-of-Evidence","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-29T21:19:03.281629Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2605.26340"},"observation_digest":"sha256:583d8bc946621e56db0dc564c1bca6b83f335c4bf2cccd0e5605dcd738f7dfbc","observation_id":"5dae8655-6ba0-4ef9-a23c-faaff92950e3","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2606.18237","last_updated":"2026-06-16T17:58:05Z","snapshot_observed_at":"2026-08-16T16:35:17.192926Z","submitted_at":"2026-06-16T17:58:05Z","title":"ReproRepo: Scaling Reproducibility Audits with GitHub Repository Issues","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T01:01:24.456670Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2606.18237"},"observation_digest":"sha256:ad678de0fc11969b89027f433251f0984f01c6d4ed143646897a1db4650b63c6","observation_id":"4cf207b5-5848-4295-b211-216a32274e48","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2606.22731","last_updated":"2026-06-22T00:18:27Z","snapshot_observed_at":"2026-08-12T13:13:56.324739Z","submitted_at":"2026-06-22T00:18:27Z","title":"Closed-loop Auto Research for Molecular Property Prediction: Discovering and Certifying Generalizable Improvements","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-26T09:13:18.849983Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2606.22731"},"observation_digest":"sha256:caaad1e4804162142adf78610a50e7acc4d2c90b37499b7b88fb771b68502aa4","observation_id":"9c519e0a-6c3e-459c-aa60-a223f3be3432","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":"2602.02905","doi":"10.48550/arxiv.2602.02905","metadata_source":"arxiv_reference","pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights","venue":"Open MIND","work_id":"44a5b6a2-b99a-4e53-b188-4943cc5f4584","year":2026},"citing_paper":{"arxiv_id":"2606.24530","last_updated":"2026-07-06T16:56:53Z","snapshot_observed_at":"2026-08-15T20:31:56.652908Z","submitted_at":"2026-06-23T12:58:23Z","title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-06-26T00:13:14.940915Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2606.24530"},"observation_digest":"sha256:0adcaeeeaf4fac1b84d88ac989c235327d339ec33a90897b5ff7b53b3a0b0f22","observation_id":"064b3e11-07ee-4cd4-8249-d90cf0837319","resolution":{"observed_at":"2026-07-14T01:19:58.512370Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T00:22:12.403194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-07-14T11:05:51.130839Z","title":"FIRE-bench: Evaluating agents on the rediscovery of scientific insights.arXiv preprint arXiv:2602.02905, 2026b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10522","last_updated":"2026-07-12T00:54:06Z","snapshot_observed_at":"2026-08-17T08:32:43.649983Z","submitted_at":"2026-07-12T00:54:06Z","title":"Towards Autonomous and Auditable Medical Imaging Model Development","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-14T11:05:51.130839Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2607.10522"},"observation_digest":"sha256:c6b05af58bed907cf1bd04b852844ad24c0d44c507b61b722b6e68b473c724a3","observation_id":"3d4f5eef-eed7-425f-823a-030ea1e3c421","resolution":{"observed_at":"2026-07-14T11:05:51.130839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-02T06:42:08.783508Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.12252","last_updated":"2026-07-15T01:47:03Z","snapshot_observed_at":"2026-08-18T01:38:34.447496Z","submitted_at":"2026-07-14T01:31:20Z","title":"FinResearchBench II: A Deep Research Benchmark with Consensus-Derived Gold Rubrics for Distinguishing Financial Report Quality","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T06:42:08.783508Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2607.12252"},"observation_digest":"sha256:ceeb118d99ca91ca823fa2df4fdc489096c7a703e15faa2365c8382dec0db09c","observation_id":"15f66bc1-57d8-4608-886c-8243edea3907","resolution":{"observed_at":"2026-08-02T06:42:08.783508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-01T22:27:24.054724Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights.arXiv preprint arXiv:2602.02905, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.15766","last_updated":"2026-07-17T08:56:43Z","snapshot_observed_at":"2026-08-19T02:20:18.444655Z","submitted_at":"2026-07-17T08:56:43Z","title":"Before the Action: Benchmarking LLMs on Prospective Hypothesis Discovery","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T22:27:24.054724Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2607.15766"},"observation_digest":"sha256:9f88d531c2387dd7a17c0e38990ee8e4f90876f3b4404bd2fdf3a80ecd36a7e3","observation_id":"e1c5551e-0e51-4b1a-be11-471aa150cbf1","resolution":{"observed_at":"2026-08-01T22:27:24.054724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-01T12:50:00.281600Z","title":"FIRE-Bench: Evaluating AI agents on the rediscovery of scientific insights.arXiv preprint arXiv:2602.02905, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19321","last_updated":"2026-07-29T16:23:28Z","snapshot_observed_at":"2026-08-12T18:30:22.824623Z","submitted_at":"2026-07-21T17:41:12Z","title":"ResearchArena: Evaluating Sabotage and Monitoring in Automated AI R&D","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T12:50:00.281600Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2607.19321"},"observation_digest":"sha256:545745c69d7904171ad5b3ed54dfdfbc64d359efc69b277cbefb084377aeb23e","observation_id":"25037601-cfe4-4099-b06a-ba3ebc732871","resolution":{"observed_at":"2026-08-01T12:50:00.281600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02905","snapshot_observed_at":"2026-08-15T23:22:10.051070Z","title":"Fire-bench: Evaluating agents on the rediscovery of scientific insights.arXiv preprint arXiv:2602.02905, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12788","last_updated":"2026-08-13T03:48:07Z","snapshot_observed_at":"2026-08-20T04:51:03.577023Z","submitted_at":"2026-08-13T03:48:07Z","title":"ARAC: Benchmarking Auto-Research's Alignment and Completeness on End-to-End Researchs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T23:22:10.051070Z"},"links":{"cited_paper":"/paper/2602.02905","citing_paper":"/paper/2608.12788"},"observation_digest":"sha256:7d63b035cdca074c8a9cef83c85f7e0beb1a237693437a2594278987781126f5","observation_id":"d0398953-3ace-4068-b09a-45f52d3761a1","resolution":{"observed_at":"2026-08-15T23:22:10.051070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2602.02905/citation-record","integrity":"/paper/2602.02905/integrity","json":"/paper/2602.02905/citation-record.json","paper":"/paper/2602.02905"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.402142Z","title":"2.Experiment Completeness: Whether all key experiments from the paper are captured in the tree structure","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.402142Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:176edbbbec7ec9ddb62f920fbe466cd4e9a50e52e2e2c3131544dad4542fb893","observation_id":"60b5d011-6f12-4057-876a-fec078a6ca27","resolution":{"observed_at":"2026-08-03T05:17:18.402142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05376","last_updated":"2023-10-02T17:03:01Z","snapshot_observed_at":"2026-08-12T21:57:52.033689Z","submitted_at":"2023-04-11T17:41:13Z","title":"ChemCrow: Augmenting large-language models with chemistry tools","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05376","snapshot_observed_at":"2026-08-03T05:17:18.382660Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.382660Z"},"links":{"cited_paper":"/paper/2304.05376","citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:8e87284cd1017fe492e54a239dabfcd7595e1be6407be0a3fa0efe55d6819084","observation_id":"85532995-9951-458c-be16-567d8fb3682f","resolution":{"observed_at":"2026-08-03T05:17:18.382660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.404462Z","title":"4.Structural Coherence: Whether the hierarchical decomposition follows a logical parent-child relationship","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.404462Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:e8d964655e8d81157c71b0edc5843777db347aabbcf07e3daee9f1f89c2e08ab","observation_id":"b3016884-444a-4dcb-8cc6-da3351167143","resolution":{"observed_at":"2026-08-03T05:17:18.404462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18102","last_updated":"2025-03-23T15:16:42Z","snapshot_observed_at":"2026-08-19T23:01:07.335547Z","submitted_at":"2025-03-23T15:16:42Z","title":"AgentRxiv: Towards Collaborative Autonomous Research","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18102","snapshot_observed_at":"2026-08-03T05:17:18.388441Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.388441Z"},"links":{"cited_paper":"/paper/2503.18102","citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:6609f409766025f36827b19301ddfa82bae15ffe80273773ff57eabf66d76053","observation_id":"97034d38-aa55-41b9-8a0a-1decd5c58c65","resolution":{"observed_at":"2026-08-03T05:17:18.388441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01848","last_updated":"2025-04-07T12:15:49Z","snapshot_observed_at":"2026-07-06T21:03:06.857885Z","submitted_at":"2025-04-02T15:55:24Z","title":"PaperBench: Evaluating AI's Ability to Replicate AI Research","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.01848","snapshot_observed_at":"2026-08-03T05:17:18.391571Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.391571Z"},"links":{"cited_paper":"/paper/2504.01848","citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:c0077d2400a8e94a636e7ca55b27ae90a72cd6a7557c3de03484dad826f4664e","observation_id":"65a82c89-567e-44b8-82cd-f489ab58f16c","resolution":{"observed_at":"2026-08-03T05:17:18.391571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.394215Z","title":"Wang, X., Chen, Y ., Yuan, L., Zhang, Y ., Li, Y ., Peng, H., and Ji, H","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.394215Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:23d381a0c0baa1dd4a68f98f5d3bb48f6e7b57ffdca0487df3277bd829f24eb9","observation_id":"74c209ba-2920-48f4-8962-66a51de19d7d","resolution":{"observed_at":"2026-08-03T05:17:18.394215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00816","last_updated":"2025-03-08T14:01:34Z","snapshot_observed_at":"2026-08-17T08:54:33.605275Z","submitted_at":"2024-10-28T08:10:21Z","title":"CycleResearcher: Improving Automated Research via Automated Review","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00816","snapshot_observed_at":"2026-08-03T05:17:18.396653Z","title":"Weng, Y ., Zhu, M., Bao, G., Zhang, H., Wang, J., Zhang, Y ., and Yang, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.396653Z"},"links":{"cited_paper":"/paper/2411.00816","citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:0007fb2699fb5b456b8de09246e75a5765d0906bb56da7e9d2e290df5c47ff1b","observation_id":"497e45f6-f199-44ca-9d36-716652175e33","resolution":{"observed_at":"2026-08-03T05:17:18.396653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.16280","last_updated":"2025-07-22T06:51:26Z","snapshot_observed_at":"2026-08-19T08:04:04.473429Z","submitted_at":"2025-07-22T06:51:26Z","title":"ResearcherBench: Evaluating Deep AI Research Systems on the Frontiers of Scientific Inquiry","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.16280","snapshot_observed_at":"2026-08-03T05:17:18.399197Z","title":"11 FIRE-Bench: Evaluating Agents on the Rediscovery of Scientific Insights Xu, T., Lu, P., Ye, L., Hu, X., and Liu, P","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.399197Z"},"links":{"cited_paper":"/paper/2507.16280","citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:f45b6069f1795632eed33043742057f0369727db5d2ab91707bdb591d3254b17","observation_id":"acd4b857-3906-4289-97d5-7ed911cbef95","resolution":{"observed_at":"2026-08-03T05:17:18.399197Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.406752Z","title":"The results, shown in Table 9, demonstrate consistently high scores across all aspects, confirming the quality and reliability of the LLM-generated problem trees used in FIRE-Bench","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.406752Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:1b47205d0c63b780f7ec42ae10f4bd1448f0ea166696ae32ad911b1e50d249a7","observation_id":"2894f4f6-dfaa-4889-8c2f-fba9c0a6fa48","resolution":{"observed_at":"2026-08-03T05:17:18.406752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.409136Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.409136Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:0b1aea24c5f80057d01d7e06dc691c8afc6de0a70a7498be0e2929f9ace52a20","observation_id":"e346ac83-7587-43e9-8a17-cadc287323e9","resolution":{"observed_at":"2026-08-03T05:17:18.409136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.411320Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.411320Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:c779991765b1882fc5d98453f8f5b005bd47b454b775584f7cea37cc0ad0bf5e","observation_id":"4b21656a-ccaa-4a2d-ab2b-6e88bf3942b6","resolution":{"observed_at":"2026-08-03T05:17:18.411320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.413489Z","title":"RAGChecker atomized claims:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.413489Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:28eb2271b8b1027cb17169155eaf1d98a3f41a53db10128fc090ff222dbd8611","observation_id":"7aa854b5-5295-4794-8fd9-7b2d3109d286","resolution":{"observed_at":"2026-08-03T05:17:18.413489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.416426Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.416426Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:1f8c7ea5ee48017b08d391f65a02264b9848f5b6a99982470e32868ab67b0d35","observation_id":"6a0ae846-524a-4573-8a5c-538ec13481f9","resolution":{"observed_at":"2026-08-03T05:17:18.416426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.418547Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.418547Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:7f54a39d5d91af28fd38f18a0524231af066fcd1a5d6a42f6c7a074d1be2d382","observation_id":"477d1391-90f4-4294-b7b4-c5c831028d43","resolution":{"observed_at":"2026-08-03T05:17:18.418547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.420532Z","title":"Metrics:TP = 3, Precision = 1.0, Recall = 1.0, F1 = 1.0 Justification:Human claim 1 matches RAGChecker claim 1 (both assert early placement improves accuracy)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.420532Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:4a6bf735f716ff720b7cb54b327e3e17e4cecf22aac0428c21109a8aebbf3bbf","observation_id":"fd17c4b8-326a-41d2-a72c-ac4588afdc22","resolution":{"observed_at":"2026-08-03T05:17:18.420532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.423022Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.423022Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:4bc11809a7f55dcfbda19c450f7a2cd164e086ead78504026597b85f4ef42ff7","observation_id":"d1e30399-aec4-42db-9e38-c74987c133c0","resolution":{"observed_at":"2026-08-03T05:17:18.423022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.425163Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.425163Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:3f288b5f37bec426bba64a2eab9477ca2a127068d2999ec25efc027db6635fc4","observation_id":"73bc7f87-dd03-46f5-b7ca-51e6a50b4a1a","resolution":{"observed_at":"2026-08-03T05:17:18.425163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.427051Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.427051Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:5cf03941cebe954c9060dee28c64bf7378d2aa3ad11ce047296298943c7b6ac5","observation_id":"22f520c5-bd3f-4c38-b472-68208029aabd","resolution":{"observed_at":"2026-08-03T05:17:18.427051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.429683Z","title":"RAGChecker atomized claims:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.429683Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:ca81718d514a75a31539ddedca8ee5c670c14d7ac6414a06f0c3878767e74056","observation_id":"2f85c355-5ccc-4117-8b3e-7ad977a3f15b","resolution":{"observed_at":"2026-08-03T05:17:18.429683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.431618Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.431618Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:89ac3af1903dacffa5aa1a374daf8810f2c1f0b6d786dd5d20ff2bd4b823cba3","observation_id":"355d5656-0101-4fdb-9432-5cf081143deb","resolution":{"observed_at":"2026-08-03T05:17:18.431618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.434028Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.434028Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:b05095753c402e193f776a0b12406c47e7b1af4bf50391c6be535819649ab0da","observation_id":"5540bfc0-a11d-49f0-817c-b4200d77ae21","resolution":{"observed_at":"2026-08-03T05:17:18.434028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.435949Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.435949Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:f6c4ff899a8d530cc82a8c2fab9ae3d08bb1b1f283f5e9142fc93f10f544f92e","observation_id":"1168059c-bb01-4367-92e5-1e8eeef2d8bc","resolution":{"observed_at":"2026-08-03T05:17:18.435949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.438132Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.438132Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:46d9e460757294b9215d57680479c40b3aa117862d4dae2a58770a771cf27176","observation_id":"6d1e2341-4c4f-4a5b-8d61-c0fc19db9310","resolution":{"observed_at":"2026-08-03T05:17:18.438132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.440157Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.440157Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:7253c0faff2e12a45a24ab0992afe7046c4a6f7ae48d8787dca99cd713348761","observation_id":"3f10cf8d-b672-4446-b153-82c2cf059a4f","resolution":{"observed_at":"2026-08-03T05:17:18.440157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.441996Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.441996Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:b03f80b1e6492c21e64ba1a640e794d99f442a199c445cbf19803666a6632074","observation_id":"bc92d0a5-1997-4c97-9c09-e1e7e2b7d3bc","resolution":{"observed_at":"2026-08-03T05:17:18.441996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.444226Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.444226Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:9666d0ae6596033f8b2a8f8fdf85ba825ca7d7274698dc59c062e9966950175c","observation_id":"d82287d3-bb57-4f8d-94b7-7915c525684f","resolution":{"observed_at":"2026-08-03T05:17:18.444226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.446247Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.446247Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:987470b7e225279bb83f5551612eefbf40949fb8eaa101e319df0944e64d7a8b","observation_id":"6bc373ca-f690-484f-be95-3b20e69ef209","resolution":{"observed_at":"2026-08-03T05:17:18.446247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.448150Z","title":"RAGChecker atomized claims:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.448150Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:9002b6fc17665097c69f573a030daf277ec7e0309ad00cfd2afb042e35d85a7d","observation_id":"552975c7-f36f-4e3f-bfcf-82c7dc9a6f0c","resolution":{"observed_at":"2026-08-03T05:17:18.448150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.450332Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.450332Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:7a74a6124cb99ff79262c2bfdb92a72be6560b3136a8ffdadd447374fae39103","observation_id":"e36a50f1-0890-420b-b975-0e36220e5cc7","resolution":{"observed_at":"2026-08-03T05:17:18.450332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.452604Z","title":"23 FIRE-Bench: Evaluating Agents on the Rediscovery of Scientific Insights","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.452604Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:7b8c1ee0bc0f4cba5e1295bb1b4ca728d0839ef50fe6f1159d21e45c89231580","observation_id":"6a2a1ba0-7e65-43f2-9229-c9b52fb6428b","resolution":{"observed_at":"2026-08-03T05:17:18.452604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.454632Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.454632Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:f9fd140bcac5e2e33031d59d7573a820f319d1ea48a642e9dafcc086b4083741","observation_id":"b8b80036-acd3-4e72-8468-47914f61206f","resolution":{"observed_at":"2026-08-03T05:17:18.454632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.456565Z","title":"Models better at using relevant information at beginning of input context and the performance drops in later position","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.456565Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:a351859fd8f0b7376027091ca46c30552a8ac73e0db07556eac067c4ec3609a3","observation_id":"0623cee0-2dd3-4153-8712-a759d2e157f3","resolution":{"observed_at":"2026-08-03T05:17:18.456565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.458929Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.458929Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:baf97d2a08ecd65fc2b5b784ef115f2df51061229ec7739ed8f6d9b5040b54ce","observation_id":"438ef993-8b29-4ac9-8fa4-72e397a28885","resolution":{"observed_at":"2026-08-03T05:17:18.458929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.461003Z","title":"first... second","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.461003Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:01d149529fa63d2b24a8945a75067757861dc1bd8777b01558d94f9e6804fe46","observation_id":"3453e61c-a3a7-482e-90c0-aedc35aad6f5","resolution":{"observed_at":"2026-08-03T05:17:18.461003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.463236Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.463236Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:aab12dffed76fa3049fa0f1f1b80a36c62a9e33a5b678fabcd84d94994570cdf","observation_id":"8eec02cb-75eb-4023-83f8-416861793dc3","resolution":{"observed_at":"2026-08-03T05:17:18.463236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.465475Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.465475Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:ad9228a7def62854bfbea772ca5be85cb49ea6716046a0e89f1a27348a0e7ab3","observation_id":"490dcc49-928d-4493-a08e-761684ea4651","resolution":{"observed_at":"2026-08-03T05:17:18.465475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.467556Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.467556Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:ecf4fc81603851551fcee1509c98973f17187f7f894aad0b691d93223deffe45","observation_id":"4202df0a-fa54-4014-a08e-d5c2be75b097","resolution":{"observed_at":"2026-08-03T05:17:18.467556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.469520Z","title":"paper\": {","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.469520Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:ede4c19fb3027a700579b23b0d9df294ce54821e94796c82c012c4cdc0729fca","observation_id":"8b623832-05c7-4c1d-963a-dcb5dbe47b3a","resolution":{"observed_at":"2026-08-03T05:17:18.469520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.473651Z","title":"{gt}\". And the false negative conclusion missed by AI research agent:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.473651Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:c264f6e60ad7ef206f29b16cfd351caada28683dda28c2165a482d42ecbd3982","observation_id":"9af276e9-f419-4a2d-add7-9cfeb2f711c4","resolution":{"observed_at":"2026-08-03T05:17:18.473651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.475908Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.475908Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:79ec2843426bda884aa20b4f6e5e3bcde77afa7514e3f461edc82b136ba420b1","observation_id":"d12c8058-1e34-4aef-8588-fd9326be0319","resolution":{"observed_at":"2026-08-03T05:17:18.475908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T05:17:18.477910Z","title":"{gt}\". And the false positive conclusion generated by an AI research agent:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.477910Z"},"links":{"citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:57e287e49cf68d00f8f98cc2998088d49391277d989fa7b3cc160541c20341b8","observation_id":"8152cf2a-c86d-4d62-8bf9-092ecde2c23b","resolution":{"observed_at":"2026-08-03T05:17:18.477910Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21248","last_updated":"2026-04-20T16:53:43Z","snapshot_observed_at":"2026-07-06T20:59:31.273397Z","submitted_at":"2025-03-27T08:09:15Z","title":"ResearchBench: Benchmarking LLMs in Scientific Discovery via Inspiration-Based Task Decomposition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21248","snapshot_observed_at":"2026-08-03T05:17:18.385882Z","title":"URL https:// aclanthology.org/2024.tacl-1.9/","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.385882Z"},"links":{"cited_paper":"/paper/2503.21248","citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:983647867289d25181e61f3935c9944838734142cf59ce29a876126620ff2de9","observation_id":"d5f29e90-b949-46ec-9c5c-79d65cf60fb8","resolution":{"observed_at":"2026-08-03T05:17:18.385882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07738","last_updated":"2025-02-09T08:15:44Z","snapshot_observed_at":"2026-08-16T14:01:45.252212Z","submitted_at":"2024-04-11T13:36:29Z","title":"ResearchAgent: Iterative Research Idea Generation over Scientific Literature with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07738","snapshot_observed_at":"2026-08-03T05:17:18.379513Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T05:17:18.379513Z"},"links":{"cited_paper":"/paper/2404.07738","citing_paper":"/paper/2602.02905"},"observation_digest":"sha256:664050a8c9203b9e5367e45c36b015ec264df50c23356b9a01860e918eb7bae8","observation_id":"da817f0d-259e-41ab-a4eb-2561511ca1a3","resolution":{"observed_at":"2026-08-03T05:17:18.379513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2602.02905","last_updated":"2026-07-10T18:07:40Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-21T02:11:00.048335Z","submitted_at":"2026-02-02T23:21:13Z","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":41,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 13 inbound Pith citation observations for arXiv:2602.02905."}