{"as_of":"2026-08-09T00:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:63d256488d5c58439ed1e22b9589dea2748508f83e1ae2abced666810e0bcfcd","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T23:30:43.662827Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:26:56.639836Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-25T08:15:33.926301Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2504.19678","last_updated":"2026-03-06T19:01:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-28T11:08:22Z","title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-15T02:57:37.873567Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2504.19678"},"observation_digest":"sha256:d815f500efa7b65d7ffc697a60694e888e0494a3f5281699097476eef955cfc3","observation_id":"94a3833b-a525-47d0-b967-5d248b0f689d","resolution":{"observed_at":"2026-05-15T02:57:38.309833Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-08-07T15:26:56.639836Z","title":"arXiv preprint arXiv:2502.08859 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-07T21:22:21.405277Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.639836Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4ab189391cd68da460f7750d3ec0e38c37c95531fc37f69cd91e8afdaef42cff","observation_id":"57530e12-cda5-41fe-bf7f-6574601f213e","resolution":{"observed_at":"2026-08-07T15:26:56.639836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-08-07T15:08:30.334466Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16135","last_updated":"2025-05-22T02:24:35Z","snapshot_observed_at":"2026-08-07T15:04:11.681192Z","submitted_at":"2025-05-22T02:24:35Z","title":"Sudoku-Bench: Evaluating creative reasoning with Sudoku variants","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:30.334466Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2505.16135"},"observation_digest":"sha256:86cdc85dcff67be41bb07002bda24280932a2da1498305c5b64ddf1e746778d5","observation_id":"e80c223a-4010-4c5b-ac75-717ba7a021e7","resolution":{"observed_at":"2026-08-07T15:08:30.334466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2506.06211","last_updated":"2026-04-21T03:40:06Z","snapshot_observed_at":"2026-08-03T01:28:16.465454Z","submitted_at":"2025-06-06T16:17:09Z","title":"PuzzleWorld: A Benchmark for Multimodal, Open-Ended Reasoning in Puzzlehunts","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-19T10:43:02.601014Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2506.06211"},"observation_digest":"sha256:3a8fc06706076c81e6ea73b75bdf3ad7cea046ad6ff5952e6aed3ea8bfd561fb","observation_id":"3b4b129e-89df-44b5-ae33-826b8ebefe91","resolution":{"observed_at":"2026-05-19T10:47:15.164738Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2510.08945","last_updated":"2026-05-21T22:37:46Z","snapshot_observed_at":"2026-07-06T22:32:17.816175Z","submitted_at":"2025-10-10T02:51:47Z","title":"FATHOMS-RAG: A Framework for the Assessment of Thinking and Observation in Multimodal Systems that use Retrieval Augmented Generation","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T08:13:05.328746Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2510.08945"},"observation_digest":"sha256:8a277e4ad333794dcc2e26e7d643d36eb633915361e1ba1de42755febddecbd8","observation_id":"048dd1b5-73d0-4c8c-aeca-b028b91d41fe","resolution":{"observed_at":"2026-05-25T08:15:33.930689Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":"2502.08859","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enigmaeval: A benchmark of long multimodal reasoning challenges","venue":null,"work_id":"0cc9ff97-d72f-4232-b771-d3c969ededbc","year":2025},"citing_paper":{"arxiv_id":"2605.14040","last_updated":"2026-05-13T19:00:57Z","snapshot_observed_at":"2026-07-06T23:25:29.856145Z","submitted_at":"2026-05-13T19:00:57Z","title":"Physics-R1: An Audited Olympiad Corpus and Recipe for Visual Physics Reasoning","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-15T05:35:32.806871Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2605.14040"},"observation_digest":"sha256:cc27027a38488bfc177c538b2ac04f7b1ee569fa8971c57307e3919d655160e8","observation_id":"8ee27dfd-b016-448f-9762-eac45709a9c7","resolution":{"observed_at":"2026-05-15T05:39:47.813355Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.08859/citation-record","integrity":"/paper/2502.08859/integrity","json":"/paper/2502.08859/citation-record.json","paper":"/paper/2502.08859"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.152706Z","title":"https://puzzledpint.org/","venue":null,"work_id":"e197f0bc-b382-4b43-b193-d33bfe557cd6","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.527786Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:618ffea45bc1b7cccc85b9955f35707d6f35bf3cf22b333d294e0aaf6698e98a","observation_id":"82346c5b-96d1-42e8-81cb-8cd0681ce094","resolution":{"observed_at":"2026-08-07T23:30:44.156381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.141905Z","title":null,"venue":null,"work_id":"5ef93580-5187-43c3-8366-df8dfb72cc8e","year":2025},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.531958Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:2c94de7e3033e3558218b701b35011e63fa00e2bad7074d8ad805c46aaf54f20","observation_id":"843cd5f7-0e66-41bd-a6cc-633ae07e1a93","resolution":{"observed_at":"2026-08-07T23:30:44.145614Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.130060Z","title":"Puzzle Potluck","venue":null,"work_id":"d52b91ff-5428-440a-b23b-c3023dc7a109","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.535582Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3349fcd3f6642ddfadfe7f84d78f147c2d90ab126b508f8981e929d21ee68a25","observation_id":"5c814eac-f57c-4476-a4be-ec1b1dcaaf93","resolution":{"observed_at":"2026-08-07T23:30:44.133987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.119456Z","title":"SOME PUZZLES by Mark Halpin","venue":null,"work_id":"1b9fe7f0-4cf0-4511-9162-b6675e8597dc","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.539340Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3afcb19050ab1f138ec985551b7660a09e906e9b72e275f59f8f6db4c68e7c40","observation_id":"07555f79-d9b5-4272-9dae-060dc3c4e83d","resolution":{"observed_at":"2026-08-07T23:30:44.123189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.107955Z","title":null,"venue":null,"work_id":"05af4f2e-6ad0-4550-b702-17293f96981b","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.542911Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:563ce477286d762c5bcf1e977efb813b8e01059a2030d893d5806ceeb0bc442b","observation_id":"5126262f-5fa0-42a8-875d-6c8ad29cd47c","resolution":{"observed_at":"2026-08-07T23:30:44.112500Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.097280Z","title":"https://puzzles.mit.edu/","venue":null,"work_id":"5e321ef2-0ed6-42a0-95e6-d0292e1a7aa2","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.546828Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:d7d6ab29abe70409caf41c80ec5b072775e33821470832fff487292c064fa6fb","observation_id":"8ae17add-ffd2-4fa9-b4b2-a6c2515e9957","resolution":{"observed_at":"2026-08-07T23:30:44.100847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.085985Z","title":"Grandmaster Puzzles","venue":null,"work_id":"7d4d5a31-906b-4952-8f8f-e87c59499702","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.550584Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:0fec223fc1babef3a65e04db8c6eeb5552f409c74b61dbf3d085054f7612a6dc","observation_id":"b2dfcc65-8744-4ff9-81e1-10de40e61f08","resolution":{"observed_at":"2026-08-07T23:30:44.090026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.554415Z","title":"Measuring mathematical problem solving with the math dataset, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.554415Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:8b7f6e0f388231495fbaf8986830b306aab76722a6348e1acb7aff6ec247c5be","observation_id":"7cd582dc-3fd8-469c-a10f-7f7fbcb531af","resolution":{"observed_at":"2026-08-07T23:30:43.554415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.557938Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.557938Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:e841fc95d511692e0ea03384ce8afd63025dfb90546fd8b737bdbcecb82f2905","observation_id":"391adf69-e187-44ba-acb9-c27e1c67651b","resolution":{"observed_at":"2026-08-07T23:30:43.557938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.060792Z","title":"Frontiermath: A benchmark for evaluating advanced mathematical reasoning in ai, 2024","venue":null,"work_id":"b1274e2b-dd59-4f59-96d2-a2c4fe784473","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.561371Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:66b59f214dd2ada7f75e1e9e6825a7445342c3d847f02b350abad523951adaf2","observation_id":"9d881114-a65c-439d-850e-23a6e4529837","resolution":{"observed_at":"2026-08-07T23:30:44.065829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.564791Z","title":"Olympiadbench: A challenging benchmark for promoting agi with olympiad-level bilingual multimodal scientific problems, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.564791Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3a6978b94c4446101a82997cb69bde78af5ef619be4ea7eb91d08b6203900c1e","observation_id":"0cd4cd2d-5df6-4280-bfbb-f1383fef72a4","resolution":{"observed_at":"2026-08-07T23:30:43.564791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.043119Z","title":"Humanity’s Last Exam, 2025","venue":null,"work_id":"c7086602-f360-4727-bd77-73b952aa0471","year":2025},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.568159Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:69457466b94b070c99a9363d9cad22daf3bb932d7ddf58e6a95345b731a4ef8a","observation_id":"7d64a4a5-0732-4d6c-bbf7-a631a43aa849","resolution":{"observed_at":"2026-08-07T23:30:44.046677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.571998Z","title":"Measuring massive multitask language understanding, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.571998Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:f71c9ccef1d5fd590ced1de3a72ca04d14bd753f8357effe4201ee58b6bffd62","observation_id":"b177cc5e-d96f-4155-9f66-ae21f995b0b1","resolution":{"observed_at":"2026-08-07T23:30:43.571998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.026335Z","title":"Mmmu: A massive multi- discipline multimodal understanding and reasoning benchmark for expert agi","venue":null,"work_id":"6eb46864-850d-467f-b9bb-212df0b91540","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.575607Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:ad7fb528f8fb6abc9c8b4010c344155535cfdd5588c6d3c1f7603c09719456ad","observation_id":"8f5d7d3e-4517-4098-8eb6-4c80358f5981","resolution":{"observed_at":"2026-08-07T23:30:44.030082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.578906Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.578906Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:8563f38f14869a53de0266de186398f9eebc106809a28b421413fd474b51ad96","observation_id":"65f7a8ab-ff2c-48cc-a61b-b9bc36381e13","resolution":{"observed_at":"2026-08-07T23:30:43.578906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:44.010171Z","title":"Vista: A rubric- based visual task assessment","venue":null,"work_id":"35be598a-a734-40b3-b803-84337149d259","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.582228Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3595cce22ce708b72fe86420d83376bdc49b8f42e5d95c816a0024390cd98d59","observation_id":"460bbc65-3f94-4551-9fc9-f93c875e9445","resolution":{"observed_at":"2026-08-07T23:30:44.013743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.585618Z","title":"On the measure of intelligence, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.585618Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3bf0a6a58a5dbb5e9ac77b4804193fa19b43a0c78d0c3b6cf9a58943a38dfd89","observation_id":"cfc927e1-817f-462c-a866-d9bbabc01ebe","resolution":{"observed_at":"2026-08-07T23:30:43.585618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.993984Z","title":"Lanzendörfer, Yannick Niedermayr, and Roger Wattenhofer","venue":null,"work_id":"b7f7aaaa-f22a-46ed-8af2-3d36fd5e9019","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.588907Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:c3a21337db187e4b2a0868bed197296d406e4e39c789df03e3fdfc9eb9bc48d5","observation_id":"8c0601a4-d84c-47d8-905a-eef165df06a6","resolution":{"observed_at":"2026-08-07T23:30:43.997755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.983687Z","title":"PuzzlePlex: A Benchmark to Evaluate the Reasoning and Planning of Large Language Models on Puzzles, 2025","venue":null,"work_id":"9d6e9343-2f93-47d2-a074-0ce4d5134811","year":2025},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.592101Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:9f2b56d2a4cc4f4061c8e407f841cf8283cf5a4cb508e52c9c39f866a04ad031","observation_id":"5ac52958-4155-4a5f-98e0-eb7b092d71c4","resolution":{"observed_at":"2026-08-07T23:30:43.987521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02611","last_updated":"2025-03-01T12:46:25Z","snapshot_observed_at":"2026-08-06T18:34:26.106835Z","submitted_at":"2024-02-04T20:56:09Z","title":"FCoReBench: Can Large Language Models Solve Challenging First-Order Combinatorial Reasoning Problems?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02611","snapshot_observed_at":"2026-08-07T23:30:43.595386Z","title":"PuzzleBench: Can LLMs Solve Challenging First-Order Combinatorial Reasoning Problems? arXiv preprint arXiv:2402.02611, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.595386Z"},"links":{"cited_paper":"/paper/2402.02611","citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:7043c1e5b871c1c6490b0c8059b45e77f08b7dd9114926ae9545362baf8de5d0","observation_id":"d18500c4-f372-48a4-a7e0-2d5e38939151","resolution":{"observed_at":"2026-08-07T23:30:43.595386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14790","last_updated":"2024-10-04T04:58:12Z","snapshot_observed_at":"2026-07-06T18:49:25.288816Z","submitted_at":"2024-07-20T07:43:07Z","title":"Step-by-Step Reasoning to Solve Grid Puzzles: Where do LLMs Falter?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14790","snapshot_observed_at":"2026-08-07T23:30:43.599486Z","title":"Step-by-Step Reasoning to Solve Grid Puzzles: Where do LLMs Falter? arXiv preprint arXiv:2407.14790, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.599486Z"},"links":{"cited_paper":"/paper/2407.14790","citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:e821059381265729bf9711aefae513fe53900a73557a932a581911fde054a320","observation_id":"46e6cb43-c7dd-4a51-8eb6-9c7d8dccd28e","resolution":{"observed_at":"2026-08-07T23:30:43.599486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00376","last_updated":"2021-07-04T22:50:32Z","snapshot_observed_at":"2026-08-09T00:04:22.828648Z","submitted_at":"2021-01-02T05:28:15Z","title":"RiddleSense: Reasoning about Riddle Questions Featuring Linguistic Creativity and Commonsense Knowledge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00376","snapshot_observed_at":"2026-08-07T23:30:43.603305Z","title":"Riddlesense: Answering riddle questions as commonsense reasoning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.603305Z"},"links":{"cited_paper":"/paper/2101.00376","citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:90bec5e913b4edb11a8ff5350b6ff7b3d7d4809c562e71d1ff75a64819d86680","observation_id":"d7a2bf39-b9b4-4d99-a75d-af9f8f300053","resolution":{"observed_at":"2026-08-07T23:30:43.603305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.973643Z","title":"https://www.melbunimathsstats.org/puzzlehunt","venue":null,"work_id":"de9e17a2-e887-4fad-9b54-06fa8281621c","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.607095Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:b85a9bfdc3f60608e11cd94243d91b11743435eacde82cf5fc0e25ed258d4d56","observation_id":"e6bf3606-0828-4f48-9380-73d8c2b2cc1d","resolution":{"observed_at":"2026-08-07T23:30:43.977237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.610435Z","title":"https://web.archive.org/web/20210725192741/https://www","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.610435Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:16feac93b4ecdb2f8919b2ffe43a08a8f63535ee834778a6f83981043f0e4948","observation_id":"b1cee88c-c0d9-446e-b0f4-45668caf367e","resolution":{"observed_at":"2026-08-07T23:30:43.610435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.964008Z","title":"https://harvardpuzzles.github.io/","venue":null,"work_id":"2675e864-cbfb-44d9-b8a6-e49206236817","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.613709Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:9f4d04196c042ee6e4c507191d18b23455a80a05ac7a6bf1bee6c9eae82e6fe7","observation_id":"342ba2ad-cb5f-49c1-ae7e-e7f7828dfbdd","resolution":{"observed_at":"2026-08-07T23:30:43.967490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.954153Z","title":"https://www.mezzacotta.net/puzzle/cisra/","venue":null,"work_id":"ec4c73ac-dc9f-4fbe-a621-92c76f8bc86f","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.616875Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:48ddba281aebd771cc64ebf3f7823534910fec71ea355f2b08a45c35bda4e202","observation_id":"c7fdeb96-070a-4dc1-aaeb-f1ba1088d752","resolution":{"observed_at":"2026-08-07T23:30:43.957817Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.944353Z","title":"https://www.janestreet.com/puzzles/archive/index.html","venue":null,"work_id":"11cb608a-1090-47aa-ac3f-aa5d5d178887","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.620473Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:3a080e282e664c67b41847d83e2a1487d3c6cf2e2a85fa6af8fab92f5321c533","observation_id":"503937d6-4c14-453e-ac0c-00bc546d01df","resolution":{"observed_at":"2026-08-07T23:30:43.947801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.934699Z","title":"https://gooooogol.theburninators.org/puzzles/","venue":null,"work_id":"c09a476b-b4bc-4e3e-8fb7-0dc7b1b3065b","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.624023Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:6d3f050777cea599dab4dcad09b8156f16689bad3b23a32a9fc39c1cd53a5bee","observation_id":"3a1b91e4-717d-4d1e-af44-e612c8e06b68","resolution":{"observed_at":"2026-08-07T23:30:43.938099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.925243Z","title":"https://playdash.org/","venue":null,"work_id":"3a238520-6b4f-4587-b771-ce62644b17cf","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.627592Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:688fcb3c978753a0066a7ea73fc28ca37a1b49af86bbdd6f5b9369a2c6d7ca8c","observation_id":"a1c18164-aca6-46c1-821a-3a837d60b995","resolution":{"observed_at":"2026-08-07T23:30:43.928572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.915660Z","title":"https://www.baphl.org/","venue":null,"work_id":"5f0360c9-0815-4280-b281-f44e3a9109d2","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.631361Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:49e44707dbe5e0bbfc44544b36f5d0ce8c96dc6eb4499b03ce748e8630da5e67","observation_id":"1776b629-139b-4d48-a687-d23cda9ee5a3","resolution":{"observed_at":"2026-08-07T23:30:43.918925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.905535Z","title":"Forbidden actions","venue":null,"work_id":"a2dc60e9-d78b-420a-9309-9a5d8a7fddf2","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.634785Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:5ba0ec028955e8045baf0e7b668a069eea83ca1ec38eb3ceb8e521c08c6058ac","observation_id":"f509d110-5f11-4786-99f7-72c04cdb207d","resolution":{"observed_at":"2026-08-07T23:30:43.909101Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.895784Z","title":null,"venue":null,"work_id":"4136f01a-cefe-4c6d-92d8-8b3ad089f192","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.638190Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:0a2c880869b176952dbabb514f9a01d9ca1a4cf22da9a6c311a5e280183938af","observation_id":"49d8ad0e-f128-40dc-9fbe-5178ccecdb90","resolution":{"observed_at":"2026-08-07T23:30:43.899181Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.885485Z","title":null,"venue":null,"work_id":"cab37904-8fb0-47d1-b930-67790dc526c1","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.641496Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:c66d613a0c5423e328a1d986734222a0ebecf6d787a20db18ca36da9443d3095","observation_id":"de4a874d-eb72-42cd-87eb-a1f8537c0f43","resolution":{"observed_at":"2026-08-07T23:30:43.889272Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.874627Z","title":null,"venue":null,"work_id":"75c5b606-e16b-4682-8ce7-f7073793e63e","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.645200Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:83121ff6a0c0a360fa739aaf0abc43d2b5e0b5761d07a45e3815075daf5a83aa","observation_id":"79fa53c8-27b7-4cb2-8e7c-6eb38e6962f6","resolution":{"observed_at":"2026-08-07T23:30:43.878286Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.863673Z","title":null,"venue":null,"work_id":"590bb88c-759e-4dda-8dbc-c5f943f4be9d","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.648403Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:11554f4fd098ea796ed279307615f0ebbeb727d5c1c37147541381425cc7b2de","observation_id":"708a5c9a-11ba-41b1-9db5-99094dc2a39c","resolution":{"observed_at":"2026-08-07T23:30:43.867120Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.853003Z","title":"Problem Web Wage","venue":null,"work_id":"225c7e9c-f81b-43ae-bde8-809cc6e20561","year":2024},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.651991Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:66a09c8fe3a67625068b01f948512d5e5e9536d27aaf3298bb97df3a6f19830d","observation_id":"25a673f7-2449-4694-8b86-389c4c22039d","resolution":{"observed_at":"2026-08-07T23:30:43.856528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.842607Z","title":null,"venue":null,"work_id":"8f70936b-389b-4a96-bad9-9c02df12f1cb","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.655828Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:2b8c1caec0f3563aaa4737b6e84229cd0824c8927032eb767584d5c44580e80a","observation_id":"ea411947-d5fb-4d7c-8130-9d9b09a33023","resolution":{"observed_at":"2026-08-07T23:30:43.846067Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.832352Z","title":null,"venue":null,"work_id":"d3ec59d8-29ba-4f69-86a5-cc4eda7ee839","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.659550Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:66b9d866a6342bcb78c2a51f18a2f37ff2542003b46712876c1dd3fef926b57a","observation_id":"cd0e5dc9-42fa-4b02-8123-e1e335477d5a","resolution":{"observed_at":"2026-08-07T23:30:43.835897Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:30:43.819164Z","title":"This structured approach to answer formats allows us to extract answers consistently and reduces ambiguity when comparing model outputs to ground-truth solutions","venue":null,"work_id":"a0f08d6b-3e75-4922-a5ae-3169996cb8ae","year":null},"citing_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T23:30:43.662827Z"},"links":{"citing_paper":"/paper/2502.08859"},"observation_digest":"sha256:33516c6458a40e632ff6d318cd28cd301e42c957d212ce684e3c1cb24e4eb406","observation_id":"d052f01f-dada-4f97-b0b3-42046345fa91","resolution":{"observed_at":"2026-08-07T23:30:43.824654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T23:24:48.758803Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":0,"verified_fuzzy":21},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 6 inbound Pith citation observations for arXiv:2502.08859."}