{"as_of":"2026-08-06T09:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:305c3fdbe6ef5b410c4a972161ff4e184b674328d0454823b94cd15332ea4e28","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T23:55:49.639185Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T16:15:06.153368Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":"2404.05955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-08T16:15:06.153368Z","title":"arXiv preprint arXiv:2404.05955 , year=","venue":"cs.CL","work_id":"b65ed183-95d3-4265-a05b-55aa2ec45c7a","year":2024},"citing_paper":{"arxiv_id":"2401.05459","last_updated":"2024-05-08T06:16:23Z","snapshot_observed_at":"2026-08-02T13:57:57.119489Z","submitted_at":"2024-01-10T09:25:45Z","title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","version":2},"reference_index":136,"source":"pdf_text","source_observed_at":"2026-05-17T00:57:26.303195Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2401.05459"},"observation_digest":"sha256:3c03aad0c578e7edec0c4dea13635c603ec65d4304eb325d20ff9ff66f444966","observation_id":"4e1379bf-4769-48f9-97d2-52098f1f293d","resolution":{"observed_at":"2026-05-17T00:57:26.547308Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":"2404.05955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-08T16:15:06.153368Z","title":"arXiv preprint arXiv:2404.05955 , year=","venue":"cs.CL","work_id":"b65ed183-95d3-4265-a05b-55aa2ec45c7a","year":2024},"citing_paper":{"arxiv_id":"2411.18279","last_updated":"2025-05-06T15:08:00Z","snapshot_observed_at":"2026-07-06T19:57:55.925634Z","submitted_at":"2024-11-27T12:13:39Z","title":"Large Language Model-Brained GUI Agents: A Survey","version":12},"reference_index":218,"source":"pdf_text","source_observed_at":"2026-05-19T11:08:27.472508Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2411.18279"},"observation_digest":"sha256:5228bb30a74536f38574876eb631af1af1adc9aec05c85677528c359212f13c1","observation_id":"012d0a51-fcba-4c5d-a8b4-f3b6f5d65818","resolution":{"observed_at":"2026-05-19T11:08:27.778355Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":"2404.05955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-08T16:15:06.153368Z","title":"arXiv preprint arXiv:2404.05955 , year=","venue":"cs.CL","work_id":"b65ed183-95d3-4265-a05b-55aa2ec45c7a","year":2024},"citing_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-02T16:13:31.498470Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-11T05:26:04.960844Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2505.07062"},"observation_digest":"sha256:1b8b0470bd5e41c82b5c47430156569a0e90b154ecdcc60c293e64090e73da56","observation_id":"45680f7d-05e7-4bf0-8c52-88abf138915e","resolution":{"observed_at":"2026-05-11T05:26:05.946519Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-08-05T23:55:49.639185Z","title":"Visualwebbench: How far have multimodal llms evolved in web page understanding and grounding? arXiv preprint arXiv:2404.05955, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04700","last_updated":"2025-08-12T15:11:53Z","snapshot_observed_at":"2026-08-05T23:55:45.338719Z","submitted_at":"2025-08-06T17:58:46Z","title":"SEAgent: Self-Evolving Computer Use Agent with Autonomous Learning from Experience","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T23:55:49.639185Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2508.04700"},"observation_digest":"sha256:2f11efecc95705321e1c3ccab131613b779be019229d67ee0dd9572af982903c","observation_id":"3419e5f9-f23f-4dee-9cbc-ad7f3ed5bbb0","resolution":{"observed_at":"2026-08-05T23:55:49.639185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-08-05T14:03:37.262733Z","title":"Visualwebbench: How far have multimodal llms evolved in web page understanding and grounding? arXiv preprint arXiv:2404.05955, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21767","last_updated":"2025-08-29T16:40:57Z","snapshot_observed_at":"2026-08-06T03:24:39.373354Z","submitted_at":"2025-08-29T16:40:57Z","title":"UItron: Foundational GUI Agent with Advanced Perception and Planning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T14:03:37.262733Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2508.21767"},"observation_digest":"sha256:9367190b2444cae7a41d58e593489b3898135feead15cff6dc61c6109b539c93","observation_id":"5edcc879-91b4-4fd9-928a-13aad209714d","resolution":{"observed_at":"2026-08-05T14:03:37.262733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-08-05T12:59:06.413484Z","title":"Visualwebbench: How far have multimodal llms evolved in web page understanding and grounding?arXivpreprintarXiv:2404.05955, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.01106","last_updated":"2025-09-11T12:40:54Z","snapshot_observed_at":"2026-08-05T22:16:12.493867Z","submitted_at":"2025-09-01T03:53:47Z","title":"Robix: A Unified Model for Robot Interaction, Reasoning and Planning","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T12:59:06.413484Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2509.01106"},"observation_digest":"sha256:415d28e9e238aa5eb3a9de00e36701de093d15b26422ec1ca7380b58c685caec","observation_id":"161fceba-5e13-4c5c-bea0-3232da40bec5","resolution":{"observed_at":"2026-08-05T12:59:06.413484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":"2404.05955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-08T16:15:06.153368Z","title":"arXiv preprint arXiv:2404.05955 , year=","venue":"cs.CL","work_id":"b65ed183-95d3-4265-a05b-55aa2ec45c7a","year":2024},"citing_paper":{"arxiv_id":"2605.18048","last_updated":"2026-05-18T08:36:04Z","snapshot_observed_at":"2026-07-06T23:28:59.503300Z","submitted_at":"2026-05-18T08:36:04Z","title":"DocOS: Towards Proactive Document-Guided Actions in GUI Agents","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-20T11:13:48.529796Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2605.18048"},"observation_digest":"sha256:702eca4ff6e7ad27450974fc2e62474bf0de5c68bf9c9b3fe22114455e8b76e6","observation_id":"9bc3b8ab-3acb-4cf4-80a6-49d71ba73a65","resolution":{"observed_at":"2026-05-20T11:18:13.836387Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":"2404.05955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-08T16:15:06.153368Z","title":"arXiv preprint arXiv:2404.05955 , year=","venue":"cs.CL","work_id":"b65ed183-95d3-4265-a05b-55aa2ec45c7a","year":2024},"citing_paper":{"arxiv_id":"2607.02032","last_updated":"2026-07-06T17:18:28Z","snapshot_observed_at":"2026-08-03T13:49:10.443476Z","submitted_at":"2026-07-02T10:59:03Z","title":"PACE: A Proxy for Agentic Capability Evaluation","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-07-03T13:34:39.350893Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2607.02032"},"observation_digest":"sha256:1883da0ac937a06903e18aa06bad4917ec6330469bfd16f90e39c55575dfae23","observation_id":"7ba40c9d-fe48-49ab-b9fb-f30ea15acccd","resolution":{"observed_at":"2026-07-03T13:38:18.458668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-12T08:29:58.561496Z","title":"arXiv preprint arXiv:2404.05955 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02032","last_updated":"2026-07-06T17:18:28Z","snapshot_observed_at":"2026-08-03T13:49:10.443476Z","submitted_at":"2026-07-02T10:59:03Z","title":"PACE: A Proxy for Agentic Capability Evaluation","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-07-12T08:29:58.561496Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2607.02032"},"observation_digest":"sha256:f85a0e0205b84c442f256251a848828812f4d3e883ca4094cfa259874be482c2","observation_id":"94f74fcc-0954-4848-b120-156b7c4611e5","resolution":{"observed_at":"2026-07-12T08:29:58.561496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":"2404.05955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-08T16:15:06.153368Z","title":"arXiv preprint arXiv:2404.05955 , year=","venue":"cs.CL","work_id":"b65ed183-95d3-4265-a05b-55aa2ec45c7a","year":2024},"citing_paper":{"arxiv_id":"2607.06118","last_updated":"2026-07-07T10:27:31Z","snapshot_observed_at":"2026-08-02T12:30:22.342026Z","submitted_at":"2026-07-07T10:27:31Z","title":"WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-08T16:11:49.725631Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2607.06118"},"observation_digest":"sha256:f6dcca753e31e78604e1e83bc50cbd130c2d0f7b13944102a6331d34ffcc13ce","observation_id":"8e37b665-4417-4cd4-acbf-bbb52ec8eb24","resolution":{"observed_at":"2026-07-08T16:15:06.155483Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-07-31T23:04:31.363636Z","title":"Visu- alwebbench: How far have multimodal llms evolved in web page understanding and grounding?arXiv preprint arXiv:2404.05955,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24112","last_updated":"2026-07-27T07:54:08Z","snapshot_observed_at":"2026-08-06T08:03:01.329757Z","submitted_at":"2026-07-27T07:54:08Z","title":"Scaling GUI Agents with Visual State Transitions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-31T23:04:31.363636Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2607.24112"},"observation_digest":"sha256:86777b6f40acd012f5150780f8791e7ab80b37b01a460d8635ca4a79271e7c58","observation_id":"c705f643-7e56-4759-9253-c5959683b8ad","resolution":{"observed_at":"2026-07-31T23:04:31.363636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-08-04T01:39:07.725503Z","title":"Visualwebbench: How far have multimodal llms evolved in web page understanding and grounding?arXiv preprint arXiv:2404.05955, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00036","last_updated":"2026-07-21T09:38:24Z","snapshot_observed_at":"2026-08-06T09:13:04.034456Z","submitted_at":"2026-07-21T09:38:24Z","title":"XL-DocBench: Benchmarking Evidence-Grounded Extra-Long Document Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T01:39:07.725503Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2608.00036"},"observation_digest":"sha256:615d338c8e21a549cc1687940b9efa26e12e3a9e6d84a8792c51dcfe9e5213cd","observation_id":"e9beb5ea-b3ec-47a3-b7b2-d89a5d70fffd","resolution":{"observed_at":"2026-08-04T01:39:07.725503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2404.05955/citation-record","integrity":"/paper/2404.05955/integrity","json":"/paper/2404.05955/citation-record.json","paper":"/paper/2404.05955"},"outbound":[],"paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2404.05955."}