{"as_of":"2026-08-19T03:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5c5c352565dd210cc29a3e8bc668172767407975a6bcb84eeb4e216829d24952","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:10:11.229978Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-14T10:38:39.034222Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"1908.10747","last_updated":"2019-08-28T14:29:13Z","snapshot_observed_at":"2026-08-17T14:09:01.254765Z","submitted_at":"2019-08-28T14:29:13Z","title":"Language Tasks and Language Games: On Methodology in Current Natural Language Processing Research","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-14T10:38:39.034222Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/1908.10747"},"observation_digest":"sha256:cbce739958276afebed0fffedebcb7fa45e67e318f6e60af82addf13f6c3a6e1","observation_id":"c44e8270-1ad1-49bd-a8b2-0d0bb52c55eb","resolution":{"observed_at":"2026-08-14T10:38:39.034222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2406.11354","last_updated":"2026-04-23T15:54:03Z","snapshot_observed_at":"2026-08-17T20:38:48.003295Z","submitted_at":"2024-06-17T09:17:40Z","title":"Preserving Knowledge in Large Language Model with Model-Agnostic Self-Decompression","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-24T00:09:52.093810Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2406.11354"},"observation_digest":"sha256:d53ffe7f2f14c43b68ef5e8a716ed30dd0331341b049c99d84967a7f07b8c838","observation_id":"8f07b351-18dd-4c1d-ab18-1cc9f23a84a7","resolution":{"observed_at":"2026-05-24T00:13:39.503725Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-11T12:04:52.066207Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.14672","last_updated":"2025-03-19T12:04:30Z","snapshot_observed_at":"2026-08-14T22:55:37.006938Z","submitted_at":"2024-12-19T09:24:10Z","title":"FiVL: A Framework for Improved Vision-Language Alignment through the Lens of Training, Evaluation and Explainability","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T12:04:52.066207Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2412.14672"},"observation_digest":"sha256:05b59aec35c88e5f57fe9db27a777c0413e82a985fc3f24dd48b199ba953cd12","observation_id":"c0666ad8-9fba-4350-963b-0d05427ba75c","resolution":{"observed_at":"2026-08-11T12:04:52.066207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-07T13:30:21.172080Z","title":"Making the V in VQA Matter: Ele- vating the Role of Image Understanding in Visual Question Answering, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.21755","last_updated":"2025-06-20T19:32:29Z","snapshot_observed_at":"2026-08-19T03:15:08.604258Z","submitted_at":"2025-05-27T20:44:44Z","title":"FRAMES-VQA: Benchmarking Fine-Tuning Robustness across Multi-Modal Shifts in Visual Question Answering","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T13:30:21.172080Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2505.21755"},"observation_digest":"sha256:c358ef76ab40297445485fb6e7aaf9779128c501119c9e9912560685894e2085","observation_id":"f59dcad5-8993-44fe-b3f2-142718843a77","resolution":{"observed_at":"2026-08-07T13:30:21.172080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-06T18:36:01.277024Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.10442","last_updated":"2025-07-10T15:26:41Z","snapshot_observed_at":"2026-08-14T17:00:14.442606Z","submitted_at":"2025-07-10T15:26:41Z","title":"Response Wide Shut? Surprising Observations in Basic Vision Language Model Capabilities","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:36:01.277024Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2507.10442"},"observation_digest":"sha256:1d4781dd403c36e2b373e1c2aa9c31c30793be7ca7b3e2bef429de16225f354a","observation_id":"58a9dfed-de02-42d1-b30e-b7a62d530626","resolution":{"observed_at":"2026-08-06T18:36:01.277024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-06T16:39:20.062426Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.15875","last_updated":"2025-07-17T09:05:34Z","snapshot_observed_at":"2026-08-14T20:52:04.127429Z","submitted_at":"2025-07-17T09:05:34Z","title":"Differential Multimodal Transformers","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T16:39:20.062426Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2507.15875"},"observation_digest":"sha256:a39bad0ae63415be02235cc772ed60d3015756cabf508a26aae7ce42626d092c","observation_id":"fa3ecfab-4878-4640-986d-eb42adafab7c","resolution":{"observed_at":"2026-08-06T16:39:20.062426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-15T17:10:11.229978Z","title":"Making the V in VQA matter: Elevating the role of image understanding in Visual Question Answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.17117","last_updated":"2026-06-27T06:19:28Z","snapshot_observed_at":"2026-08-17T14:03:11.442053Z","submitted_at":"2025-08-23T19:04:57Z","title":"PlantExpertVQA: A Visual Question Answering Dataset for Benchmarking Vision-Language Models in Plant Science","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-15T17:10:11.229978Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2508.17117"},"observation_digest":"sha256:3b3a6026b83d96b1587eefae8001e56b9600d789436d887a9dc29188d057bcf7","observation_id":"a9bdf05d-29e1-4dd4-8334-a4e19eb2017e","resolution":{"observed_at":"2026-08-15T17:10:11.229978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2604.16930","last_updated":"2026-04-18T09:28:23Z","snapshot_observed_at":"2026-08-15T09:12:04.785679Z","submitted_at":"2026-04-18T09:28:23Z","title":"CoGR-MoE: Concept-Guided Expert Routing with Consistent Selection and Flexible Reasoning for Visual Question Answering","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T07:09:48.239662Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2604.16930"},"observation_digest":"sha256:f6777ecf5d2693238a9625e0120fa470a6cfc2b650b7edd054f02e9c63e57fdb","observation_id":"59b73ab2-1193-4a1f-ab8a-02923d8feecb","resolution":{"observed_at":"2026-05-10T07:11:53.200709Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2604.18803","last_updated":"2026-04-25T21:48:41Z","snapshot_observed_at":"2026-08-12T17:43:47.112004Z","submitted_at":"2026-04-20T20:21:27Z","title":"LLM-as-Judge Framework for Evaluating Tone-Induced Hallucination in Vision-Language Models","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T05:11:01.039309Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2604.18803"},"observation_digest":"sha256:632c05d5a20b9e4d6e57c9aef3fb14127487b1b99f2cfabb52b88ec038655f77","observation_id":"25482ded-30fc-4dd8-9c94-f3f2d3eac16a","resolution":{"observed_at":"2026-05-10T05:15:50.368821Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2604.22851","last_updated":"2026-07-07T12:47:51Z","snapshot_observed_at":"2026-08-16T16:27:57.656946Z","submitted_at":"2026-04-22T07:49:02Z","title":"EgoDyn-Bench: Evaluating Ego-Motion Understanding in Vision-Centric Foundation Models for Autonomous Driving","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T00:09:18.068337Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2604.22851"},"observation_digest":"sha256:d065a7323b6eff0822e1a3dc7c5dadb6f5f462a20a2e000307afb96db5a750fc","observation_id":"25d1d983-259b-459b-bc68-c575850fe223","resolution":{"observed_at":"2026-05-10T00:19:47.174538Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2606.03569","last_updated":"2026-06-02T12:36:24Z","snapshot_observed_at":"2026-08-15T05:36:49.314849Z","submitted_at":"2026-06-02T12:36:24Z","title":"When Attention Collapses: Stage-Aware Visual Token Pruning from Structure to Semantics","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T10:51:38.455604Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2606.03569"},"observation_digest":"sha256:e2ec2f6b9d653177469fe74e73368ae210b7a925c2c03f475a708859f65e647f","observation_id":"a832c5d0-bdd4-47c6-94f1-2f74dcdc0bfa","resolution":{"observed_at":"2026-07-02T02:36:26.509150Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2606.20643","last_updated":"2026-06-05T21:06:39Z","snapshot_observed_at":"2026-08-14T14:02:25.540335Z","submitted_at":"2026-06-05T21:06:39Z","title":"SPARC: A Multi-Agent System for Electrical Circuit Question Answering","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T21:44:42.020128Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2606.20643"},"observation_digest":"sha256:41a00a3472729d109bf006c7aed70a0c8fa192215eb31e0bb83a363d187b107a","observation_id":"6cc0774c-d953-4fd2-a1f2-000112688b05","resolution":{"observed_at":"2026-07-02T18:57:17.102609Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2606.26587","last_updated":"2026-06-25T04:19:04Z","snapshot_observed_at":"2026-08-17T15:55:33.333568Z","submitted_at":"2026-06-25T04:19:04Z","title":"SharQ: Bridging Activation Sparsity and FP4 Quantization for LLM Inference","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-26T05:41:39.052865Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2606.26587"},"observation_digest":"sha256:b49d6e025b48546b70648a3ccf3455fefe3bd25be289a82b9e9c4ec00c0793f4","observation_id":"fb7a23d7-1e3a-44c9-af28-11d6a880f651","resolution":{"observed_at":"2026-07-04T12:59:52.416002Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":"1612.00837","doi":"10.48550/arxiv.1612.00837","metadata_source":"pith","pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","venue":"cs.CV","work_id":"5224ce0f-65a2-404a-8b9d-27604f17acb9","year":2016},"citing_paper":{"arxiv_id":"2607.05290","last_updated":"2026-07-06T16:30:41Z","snapshot_observed_at":"2026-08-13T01:10:38.072736Z","submitted_at":"2026-07-06T16:30:41Z","title":"ChatImage: Navigating Long-Form LLM Answers through Interactive Images","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-07T19:32:20.095114Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2607.05290"},"observation_digest":"sha256:3401ed2fd96894b319d91dc9ac02eafce9bc08c3d2a467c917bdb30a569d43f6","observation_id":"a32ba699-9e1a-4bd8-9cbc-a51f9eeb0929","resolution":{"observed_at":"2026-07-07T19:34:06.385691Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-01T19:13:26.665422Z","title":"Jordan Hoffmann, Sebastian Borgeaud, Arthur Mensch, Elena Buchatskaya, Trevor Cai, Eliza Rutherford, Diego de Las Casas, Lisa Anne Hendricks, Johannes Welbl, Aidan Clark, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17052","last_updated":"2026-07-19T03:43:26Z","snapshot_observed_at":"2026-08-17T13:40:59.275943Z","submitted_at":"2026-07-19T03:43:26Z","title":"Searching for Task-Specific Vision Paths: Evolutionary Block Pruning Across Vision-Language Models","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-01T19:13:26.665422Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2607.17052"},"observation_digest":"sha256:c40a8f3b4247532a225faf9bcc9788eea984e145d1c3e6173833c94a4dc0286c","observation_id":"9a380d2e-0e92-4e99-ab94-937e0b96ea69","resolution":{"observed_at":"2026-08-01T19:13:26.665422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T16:41:28.885987Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answering, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.03580","last_updated":"2026-08-04T12:34:46Z","snapshot_observed_at":"2026-08-16T03:40:03.444006Z","submitted_at":"2026-08-04T12:34:46Z","title":"SlimVLM: Sensitivity-aware Dynamic Structured Pruning with Adaptive Visual Token Selection for Efficient Vision-Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T16:41:28.885987Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2608.03580"},"observation_digest":"sha256:39743022076e5be5a045893fdf26f6c5b4b5f1d2d16e481e77f6c9acbe92d182","observation_id":"bc3fd0ed-4c5a-4351-870e-0fd3dbf05f7f","resolution":{"observed_at":"2026-08-05T16:41:28.885987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-05T15:19:02.790931Z","title":"Making the V in VQA matter: Elevating the role of image understanding in visual question answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.03649","last_updated":"2026-08-04T13:33:11Z","snapshot_observed_at":"2026-08-08T01:06:38.251380Z","submitted_at":"2026-08-04T13:33:11Z","title":"When Do Fewer Visual Tokens Accelerate Multimodal Inference? A Break-Even Study Across Decision Locations and Hardware","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T15:19:02.790931Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2608.03649"},"observation_digest":"sha256:9c23f4c060c833737001b7d186920128cc4669a95d8f9ba61a06337a86c55086","observation_id":"12a72b7e-cfe8-4a6d-a7f5-4f4b45bf6057","resolution":{"observed_at":"2026-08-05T15:19:02.790931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-08T05:29:59.232774Z","title":"arXiv preprint arXiv:1612.00837 (2017)","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.05616","last_updated":"2026-08-06T05:29:02Z","snapshot_observed_at":"2026-08-18T14:02:52.698334Z","submitted_at":"2026-08-06T05:29:02Z","title":"TruthLens: Object Hallucination Detection via Self-Evaluating Truthfulness Scores in LVLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T05:29:59.232774Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2608.05616"},"observation_digest":"sha256:1a210e8371741e9ee3ee4d5978879f31615ba671b454bd5c1fa7b441239aa441","observation_id":"65c7b628-816a-453e-987e-e9bcfe67fc42","resolution":{"observed_at":"2026-08-08T05:29:59.232774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-14T04:31:22.208670Z","title":"https://doi.org/10.48550/ARXIV.1612.00837, https://arxiv.org/abs/ 1612.008373","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.08727","last_updated":"2026-08-09T14:20:49Z","snapshot_observed_at":"2026-08-14T17:47:13.254752Z","submitted_at":"2026-08-09T14:20:49Z","title":"TomaMMU: A Comprehensive Multimodal Understanding Benchmark for Tomato Leaf Diseases","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-14T04:31:22.208670Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2608.08727"},"observation_digest":"sha256:e0b228ec43ec3cbb6995c1f3efc819cd55d0829d2173d096d150802d9b613a39","observation_id":"794518a2-b532-4a00-8c5c-7777234decef","resolution":{"observed_at":"2026-08-14T04:31:22.208670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00837","snapshot_observed_at":"2026-08-14T12:25:20.428086Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.13385","last_updated":"2026-08-13T15:46:50Z","snapshot_observed_at":"2026-08-18T09:55:49.269753Z","submitted_at":"2026-08-13T15:46:50Z","title":"When Is a Task Vector Enough? An Empirical Theory of Implicit Multimodal ICL","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-14T12:25:20.428086Z"},"links":{"cited_paper":"/paper/1612.00837","citing_paper":"/paper/2608.13385"},"observation_digest":"sha256:e1579f994ee9034ec2f25f3736763d67751badd6865e8fc778e1073b30d976a9","observation_id":"bbaafca0-6a2d-4307-b0b9-f085588084db","resolution":{"observed_at":"2026-08-14T12:25:20.428086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1612.00837/citation-record","integrity":"/paper/1612.00837/integrity","json":"/paper/1612.00837/citation-record.json","paper":"/paper/1612.00837"},"outbound":[],"paper":{"arxiv_id":"1612.00837","last_updated":"2017-05-15T17:58:49Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T08:20:05.087753Z","submitted_at":"2016-12-02T20:57:07Z","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:1612.00837."}