{"as_of":"2026-08-12T16:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8743a9dd4ee481a6ba1756c454c93b09fe0ad978fe609629111ded80e08394d8","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T05:01:12.829064Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:35:01.366536Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:18:57.044501Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00832","snapshot_observed_at":"2026-08-07T05:35:01.366536Z","title":"arXiv preprint arXiv:2412.00832 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07627","last_updated":"2025-06-09T10:45:35Z","snapshot_observed_at":"2026-08-07T06:32:54.058652Z","submitted_at":"2025-06-09T10:45:35Z","title":"Event-Priori-Based Vision-Language Model for Efficient Visual Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:35:01.366536Z"},"links":{"cited_paper":"/paper/2412.00832","citing_paper":"/paper/2506.07627"},"observation_digest":"sha256:5da7bb1fa618754ae84e1ca91ac6ea1d7830012e4af0af848a03d3f324eb4d41","observation_id":"e341bfd4-a294-4028-a135-048efe39d61c","resolution":{"observed_at":"2026-08-07T05:35:01.366536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"cited_work":{"arxiv_id":"2412.00832","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.00832","snapshot_observed_at":"2026-07-03T20:18:57.044501Z","title":"Event- GPT: Event stream understanding with multimodal large lan- guage models.arXiv preprint arXiv:2412.00832, 2024","venue":null,"work_id":"9cf69835-7008-4f64-a531-c1c8dd315ad6","year":2024},"citing_paper":{"arxiv_id":"2606.18242","last_updated":"2026-06-16T17:58:40Z","snapshot_observed_at":"2026-08-02T10:04:20.980054Z","submitted_at":"2026-06-16T17:58:40Z","title":"EventDrive: Event Cameras for Vision-Language Driving Intelligence","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-27T01:26:43.752335Z"},"links":{"cited_paper":"/paper/2412.00832","citing_paper":"/paper/2606.18242"},"observation_digest":"sha256:c5ef6e8b892fb0bcc1b3d92ff961544068e0916956b4f85855901885f09de41f","observation_id":"cee59036-540e-4c27-8a3f-e3af7a893453","resolution":{"observed_at":"2026-07-03T20:18:57.048104Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"cited_work":{"arxiv_id":"2412.00832","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.00832","snapshot_observed_at":"2026-07-03T20:18:57.044501Z","title":"Event- GPT: Event stream understanding with multimodal large lan- guage models.arXiv preprint arXiv:2412.00832, 2024","venue":null,"work_id":"9cf69835-7008-4f64-a531-c1c8dd315ad6","year":2024},"citing_paper":{"arxiv_id":"2606.31654","last_updated":"2026-07-02T11:38:15Z","snapshot_observed_at":"2026-07-07T00:05:22.503928Z","submitted_at":"2026-06-30T13:33:48Z","title":"DynFly: Dynamic-Aware Continuous Trajectory Generation for UAV Vision-Language Navigation in Urban Environments","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-01T05:35:32.213470Z"},"links":{"cited_paper":"/paper/2412.00832","citing_paper":"/paper/2606.31654"},"observation_digest":"sha256:3f69439529f249b3d270b223ff5ac7730e0c9d7d4487fc979a7c3cfac347bbf7","observation_id":"7ec6b07c-ae39-4d48-b2f3-a86ee3c2bfd6","resolution":{"observed_at":"2026-07-01T10:25:41.246070Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.00832/citation-record","integrity":"/paper/2412.00832/integrity","json":"/paper/2412.00832/citation-record.json","paper":"/paper/2412.00832"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.365328Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"d0b8de2c-353a-47a7-87a6-0ec5b4a0cfe5","year":2022},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.146077Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:4f131dfaef921abec252cfabc35eefe12e8f50d3af8ca7d30c5be88a6f4d3aaa","observation_id":"f27a5cf4-8386-4842-b18b-319ef0b3c0ea","resolution":{"observed_at":"2026-08-12T05:01:15.376332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:12.191004Z","title":"The (r) evolution of multi- modal large language models: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.191004Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:9fbd54829a234143e44c2cc3db5b861de2193e55a2561fc867b5988e64057cd5","observation_id":"cda93b88-d0ba-4f75-8fc4-b88416fabf8d","resolution":{"observed_at":"2026-08-12T05:01:12.191004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13380","last_updated":"2023-06-23T09:02:25Z","snapshot_observed_at":"2026-08-12T06:52:31.417664Z","submitted_at":"2023-06-23T09:02:25Z","title":"First Place Solution to the CVPR'2023 AQTC Challenge: A Function-Interaction Centric Approach with Spatiotemporal Visual-Language Alignment","version":1},"cited_work":{"arxiv_id":"2306.13380","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.13380","snapshot_observed_at":"2026-08-12T05:01:14.023494Z","title":"First Place Solution to the CVPR'2023 AQTC Challenge: A Function-Interaction Centric Approach with Spatiotemporal Visual-Language Alignment","venue":"cs.CV","work_id":"c6608865-f0ea-4ab8-b8e4-1ea7169b50d3","year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.213248Z"},"links":{"cited_paper":"/paper/2306.13380","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:1cc448be53ca52b7c28f729b8b3121aa7bf1ae6ccbb67a2ad90aeea080e4ae1a","observation_id":"877ca8af-dd48-45b6-a512-3ebb13ac7acc","resolution":{"observed_at":"2026-08-12T05:01:14.030175Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.334185Z","title":"Dress: Instructing large vision-language models to align and interact with humans via natural lan- guage feedback","venue":null,"work_id":"ed5cdcd9-6ea8-4ed9-b94c-2a681db50f4c","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.221664Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:19e70758a66a9dd6e09f214ec03af3afecc02f4537205a3ad1793e33c7d1e471","observation_id":"d4b315fa-d1df-4ebc-a4fb-60c6b9f3e872","resolution":{"observed_at":"2026-08-12T05:01:15.345875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.295058Z","title":"Internvl: Scaling up vision foundation mod- els and aligning for generic visual-linguistic tasks","venue":null,"work_id":"ee46f12d-a868-46e6-a0de-5de76aad8332","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.233969Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:6f83ae3dce9a471abed14971bee7cb95de9273be8c508000b7a729203e9becab","observation_id":"3f724c32-951e-477f-9df4-0600bbbe027b","resolution":{"observed_at":"2026-08-12T05:01:15.311023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.262256Z","title":"Reproducible scal- ing laws for contrastive language-image learning","venue":null,"work_id":"5987a98e-3cec-4420-bcc0-a0103b953af0","year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.246718Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:f1dcd8e552252fd4e54c1255dc7fbf89d7bf871341f95324c35615dd7bf3ed41","observation_id":"2bc6fd32-242c-49f5-a715-bb0e4893d8a9","resolution":{"observed_at":"2026-08-12T05:01:15.275105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.215917Z","title":"Event-based vision: A survey","venue":null,"work_id":"a9952b51-816b-4e53-8f28-8b688153e8a2","year":null},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.264858Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:4b6c2c8024f148b0997527b2adc3f4cb038597df9b6a56cd892e8976710f59fe","observation_id":"05333c3c-b55a-4dcb-91f5-368cdc6e3118","resolution":{"observed_at":"2026-08-12T05:01:15.224190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:12.272493Z","title":"Low-latency auto- motive vision with event cameras","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.272493Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:0e62e5373638404b4b70442bd35b444cd82db6ec93b3769872b713ba2a9934da","observation_id":"3a8da49b-83ad-49ee-bb36-20e4c81e0fea","resolution":{"observed_at":"2026-08-12T05:01:12.272493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.123463Z","title":"Eklt: Asynchronous photometric feature tracking using events and frames","venue":null,"work_id":"f7e70d1c-d57a-4b4e-bded-6c5178807f0a","year":null},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.299020Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:199d65e326f2831e6def0fbefbbb0f37d4674b41841dd56e1622ed474eea9ad3","observation_id":"dddee9e6-87e8-4149-bb7e-7306ef5af691","resolution":{"observed_at":"2026-08-12T05:01:15.147226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.087338Z","title":"Recurrent vision transformers for object detection with event cameras","venue":null,"work_id":"50690d4b-2e22-41eb-9d7d-750c0e4141d1","year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.307967Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:af278386455c4120511e248461bb9ddc5ada78761baf6714974d2e107cee1457","observation_id":"6c17c68a-1f27-4629-90f8-718b53456217","resolution":{"observed_at":"2026-08-12T05:01:15.094373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.042736Z","title":"Dsec: A stereo event camera dataset for driving scenarios","venue":null,"work_id":"94396049-8b23-410f-aa1c-5e2ceccf8b51","year":2021},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.319807Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:1ca14840189de13c95a6c6154c7fdb69039b605f34ce683a91fcaabab384bf3c","observation_id":"fe9d2bdf-22a2-4961-a51e-a9d697ac4c09","resolution":{"observed_at":"2026-08-12T05:01:15.058056Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09793","last_updated":"2024-03-22T10:36:32Z","snapshot_observed_at":"2026-07-06T15:17:33.651803Z","submitted_at":"2023-04-19T16:21:14Z","title":"Event-based Simultaneous Localization and Mapping: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09793","snapshot_observed_at":"2026-08-12T05:01:12.325692Z","title":"Event-based simultaneous localization and mapping: A com- prehensive survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.325692Z"},"links":{"cited_paper":"/paper/2304.09793","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:233a8eddc5d4389db391c86505d1e8a5e3ef23d395a5a6bf9bf765b34301adda","observation_id":"d9220487-9589-44c5-9e3a-af4cfab38a60","resolution":{"observed_at":"2026-08-12T05:01:12.325692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04594","last_updated":"2024-12-19T11:04:20Z","snapshot_observed_at":"2026-08-11T17:49:38.542327Z","submitted_at":"2024-08-08T17:10:16Z","title":"Img-Diff: Contrastive Data Synthesis for Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04594","snapshot_observed_at":"2026-08-12T05:01:12.370094Z","title":"Img-diff: Contrastive data synthesis for multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.370094Z"},"links":{"cited_paper":"/paper/2408.04594","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:42e978db1689f84656e85b443d8c2144bd7419a2aba10e6df38d822d23170ea7","observation_id":"d87a4c97-c318-4cce-b567-56c9cff7cff2","resolution":{"observed_at":"2026-08-12T05:01:12.370094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:15.000044Z","title":"Real-time 3d reconstruction and 6-dof tracking with an event camera","venue":null,"work_id":"993d4088-73a2-4fe2-b2f3-1c2c31c2306e","year":2016},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.380301Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:535efb62c60e58a591a1a10d3c294dc39a4e1129e9c80c457a9182e5fdfa9952","observation_id":"8696cb31-8981-4de7-8ad8-0e9c6db56ac8","resolution":{"observed_at":"2026-08-12T05:01:15.013695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.954081Z","title":"N-imagenet: Towards robust, fine-grained object recognition with event cameras","venue":null,"work_id":"c350a8c3-f81d-47c7-9c6a-494b662156c9","year":2021},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.389515Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:ac9aa9db3b8639127244395570796eefe07a079b69e009720ead075c9d0920fc","observation_id":"9a76f3d8-4de0-4523-a974-9f71ad70a25b","resolution":{"observed_at":"2026-08-12T05:01:14.969940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.911046Z","title":"Sodformer: Streaming object detection with transformer using events and frames","venue":null,"work_id":"8d07e81d-f5f8-4793-bfcb-a1b624de737c","year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.407322Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:ff6847c9b25f4eb3d5cb29d1cedd8017986b14625ff49d6816bff5a6552a940c","observation_id":"164a4524-4963-4a2c-ac7e-28b526a83323","resolution":{"observed_at":"2026-08-12T05:01:14.921551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:12.419221Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.419221Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:4c2a3c8ed8954db158a25e3270f82dc495f6372341ddee388389d148d7fef199","observation_id":"2d517bce-5fa1-43e7-9785-691d9a70c3d4","resolution":{"observed_at":"2026-08-12T05:01:12.419221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.849131Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"aa45659f-103c-4c03-9b71-ee9670cc5469","year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.432463Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:8135d3ad73dddf543745a6db04621108ea280eb52ef8b317539089bfb1c3c4cc","observation_id":"33a74984-41e2-4f56-931e-d1f364193f75","resolution":{"observed_at":"2026-08-12T05:01:14.861678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.805471Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":"e3c2e9f2-20eb-4bb1-8ac3-2f6f29524c50","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.451299Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:92b4be568400100f495ac51da30fb304f96640c2124972c6d3a91fdaa72e2ac6","observation_id":"8f8b143d-538b-4088-bcd4-f14b415b7773","resolution":{"observed_at":"2026-08-12T05:01:14.815717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:12.463039Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.463039Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:669e2cc6036c2fbdd949c53521f0cd82f4cde2010eced588fff979d662213229","observation_id":"f159e167-0cad-49fc-9fc7-b59820cb659b","resolution":{"observed_at":"2026-08-12T05:01:12.463039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.732695Z","title":"Visual instruction tuning","venue":null,"work_id":"6a2e71b0-cd2e-4e02-bf33-b59b7b928d0b","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.471738Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:73b8fe9ee81778e568613f7746e93419f8a38fc7e8477c7ec7786dd78c94c412","observation_id":"de61ff03-66d6-4813-9971-792a65dd3047","resolution":{"observed_at":"2026-08-12T05:01:14.748627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-12T05:01:12.482716Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.482716Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:1a93d9d3ce75eba2b08aa91d8e3b4aae74bd8c7916cfc2c64bb53e9cf36f431d","observation_id":"ae5221da-37bc-420b-875f-228503fba171","resolution":{"observed_at":"2026-08-12T05:01:12.482716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05525","last_updated":"2024-03-11T16:47:41Z","snapshot_observed_at":"2026-08-11T01:38:59.827005Z","submitted_at":"2024-03-08T18:46:00Z","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05525","snapshot_observed_at":"2026-08-12T05:01:12.494304Z","title":"Deepseek-vl: towards real-world vision- language understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.494304Z"},"links":{"cited_paper":"/paper/2403.05525","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:48daa8fbc7e6abcd598a6d7f59edd689a93c5318952037c0f2362e8f501d08dd","observation_id":"79b9cb82-abac-45af-85a3-abd0e06f9dd0","resolution":{"observed_at":"2026-08-12T05:01:12.494304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-12T05:01:12.504000Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.504000Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:0a7a49e160eb1a62d77ff6895f756d353ad6024f9dd6adf8ad7db2bae217fa18","observation_id":"33e8d8d2-9aad-4e1b-8346-d4062f9c4626","resolution":{"observed_at":"2026-08-12T05:01:12.504000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.701020Z","title":"Data-driven feature tracking for event cameras","venue":null,"work_id":"0ee9e008-0138-486f-a78e-db643d7448bf","year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.516781Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:157e0d678615f29eb3296069c5d91e79481f7a65a7e762f3613d0fa9bfd7ce95","observation_id":"1e1285f9-2f14-4f3a-9e42-ac08ff1bbb71","resolution":{"observed_at":"2026-08-12T05:01:14.712020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.648664Z","title":"Esl: Event-based structured light","venue":null,"work_id":"947ed4b3-8248-4fba-8775-50f286b62a43","year":2021},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.526710Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:1c0c8799b9fa8f13a0209496eccad72093eada0d91143f782c1fb1affbe3a934","observation_id":"36c26336-66e5-4419-a6ec-152428026790","resolution":{"observed_at":"2026-08-12T05:01:14.660582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10549","last_updated":"2023-07-04T13:18:29Z","snapshot_observed_at":"2026-08-11T19:45:35.325837Z","submitted_at":"2022-12-20T18:53:14Z","title":"Cross-modal Attention Congruence Regularization for Vision-Language Relation Alignment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10549","snapshot_observed_at":"2026-08-12T05:01:12.536142Z","title":"Cross-modal attention congruence regularization for vision-language relation align- ment","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.536142Z"},"links":{"cited_paper":"/paper/2212.10549","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:cbbb51e8ce15da3c7de1a89fc28876bf253d56882b4e6c58586f0e9f44d83cf4","observation_id":"159208c6-faaf-4660-80bb-fd0348074fdb","resolution":{"observed_at":"2026-08-12T05:01:12.536142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:12.546821Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.546821Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:a5e53c586496ddbd5675e6e3293770c47e82080743068d44a8afda21c4f6554d","observation_id":"f13b0614-776c-4379-9b42-f765ae47f85d","resolution":{"observed_at":"2026-08-12T05:01:12.546821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.571552Z","title":"Emvs: Event-based multi-view stereo—3d 9 reconstruction with an event camera in real-time","venue":null,"work_id":"c46933f6-ef48-4894-9e87-51b984604fa2","year":2018},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.559931Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:5a8b97b4818a3987631c706848415324c0fab6accad76c523ab8fbb7219b441f","observation_id":"968532f8-6901-4f09-a2a3-5980694820d2","resolution":{"observed_at":"2026-08-12T05:01:14.580424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.525450Z","title":"Events-to-video: Bringing modern computer vision to event cameras","venue":null,"work_id":"f8bc7702-38be-4c95-8edc-9fbab1ca17e6","year":2019},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.576078Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:158b3dbc5d674001c252f10d5f0e83e8250ed0b3fa055bc05d696709829c2b7d","observation_id":"53c8eda1-4a95-47ef-957c-7d272039650d","resolution":{"observed_at":"2026-08-12T05:01:14.535641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14159","last_updated":"2024-01-25T13:12:09Z","snapshot_observed_at":"2026-07-06T17:20:25.138890Z","submitted_at":"2024-01-25T13:12:09Z","title":"Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14159","snapshot_observed_at":"2026-08-12T05:01:12.585257Z","title":"Grounded sam: Assembling open-world models for diverse visual tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.585257Z"},"links":{"cited_paper":"/paper/2401.14159","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:1f4ab7bafd7af4f9835658353333cf3fe8b859d8de32bc58b9a075a2a68f986f","observation_id":"6528a202-e7dc-44f5-9e4f-6df6a237dd1f","resolution":{"observed_at":"2026-08-12T05:01:12.585257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.493868Z","title":"Aligning and prompting everything all at once for univer- sal visual perception","venue":null,"work_id":"a416f9fd-a666-47c8-9e0d-5ba35fcc411d","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.592175Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:dd6d2958670b4be803f0a71349a1c8c498f43b249caad239daefbbe55591ab6f","observation_id":"bf38fdca-0f41-4b83-94e2-aa4adf5b7bc6","resolution":{"observed_at":"2026-08-12T05:01:14.504155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17981","last_updated":"2025-07-31T18:26:09Z","snapshot_observed_at":"2026-08-05T15:13:52.977828Z","submitted_at":"2024-09-26T15:54:18Z","title":"BlinkTrack: Feature Tracking over 80 FPS via Events and Images","version":2},"cited_work":{"arxiv_id":"2409.17981","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.17981","snapshot_observed_at":"2026-08-12T05:01:13.597974Z","title":"BlinkTrack: Feature Tracking over 80 FPS via Events and Images","venue":"cs.CV","work_id":"0dcf9818-fc9d-435f-907a-aa31811619ba","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.601049Z"},"links":{"cited_paper":"/paper/2409.17981","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:11e118fbbf3101b9b5275e840c61c978a225eb3374203ed7c378ad2c31e089a2","observation_id":"c169dbb5-d2d0-4f4b-b4bc-83397472b7aa","resolution":{"observed_at":"2026-08-12T05:01:13.614829Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.462771Z","title":"Flava: A foundational language and vision alignment model","venue":null,"work_id":"e23410f8-82c8-4904-9541-0b1fa521db13","year":2022},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.615369Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:5c510278d5706e0f2ebb15ecef5aa1d8aedf30988313316032a98e63745b9fb4","observation_id":"4ef5e9f7-7c74-4902-9b55-dca0f2c8a767","resolution":{"observed_at":"2026-08-12T05:01:14.470618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.403071Z","title":"Cloud-device collaborative learning for multimodal large language models","venue":null,"work_id":"d7bb3757-bc30-416e-9fd2-5c58534fdec4","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.629470Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:64ca5166c26e9daba40a15a167556da2b3f83700b280110c8b4824515cb06082","observation_id":"7c054af0-cdea-4ef8-ad61-d2753fc46196","resolution":{"observed_at":"2026-08-12T05:01:14.410556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-12T05:01:12.644835Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.644835Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:374601e923a8b63e6750f75f9415c37cf4f4678369cf568b4bff8c5ecfecf2cc","observation_id":"53e4539d-0540-4ab5-8523-50f11d46a391","resolution":{"observed_at":"2026-08-12T05:01:12.644835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.05519","last_updated":"2024-06-25T05:01:09Z","snapshot_observed_at":"2026-08-09T22:04:45.837713Z","submitted_at":"2023-09-11T15:02:25Z","title":"NExT-GPT: Any-to-Any Multimodal LLM","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.05519","snapshot_observed_at":"2026-08-12T05:01:12.652501Z","title":"Next-gpt: Any-to-any multimodal llm","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.652501Z"},"links":{"cited_paper":"/paper/2309.05519","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:b5a890f3f660cb5fe0eac76f87cecbe0f0db075d5ec38e556d8fa814f1fe5b2b","observation_id":"a051b0d7-b2bb-4400-8e40-870824ed9ee4","resolution":{"observed_at":"2026-08-12T05:01:12.652501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.06354","last_updated":"2023-11-16T19:26:02Z","snapshot_observed_at":"2026-08-06T23:14:05.530072Z","submitted_at":"2023-06-10T06:05:35Z","title":"EventCLIP: Adapting CLIP for Event-based Object Recognition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.06354","snapshot_observed_at":"2026-08-12T05:01:12.664201Z","title":"Eventclip: Adapting clip for event-based object recognition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.664201Z"},"links":{"cited_paper":"/paper/2306.06354","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:b55885433ef3d009354148c2bd39b995695bf3bf408c1f716a5d573e7d1179d8","observation_id":"f09726e2-4787-4789-964c-78d9265993ee","resolution":{"observed_at":"2026-08-12T05:01:12.664201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.356036Z","title":"Leod: Label-efficient object detection for event cameras","venue":null,"work_id":"efa01a56-089c-4c6d-9db8-0cdac0e71a74","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.675218Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:c24857627a09db68b261ce652ec350267185ce44508df7a67e789463fa1f86b4","observation_id":"177c1df0-cd5d-4dfd-ac9e-bfeacf6a6b63","resolution":{"observed_at":"2026-08-12T05:01:14.377574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:12.682307Z","title":"xgen-mm (blip-3): A family of open large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.682307Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:fec6ec41f35f459066d456c4ab08378fc67b006e40a6ec05604d00ba50cac4ff","observation_id":"e795f0ce-43e0-4479-91b5-ba390758a38f","resolution":{"observed_at":"2026-08-12T05:01:12.682307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13549","snapshot_observed_at":"2026-08-12T05:01:12.701293Z","title":"A survey on multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.701293Z"},"links":{"cited_paper":"/paper/2306.13549","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:ddc199332d56a79ed9220e1200fb906419b0ba89b7ccc994fea1439e6371b8e7","observation_id":"ce77d6d9-3793-4b57-a355-011819caa0ce","resolution":{"observed_at":"2026-08-12T05:01:12.701293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.318926Z","title":"Eventps: Real-time photometric stereo using an event camera","venue":null,"work_id":"8197b226-5b45-4f02-88c3-a50c4f5a677d","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.711889Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:04501031061dc8f71e0641adac095ad5d2490cfb719a5638e528ab9ee3e1d010","observation_id":"1f35d5f7-59f1-43ca-8419-a13b448ebc7d","resolution":{"observed_at":"2026-08-12T05:01:14.326105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12226","last_updated":"2025-09-08T07:04:17Z","snapshot_observed_at":"2026-07-06T17:32:15.447061Z","submitted_at":"2024-02-19T15:33:10Z","title":"AnyGPT: Unified Multimodal LLM with Discrete Sequence Modeling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12226","snapshot_observed_at":"2026-08-12T05:01:12.725243Z","title":"Anygpt: Unified multimodal llm with dis- crete sequence modeling","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.725243Z"},"links":{"cited_paper":"/paper/2402.12226","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:91f48bf41ba95680309419c3f79e17d48489b2d200758b291ab466b472e9abc0","observation_id":"e8e3c9be-af2f-4cef-8bed-155f1dfaf6bb","resolution":{"observed_at":"2026-08-12T05:01:12.725243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13601","last_updated":"2024-05-28T05:36:23Z","snapshot_observed_at":"2026-08-10T21:32:36.140961Z","submitted_at":"2024-01-24T17:10:45Z","title":"MM-LLMs: Recent Advances in MultiModal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13601","snapshot_observed_at":"2026-08-12T05:01:12.746090Z","title":"Mm-llms: Recent ad- vances in multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.746090Z"},"links":{"cited_paper":"/paper/2401.13601","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:7a153a1ce8f1011e94c091075a128f2b12e8bbb2303cf5bca954c7fa401e601a","observation_id":"173d49c1-b66d-4209-a7f9-6e1e7aa1f175","resolution":{"observed_at":"2026-08-12T05:01:12.746090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.273913Z","title":"Spiking transform- ers for event-based single object tracking","venue":null,"work_id":"10a12de5-dba7-4852-8f6f-97df8fd9b596","year":2022},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.767968Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:c2b21b96f7485ec7f14d22ff009da09fba6865cf6590d87a748613551c9c093e","observation_id":"b418c2d4-6d9e-4efc-829c-9033f9a779fa","resolution":{"observed_at":"2026-08-12T05:01:14.289514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19389","last_updated":"2024-10-01T06:07:24Z","snapshot_observed_at":"2026-08-12T06:51:45.124782Z","submitted_at":"2024-06-27T17:59:01Z","title":"OMG-LLaVA: Bridging Image-level, Object-level, Pixel-level Reasoning and Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19389","snapshot_observed_at":"2026-08-12T05:01:12.778547Z","title":"Omg-llava: Bridging image-level, object-level, pixel-level reasoning and understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.778547Z"},"links":{"cited_paper":"/paper/2406.19389","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:5ecde7761d122e05d5f55656a3d8a214f6a08e48db61d8b72b70ef5ae922095d","observation_id":"93adb63c-5752-49c7-90b4-9f28c8c7bfb3","resolution":{"observed_at":"2026-08-12T05:01:12.778547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08890","last_updated":"2024-04-11T15:34:46Z","snapshot_observed_at":"2026-08-07T13:40:38.719385Z","submitted_at":"2023-02-17T14:19:28Z","title":"Deep Learning for Event-based Vision: A Comprehensive Survey and Benchmarks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08890","snapshot_observed_at":"2026-08-12T05:01:12.790227Z","title":"Deep learning for event-based vision: A comprehensive survey and bench- marks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.790227Z"},"links":{"cited_paper":"/paper/2302.08890","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:b32419e3fe770120e500f9afc54d8e992a99283364e9563d6bdb236f14d0f1a6","observation_id":"e8fd68e8-126b-481c-884d-b0955e5130d9","resolution":{"observed_at":"2026-08-12T05:01:12.790227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.242886Z","title":"E- clip: Towards label-efficient event-based open-world under- standing by clip","venue":null,"work_id":"ec5be58e-a00d-4b75-858d-199fe5135b8f","year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.799667Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:e23b77a82014af062748c1f281e802f706298cc360926d04bdc378961d30dfb6","observation_id":"b86f87bf-1754-44f4-86c0-e7b6f28dd19f","resolution":{"observed_at":"2026-08-12T05:01:14.250690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T05:01:14.207953Z","title":"Ex- act: Language-guided conceptual reasoning and uncertainty estimation for event-based action recognition and more","venue":null,"work_id":"24ebcdaf-c114-4dd3-a2da-cce751f3bd3b","year":2024},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.811621Z"},"links":{"citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:6e6cf921e15706fd7925c2adb402d25a9932e407fe5798979ad86713e37c0687","observation_id":"74cf5c2a-b2db-4702-b0a2-d23a41979662","resolution":{"observed_at":"2026-08-12T05:01:14.224985Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-12T05:01:12.822819Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.822819Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:445c1861e63be887fd0f0c18aed9c9f0ce476a670f907ec19e3e6682fb25a59e","observation_id":"1e43ee78-5e59-493c-99cb-f23f3c94085b","resolution":{"observed_at":"2026-08-12T05:01:12.822819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.09251","last_updated":"2023-12-14T18:59:43Z","snapshot_observed_at":"2026-08-12T12:55:00.219026Z","submitted_at":"2023-12-14T18:59:43Z","title":"VL-GPT: A Generative Pre-trained Transformer for Vision and Language Understanding and Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.09251","snapshot_observed_at":"2026-08-12T05:01:12.829064Z","title":"Vl-gpt: A generative pre-trained transformer for vision and language understanding and generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T05:01:12.829064Z"},"links":{"cited_paper":"/paper/2312.09251","citing_paper":"/paper/2412.00832"},"observation_digest":"sha256:ac938ce5e0782d7268a268af4b8fd1b2076f6f4154e712249328bff02a9d29ee","observation_id":"517844bd-825e-4b7e-bd36-135d6f994670","resolution":{"observed_at":"2026-08-12T05:01:12.829064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.00832","last_updated":"2024-12-01T14:38:40Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T15:33:28.648704Z","submitted_at":"2024-12-01T14:38:40Z","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":2,"verified_fuzzy":26},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 3 inbound Pith citation observations for arXiv:2412.00832."}