{"as_of":"2026-08-07T21:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4abfba807fdbb289a86efea19c48ec3ac87057cf942410427af71539c35eb521","coverage":[{"denominator":85,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":85,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:34:12.154340Z","state":"measured"},{"denominator":87,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":87,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T19:17:19.634242Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T18:51:15.880853Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2506.10395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10395","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pisces: An auto-regressive foundation model for image understanding and generation","venue":null,"work_id":"277fb66a-52e9-4f2d-a884-d192518d39f1","year":2025},"citing_paper":{"arxiv_id":"2506.15564","last_updated":"2025-09-22T01:24:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-18T15:39:15Z","title":"Show-o2: Improved Native Unified Multimodal Models","version":3},"reference_index":132,"source":"pdf_text","source_observed_at":"2026-05-12T18:51:15.428692Z"},"links":{"cited_paper":"/paper/2506.10395","citing_paper":"/paper/2506.15564"},"observation_digest":"sha256:07d6bcb4efd559b1666f2ab380273ff57c41f0bea70a21617aa6fb3e565ee63d","observation_id":"00f45ee6-cf2c-4edd-b665-fcf19c93116a","resolution":{"observed_at":"2026-05-12T18:51:15.884365Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.10395","snapshot_observed_at":"2026-07-11T19:17:19.634242Z","title":"arXiv preprint arXiv:2506.10395 (2025) 22 Kang et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04423","last_updated":"2026-07-07T07:27:22Z","snapshot_observed_at":"2026-08-04T06:57:07.480777Z","submitted_at":"2026-07-05T17:33:59Z","title":"Transferability Between Understanding and Generation in Unified Multimodal Models","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:19.634242Z"},"links":{"cited_paper":"/paper/2506.10395","citing_paper":"/paper/2607.04423"},"observation_digest":"sha256:24a33e5a7c8c3f622d86bb8770cd04b508508a3191de3e350ab9e0213f990d84","observation_id":"98aa2f0d-0924-40d1-a253-9fb7348f8004","resolution":{"observed_at":"2026-07-11T19:17:19.634242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.10395/citation-record","integrity":"/paper/2506.10395/integrity","json":"/paper/2506.10395/citation-record.json","paper":"/paper/2506.10395"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:33:59.234925Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T04:33:59.234925Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:1bdb61d85dabc4eac3b530c0ced335f955f6331fcf37fa7ffb2e5956dd440b4f","observation_id":"eebf5b90-1bdb-45b1-a9e5-af521b65bd51","resolution":{"observed_at":"2026-08-07T04:33:59.234925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T04:33:59.321702Z","title":"https://arxiv.org/abs/2407.21783","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T04:33:59.321702Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:c73c239801aac7cadc14281dedc95e5cf5f8dc3ffb563c72f51e39c4713ada4e","observation_id":"a020a70d-a6a7-4c98-9182-37b551e521db","resolution":{"observed_at":"2026-08-07T04:33:59.321702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.07520","last_updated":"2022-01-19T10:45:38Z","snapshot_observed_at":"2026-07-06T12:28:43.097237Z","submitted_at":"2022-01-19T10:45:38Z","title":"CM3: A Causal Masked Multimodal Model of the Internet","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.07520","snapshot_observed_at":"2026-08-07T04:33:59.440871Z","title":"CM3: A causal masked multimodal model of the internet","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T04:33:59.440871Z"},"links":{"cited_paper":"/paper/2201.07520","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:bcceffd31955f987cded378e7fc2cf490f1693d6ef3655bfce97052ab0986992","observation_id":"abc4ea03-17c9-4738-936f-47ed169a939d","resolution":{"observed_at":"2026-08-07T04:33:59.440871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:33:59.594642Z","title":"Menick, Sebastian Borgeaud, Andy Brock, Aida Nematzadeh, Sahand Sharifzadeh, Mikolaj Binkowski, Ricardo Barreira, Oriol Vinyals, Andrew Zisserman, and Kar \\' e n Simonyan","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T04:33:59.594642Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:10c14ff9c5840c6ab7e76d8ea022701d6e8c6e7bea2de32f3b38a64c5b87ac06","observation_id":"28a22b10-b9cf-4c6f-9bdd-35e81f4db077","resolution":{"observed_at":"2026-08-07T04:33:59.594642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:33:59.735365Z","title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T04:33:59.735365Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:fd055cebb918dfe615bbc1de9cd716b222bc95dbfbb02facb05992c0eff9b545","observation_id":"9c6ba7fd-8ebb-4380-a453-fdc378c12828","resolution":{"observed_at":"2026-08-07T04:33:59.735365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:00.049480Z","title":"Kwok, Ping Luo, Huchuan Lu, and Zhenguo Li","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:00.049480Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:7860c76921f3a4c2b0ed5971370320d8085df79cb51b518da8cd1bb6a9eaef33","observation_id":"85ca09f3-7f82-4bec-868e-88e0543cf97a","resolution":{"observed_at":"2026-08-07T04:34:00.049480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10733","last_updated":"2025-05-18T21:17:22Z","snapshot_observed_at":"2026-07-06T19:33:12.610676Z","submitted_at":"2024-10-14T17:15:07Z","title":"Deep Compression Autoencoder for Efficient High-Resolution Diffusion Models","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10733","snapshot_observed_at":"2026-08-07T04:34:00.161811Z","title":"Deep compression autoencoder for efficient high-resolution diffusion models, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:00.161811Z"},"links":{"cited_paper":"/paper/2410.10733","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:993bb844e6614c8eb83f1b212fcde8d8cf9ed5cee7640a75a506f32dbd2fec2f","observation_id":"0f708b75-a8e2-4530-91a4-596384414343","resolution":{"observed_at":"2026-08-07T04:34:00.161811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-04T08:17:54.774738Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12793","snapshot_observed_at":"2026-08-07T04:34:00.326020Z","title":"Sharegpt4v: Improving large multi-modal models with better captions, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:00.326020Z"},"links":{"cited_paper":"/paper/2311.12793","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:ec7ee0b2efc6c8fabd176278a971d1ab3adeba0e47ce63769238450651f9cebe","observation_id":"2af2ac08-d9d3-4a7e-823b-559c41fbd359","resolution":{"observed_at":"2026-08-07T04:34:00.326020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20330","last_updated":"2024-04-09T15:17:50Z","snapshot_observed_at":"2026-08-07T12:15:30.838846Z","submitted_at":"2024-03-29T17:59:34Z","title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20330","snapshot_observed_at":"2026-08-07T04:34:00.478226Z","title":"Are we on the right way for evaluating large vision-language models? arXiv preprint arXiv:2403.20330, 2024 c","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:00.478226Z"},"links":{"cited_paper":"/paper/2403.20330","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:33780b2e30b46fb45cd51f96bbc8970e800b25bde2693e6d7ac9c348d7ee1352","observation_id":"810f32d4-c9b2-4610-8376-0015a635568e","resolution":{"observed_at":"2026-08-07T04:34:00.478226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06500","last_updated":"2023-06-15T08:00:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-11T00:38:10Z","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06500","snapshot_observed_at":"2026-08-07T04:34:00.637220Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:00.637220Z"},"links":{"cited_paper":"/paper/2305.06500","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:3a223ed4576b47cbd504aed64cd6ba2375e70d5e983960bd9feedb5b4aa18fc1","observation_id":"fdd57776-9a51-47f4-9f03-ec46aa1e85e7","resolution":{"observed_at":"2026-08-07T04:34:00.637220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:00.829274Z","title":"Dream LLM : Synergistic multimodal comprehension and creation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:00.829274Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:fe4907038eac089f655091e91454278b2a185252ae93184a4b868ce7ad40b289","observation_id":"282b3c3f-0bf0-434e-a7b2-4d2c58bc8d31","resolution":{"observed_at":"2026-08-07T04:34:00.829274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:00.945047Z","title":"Taming transformers for high-resolution image synthesis","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:00.945047Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:ee5eb41c359aa26ddfd01e38dd08dfef3eba0eb1a28ac9887da7e006cff3432a","observation_id":"31c365d3-95ab-449f-a0bc-836c440bcaa2","resolution":{"observed_at":"2026-08-07T04:34:00.945047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:18.728605Z","title":"Scaling rectified flow transformers for high-resolution image synthesis","venue":null,"work_id":"cc5b9adf-c5da-4cf4-a935-0e619882ac2e","year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:01.070818Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:83b038a4a60f733be05aef053149122386f97613407f4ee8054ca21066b7e02f","observation_id":"0a84b836-d5e9-4f8a-b917-d0214a4a70ce","resolution":{"observed_at":"2026-08-07T04:34:18.805912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:01.269001Z","title":"Mme: A comprehensive evaluation benchmark for multimodal large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:01.269001Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:135b3bb04657f3f2ef7a2d97d979fdffecfa3c43c2fc5c0a2065b7192a325a1d","observation_id":"005b0f3d-381c-4faa-8df6-bd1e16f6c3e6","resolution":{"observed_at":"2026-08-07T04:34:01.269001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14396","last_updated":"2025-03-02T07:53:44Z","snapshot_observed_at":"2026-07-06T18:03:51.687441Z","submitted_at":"2024-04-22T17:56:09Z","title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14396","snapshot_observed_at":"2026-08-07T04:34:01.423591Z","title":"Seed-x: Multimodal models with unified multi-granularity comprehension and generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:01.423591Z"},"links":{"cited_paper":"/paper/2404.14396","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:040da3f16ed08e1b602ba9bad8de2e16886e0e60320efa9c553dd3974246b23a","observation_id":"dd691fa1-fb58-40bf-b061-aef520ac4b56","resolution":{"observed_at":"2026-08-07T04:34:01.423591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:18.533791Z","title":"Geneval: An object-focused framework for evaluating text-to-image alignment","venue":null,"work_id":"a29e0517-ecde-4e9a-9137-7320c3d58d57","year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:01.595680Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:0bba0baca9e8fb27c4eaa5cb961db06451ed4e79ab7d8808d91ed9e558c1c764","observation_id":"176c3921-7018-4cdd-8b03-1bf1c652af5b","resolution":{"observed_at":"2026-08-07T04:34:18.610344Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:18.289760Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answering","venue":null,"work_id":"40865bd2-90dc-411b-9f23-902b4aa712f7","year":2017},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:01.784116Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:5654ab6f1c1015a9da804d08cec79c5c2114e2333fa78281135c093b0b09061e","observation_id":"8d0a0171-406d-4d70-be15-7df991db1f82","resolution":{"observed_at":"2026-08-07T04:34:18.408391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:18.087099Z","title":"Vizwiz grand challenge: Answering visual questions from blind people","venue":null,"work_id":"31a874ff-d4f5-4b07-b1c5-a930a2a0cb34","year":2018},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:01.983831Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:53dfd600a4a51929081e03446bb24254bf5f7def14adade496042edccc1c891c","observation_id":"4cc5fe1e-282d-46d5-9df9-8d11ba897449","resolution":{"observed_at":"2026-08-07T04:34:18.190173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.06377","last_updated":"2021-12-19T19:23:25Z","snapshot_observed_at":"2026-07-31T02:35:46.853381Z","submitted_at":"2021-11-11T18:46:40Z","title":"Masked Autoencoders Are Scalable Vision Learners","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.06377","snapshot_observed_at":"2026-08-07T04:34:02.155166Z","title":"Girshick","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:02.155166Z"},"links":{"cited_paper":"/paper/2111.06377","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:fd140f96059c4d800e3ee54cb7e7e20467c737d5b7bb01004863eedf02755295","observation_id":"3ecddb5c-dceb-48f4-9232-86434787a8e4","resolution":{"observed_at":"2026-08-07T04:34:02.155166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:17.877980Z","title":"Gqa: A new dataset for real-world visual reasoning and compositional question answering","venue":null,"work_id":"72eb5cc7-f21c-4bae-86b8-9e1c6d85fab6","year":2019},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:02.302887Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:f852107246595ddfcf59a4c6e13ac342aa854ce2b00433550439e371a1c91d91","observation_id":"29f63196-c174-40ca-bc2c-c11ce4af7382","resolution":{"observed_at":"2026-08-07T04:34:17.952933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2309.04669","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:13.131343Z","title":"Unified language-vision pretraining in LLM with dynamic discrete visual tokenization","venue":null,"work_id":"4910c218-0968-4ac8-b089-99336f9f546d","year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:02.418863Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:303e06bff759c59246dac934038f40ec7d22124465bed0027dc67f6fd81f0f2e","observation_id":"c7a3e172-d7a5-4053-9f1c-a43cb76fc897","resolution":{"observed_at":"2026-08-07T04:34:13.304814Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:17.581496Z","title":"A diagram is worth a dozen images","venue":null,"work_id":"89f264e9-7fba-4761-a044-76a27e0fff79","year":2016},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:02.539819Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:c0bb12bec0d6b3a8cbe6a660234771d24d3f137b2f045cd97aee67376d6f5cfa","observation_id":"61a80a70-4b79-4239-9a52-ac12d2f74da1","resolution":{"observed_at":"2026-08-07T04:34:17.721711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:17.421399Z","title":"Generating images with multimodal language models","venue":null,"work_id":"98273f6f-625a-4812-8dcc-289b8a919729","year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:02.755733Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:d4f7f8946043486a5f7e76cb8b62d73154b8b75cb1b713f09431de68ce08a148","observation_id":"a52519a8-677c-4275-bdfe-737ef7d2a1d5","resolution":{"observed_at":"2026-08-07T04:34:17.490002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05425","last_updated":"2023-06-08T17:59:56Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"MIMIC-IT: Multi-Modal In-Context Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05425","snapshot_observed_at":"2026-08-07T04:34:02.916132Z","title":"MIMIC-IT: multi-modal in-context instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:02.916132Z"},"links":{"cited_paper":"/paper/2306.05425","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:0e20f6d3d5092bc83eb6461c5fd989400e7ea563457ba06814f60b5eb4971a7f","observation_id":"e870ab70-d91a-408c-ad38-3c090f0f4a8a","resolution":{"observed_at":"2026-08-07T04:34:02.916132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16125","last_updated":"2023-08-02T08:02:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-30T04:25:16Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16125","snapshot_observed_at":"2026-08-07T04:34:03.070205Z","title":"Seed-bench: Benchmarking multimodal llms with generative comprehension","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:03.070205Z"},"links":{"cited_paper":"/paper/2307.16125","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:f6ec814e55aa4fca3f8785163ecd00529d186a38270534d30b041fb0eafa0968","observation_id":"79fde0c5-1a91-4e10-926f-c9d7b402a850","resolution":{"observed_at":"2026-08-07T04:34:03.070205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-07-06T18:44:24.873040Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07895","snapshot_observed_at":"2026-08-07T04:34:03.206849Z","title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:03.206849Z"},"links":{"cited_paper":"/paper/2407.07895","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:57e248561087c033a26b1af018ff5e375303d16af469493b0df5e60defbb6a13","observation_id":"12899f82-3563-44a6-9a78-941a20b1afd3","resolution":{"observed_at":"2026-08-07T04:34:03.206849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:17.159431Z","title":null,"venue":null,"work_id":"d8395f6e-f513-4a6a-a4e2-33d48f43c053","year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:03.339774Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:80e6e0e44dd5335121e1f91a6b5d02c67f797be1e552398964ca5c5d0266a67b","observation_id":"273f10d9-7205-4d87-8284-ca2e3c2ef32a","resolution":{"observed_at":"2026-08-07T04:34:17.290173Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18814","last_updated":"2024-03-27T17:59:04Z","snapshot_observed_at":"2026-07-31T05:41:28.385099Z","submitted_at":"2024-03-27T17:59:04Z","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18814","snapshot_observed_at":"2026-08-07T04:34:03.476230Z","title":"Mini-gemini: Mining the potential of multi-modality vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:03.476230Z"},"links":{"cited_paper":"/paper/2403.18814","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:1ae9d922ebe5bdff58b74794f3951b13f1a91d3688d9e8ba0faa22847eafb87b","observation_id":"9a25378b-887d-4c8f-b888-4c69caf597a7","resolution":{"observed_at":"2026-08-07T04:34:03.476230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10355","last_updated":"2023-10-26T02:52:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T16:34:01Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10355","snapshot_observed_at":"2026-08-07T04:34:03.678107Z","title":"Evaluating object hallucination in large vision-language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:03.678107Z"},"links":{"cited_paper":"/paper/2305.10355","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:40d8853301cb10088ec687e3a21929d46789b06bda5f09f38b2b35b9d4ca4387","observation_id":"e07f2fb3-48c4-46de-91da-d8bf02d63ddb","resolution":{"observed_at":"2026-08-07T04:34:03.678107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:03.799566Z","title":"Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll \\' a r, and C","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:03.799566Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:303a744874f0a57e89e790db5320a85c2b3909eb058bafae65740896eed43c35","observation_id":"c4ad44f6-9d62-4bf8-baf6-b5e6f06e999c","resolution":{"observed_at":"2026-08-07T04:34:03.799566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.14566","last_updated":"2024-03-25T06:05:24Z","snapshot_observed_at":"2026-08-02T21:20:28.501823Z","submitted_at":"2023-10-23T04:49:09Z","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.14566","snapshot_observed_at":"2026-08-07T04:34:03.967311Z","title":"Hallusionbench: You see what you think? or you think what you see? an image-context reasoning benchmark challenging for gpt-4v (ision), llava-1.5, and other multi-modality models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:03.967311Z"},"links":{"cited_paper":"/paper/2310.14566","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:09808fbbbd5ce1f1a35ddf0eb2247d23628213d737fc1c89f98a707c56883aed","observation_id":"98e74cdd-e023-4e57-82bb-a755b74a2fad","resolution":{"observed_at":"2026-08-07T04:34:03.967311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-07T04:34:04.091274Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:04.091274Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:a4b8ae93227d8595d68d6d89168efb7f2ddd8383cef73cdc09b9ae3689466d8a","observation_id":"f8a241d6-d815-4b89-b308-3c81e11cf694","resolution":{"observed_at":"2026-08-07T04:34:04.091274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08485","last_updated":"2023-12-11T17:46:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-17T17:59:25Z","title":"Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08485","snapshot_observed_at":"2026-08-07T04:34:04.316929Z","title":"Visual instruction tuning, 2023 c","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:04.316929Z"},"links":{"cited_paper":"/paper/2304.08485","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:4a8bdd5c96dc2dc78d5bf56241810ea7f7b2d3c2b587368d285d6560f011f313","observation_id":"fc9d2ee5-14e5-4b01-92b2-1e4f746c6f5e","resolution":{"observed_at":"2026-08-07T04:34:04.316929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:17.008362Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, January 2024 a","venue":null,"work_id":"8cfc62fc-9ad4-4026-9dad-03f90a355bc0","year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:04.448654Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:88c5e5a629a98a8ad84be7cc3073f77b6a617fd97e7c6fa7cf58caf61cacd4da","observation_id":"aa3dcf34-721e-4f12-8389-df32b9d9580c","resolution":{"observed_at":"2026-08-07T04:34:17.057930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:04.568020Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:04.568020Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:588f8ff7e8b9f0c993a768422bfbcd2df3d8883eb6dd86d567a1078ee9aefe99","observation_id":"ce5fc821-db12-452d-8655-ac7a47e1b085","resolution":{"observed_at":"2026-08-07T04:34:04.568020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.06281","last_updated":"2024-08-20T03:56:03Z","snapshot_observed_at":"2026-07-06T15:53:19.485466Z","submitted_at":"2023-07-12T16:23:09Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.06281","snapshot_observed_at":"2026-08-07T04:34:04.747893Z","title":"Mmbench: Is your multi-modal model an all-around player? arXiv preprint arXiv:2307.06281, 2023 d","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:04.747893Z"},"links":{"cited_paper":"/paper/2307.06281","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:ed20bb15b9a50e04ae1229f8f96d513abd7a16ba989fb9776372d2deaded08fc","observation_id":"396b95b1-89d1-4a06-9d74-74649dc13acf","resolution":{"observed_at":"2026-08-07T04:34:04.747893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:16.909766Z","title":"On the hidden mystery of ocr in large multimodal models, 2024 c","venue":null,"work_id":"d28c03cf-6644-4897-abcb-b3be89ade0c2","year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:04.934626Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:6b0a35e8fdc8c326fee99da2e7df013406e8e424a9fe73a929be699209399b4b","observation_id":"e6ea8826-9138-48fc-b20d-1235829373dc","resolution":{"observed_at":"2026-08-07T04:34:16.931157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:05.044523Z","title":"Learn to explain: Multimodal reasoning via thought chains for science question answering","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:05.044523Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:9bd453b2db14bc29b7d2a863167debec777dfcf73fdc3d12742e4a38c38974ca","observation_id":"74ab8d04-505b-4eb6-a6e8-66c7fc7cfcfe","resolution":{"observed_at":"2026-08-07T04:34:05.044523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02255","last_updated":"2024-01-21T03:47:06Z","snapshot_observed_at":"2026-07-06T16:27:15.027202Z","submitted_at":"2023-10-03T17:57:24Z","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02255","snapshot_observed_at":"2026-08-07T04:34:05.176742Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:05.176742Z"},"links":{"cited_paper":"/paper/2310.02255","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:ccb0106eed0ad2b6916f4944aa1ce3cbb607f677ccb1e5c756bb1041f65d51c7","observation_id":"b37d7701-e2da-4f1e-88e1-4496d035711c","resolution":{"observed_at":"2026-08-07T04:34:05.176742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09093","last_updated":"2023-06-15T12:45:25Z","snapshot_observed_at":"2026-07-06T15:42:56.285045Z","submitted_at":"2023-06-15T12:45:25Z","title":"Macaw-LLM: Multi-Modal Language Modeling with Image, Audio, Video, and Text Integration","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.09093","snapshot_observed_at":"2026-08-07T04:34:05.315938Z","title":"Macaw-llm: Multi-modal language modeling with image, audio, video, and text integration","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:05.315938Z"},"links":{"cited_paper":"/paper/2306.09093","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:43e54bcb214100ee745dc2596658b4ace581434bc2142fd6132b74430c25d40d","observation_id":"87ae0daa-a108-4529-bfcc-022c9b0aa4aa","resolution":{"observed_at":"2026-08-07T04:34:05.315938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:05.474321Z","title":"Docvqa: A dataset for vqa on document images","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:05.474321Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:458740564e0fbaa82926c191ad2fc85849644f1728d7a99e7e065212d63224cb","observation_id":"f1179f3e-2376-42ea-8792-0760c0b39fc7","resolution":{"observed_at":"2026-08-07T04:34:05.474321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:05.634106Z","title":"Infographicvqa","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:05.634106Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:371d83a1cb3f3d87f3e1162d7fb4563325b618a414247fa06e7bb3b0a9cb0a76","observation_id":"f0ddd885-f659-40fb-a918-8f9af08c3c99","resolution":{"observed_at":"2026-08-07T04:34:05.634106Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.01952","last_updated":"2023-07-04T23:04:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-04T23:04:57Z","title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.01952","snapshot_observed_at":"2026-08-07T04:34:05.794775Z","title":"SDXL: improving latent diffusion models for high-resolution image synthesis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:05.794775Z"},"links":{"cited_paper":"/paper/2307.01952","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:4bfd3c7a6d535a0565b5eccdd5b8fddf469726ecf2de50ce188101fb13c59fa3","observation_id":"2838699e-86a7-441a-b0b1-7da3caea2a41","resolution":{"observed_at":"2026-08-07T04:34:05.794775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:16.593942Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"3c666df9-8d55-4e20-8d6b-dae6a2ab2ae8","year":2021},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:05.937934Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:cb38a6230267859f7ef2cd75d08899a6ae052253116139945febf506c076018c","observation_id":"c74be96a-30ec-4c99-aca8-4fa8592c5a13","resolution":{"observed_at":"2026-08-07T04:34:16.730462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:16.317030Z","title":null,"venue":null,"work_id":"42bd2a8e-ee72-4f6a-a2a1-27145d8fef4a","year":2020},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:06.067878Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:3203baeb6d28f7869b35e94938ff4500bf272fa8ab9f48869533e4ae0bccedd0","observation_id":"20385547-d54e-4e6f-ad8d-2dfec6e24874","resolution":{"observed_at":"2026-08-07T04:34:16.439186Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.06125","last_updated":"2022-04-13T01:10:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-13T01:10:33Z","title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.06125","snapshot_observed_at":"2026-08-07T04:34:06.226877Z","title":"Hierarchical text-conditional image generation with CLIP latents","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:06.226877Z"},"links":{"cited_paper":"/paper/2204.06125","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:d726fd61be588252c1d723e590673a78134c642a5f74c531fc6b7be9942107b4","observation_id":"04974964-d317-4823-be2f-d05af2983131","resolution":{"observed_at":"2026-08-07T04:34:06.226877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:06.371337Z","title":"High-resolution image synthesis with latent diffusion models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:06.371337Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:c9043f4698a134d58e51a036746a45938a4b6bf74d7362110e808d4ba831a594","observation_id":"607662ee-0c37-4bf5-a7bb-6f381a55fe69","resolution":{"observed_at":"2026-08-07T04:34:06.371337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:06.504245Z","title":"High-resolution image synthesis with latent diffusion models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:06.504245Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:eef878ccc416ea0b6cc78c4bb6efc90688a526409ae35ed131b06418dec4b331","observation_id":"143f32cf-f12e-4123-b98b-f3ab38e3a9b0","resolution":{"observed_at":"2026-08-07T04:34:06.504245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15896","last_updated":"2024-12-06T00:41:15Z","snapshot_observed_at":"2026-07-06T17:34:57.509162Z","submitted_at":"2024-02-24T20:15:31Z","title":"Multimodal Instruction Tuning with Conditional Mixture of LoRA","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15896","snapshot_observed_at":"2026-08-07T04:34:06.624192Z","title":"Multimodal instruction tuning with conditional mixture of lora","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:06.624192Z"},"links":{"cited_paper":"/paper/2402.15896","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:09db4e7e04956d2be7b112f0e96af9de96157b6ecb1bb339ca970a1bcc562d4c","observation_id":"1ee84ecc-3961-4ef5-a0ae-76d81a0a4246","resolution":{"observed_at":"2026-08-07T04:34:06.624192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:16.025165Z","title":"Towards vqa models that can read","venue":null,"work_id":"84833f42-49f4-4469-a3b2-349abda0b4f7","year":2019},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:06.792015Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:46fd10d9a456c191ebd36de83fe4995ed4b20c9de8a5e422e6664a292016603d","observation_id":"ec43e49d-4005-475c-9c85-60df80513c47","resolution":{"observed_at":"2026-08-07T04:34:16.178169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06525","last_updated":"2024-06-10T17:59:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-10T17:59:52Z","title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06525","snapshot_observed_at":"2026-08-07T04:34:06.956907Z","title":"Autoregressive model beats diffusion: Llama for scalable image generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:06.956907Z"},"links":{"cited_paper":"/paper/2406.06525","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:2d571765cb299a17b8804a4efd1b56b0e50cb99eaeda160be57cd39fbd4e6b0f","observation_id":"b455ec78-c8a9-44bc-9ca5-25d79365c58f","resolution":{"observed_at":"2026-08-07T04:34:06.956907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15389","last_updated":"2023-03-27T17:02:21Z","snapshot_observed_at":"2026-07-06T15:08:34.018146Z","submitted_at":"2023-03-27T17:02:21Z","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15389","snapshot_observed_at":"2026-08-07T04:34:07.113869Z","title":"EVA-CLIP: improved training techniques for CLIP at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:07.113869Z"},"links":{"cited_paper":"/paper/2303.15389","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:2d5ec8b82700e17a14e34e07af7f2872227ed053c90c79a17227a34de889a82e","observation_id":"b6fb765c-2f9b-4a37-8260-721aab517faa","resolution":{"observed_at":"2026-08-07T04:34:07.113869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.05222","last_updated":"2024-05-08T02:46:43Z","snapshot_observed_at":"2026-08-06T07:46:37.713769Z","submitted_at":"2023-07-11T12:45:39Z","title":"Emu: Generative Pretraining in Multimodality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.05222","snapshot_observed_at":"2026-08-07T04:34:07.236721Z","title":"Generative pretraining in multimodality","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:07.236721Z"},"links":{"cited_paper":"/paper/2307.05222","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:6332d2de98557c2ada764ef528bdc734a38e03a0f34ac6e73b6f667731be5116","observation_id":"b107e098-ca99-4fa3-9c50-ba1aa94d0951","resolution":{"observed_at":"2026-08-07T04:34:07.236721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:15.679631Z","title":"Generative multimodal models are in-context learners","venue":null,"work_id":"e7b8464d-b6e8-4e6c-9f4a-3745646a33d1","year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:07.381855Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:a472ec2bfe821c4e04af1cf958e46e4d5a3119d0688fe4911105f00c62202167","observation_id":"9157b874-4a5e-4133-bc4e-ebce19cfbb08","resolution":{"observed_at":"2026-08-07T04:34:15.826498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18775","last_updated":"2023-11-30T18:21:25Z","snapshot_observed_at":"2026-07-06T16:55:09.267685Z","submitted_at":"2023-11-30T18:21:25Z","title":"CoDi-2: In-Context, Interleaved, and Interactive Any-to-Any Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18775","snapshot_observed_at":"2026-08-07T04:34:07.537541Z","title":"Codi-2: In-context, interleaved, and interactive any-to-any generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:07.537541Z"},"links":{"cited_paper":"/paper/2311.18775","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:8264b60d1f7b2cd5a0d4a0849b83452590b7953c361e988c66a695b6b37ae981","observation_id":"b6b0a231-7596-49d1-8b53-45a7d0f0e3ad","resolution":{"observed_at":"2026-08-07T04:34:07.537541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:15.372153Z","title":"Any-to-any generation via composable diffusion","venue":null,"work_id":"a3025744-038c-41ea-9d6d-e586b5787a2b","year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:07.678452Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:23635b41f57ee8f920e927a0a404a0d018d759bd6e7463cb93a9080057182f7c","observation_id":"ff032ef9-9d8c-4eac-9f1c-0536508ec074","resolution":{"observed_at":"2026-08-07T04:34:15.538723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:07.871098Z","title":"Chameleon: Mixed-modal early-fusion foundation models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:07.871098Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:ff5547212ebde90dd2fffb684b672270e7432cfe262923d657de554598911370","observation_id":"f3aefd29-efa1-4137-8cf6-000253965e3c","resolution":{"observed_at":"2026-08-07T04:34:07.871098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10208","last_updated":"2024-04-02T09:20:50Z","snapshot_observed_at":"2026-07-06T17:17:31.008185Z","submitted_at":"2024-01-18T18:50:16Z","title":"MM-Interleaved: Interleaved Image-Text Generative Modeling via Multi-modal Feature Synchronizer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10208","snapshot_observed_at":"2026-08-07T04:34:08.000903Z","title":"Mm-interleaved: Interleaved image-text generative modeling via multi-modal feature synchronizer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:08.000903Z"},"links":{"cited_paper":"/paper/2401.10208","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:70b2ff239e964e296ceccd1210ecf63398aa0c335c453e56143c1e51f36bde94","observation_id":"ff2de168-136a-4a9c-a1e1-8385b4d6a2fe","resolution":{"observed_at":"2026-08-07T04:34:08.000903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16860","snapshot_observed_at":"2026-08-07T04:34:08.124461Z","title":"Cambrian-1: A fully open, vision-centric exploration of multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:08.124461Z"},"links":{"cited_paper":"/paper/2406.16860","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:7be8e767b29ebbe3477ce678de35453b107aa787cecdbe5241cee56dbb18f771","observation_id":"0129f496-dcae-4c81-bc71-311c6cc4aecf","resolution":{"observed_at":"2026-08-07T04:34:08.124461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:15.067690Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":null,"work_id":"c1454c1f-e791-4ea5-8212-fb09cdf65b76","year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:08.316611Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:ce26cca688cec0c241fac1599a07f715d1006ec1d92523ec58696c8475f4142c","observation_id":"0b5b9bf6-2cc9-4da9-b2db-d3b84535404f","resolution":{"observed_at":"2026-08-07T04:34:15.221743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T04:34:08.477979Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:08.477979Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:1f8a9a5d2882210d199b169e98643bba9f06f5a61ab5b9aa80be9f8be20b15f6","observation_id":"969ed406-5d8c-4ea7-9be1-56d92ad587cb","resolution":{"observed_at":"2026-08-07T04:34:08.477979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T04:34:08.648626Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:08.648626Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:95517d510d7a4be18dc92643759762515147b8a129f2e0fcb47e83ec98bbb522","observation_id":"0ffe2784-43c0-407f-87e9-88ba41cf170c","resolution":{"observed_at":"2026-08-07T04:34:08.648626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07574","last_updated":"2023-11-29T15:37:24Z","snapshot_observed_at":"2026-08-06T03:52:00.226360Z","submitted_at":"2023-11-13T18:59:31Z","title":"To See is to Believe: Prompting GPT-4V for Better Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07574","snapshot_observed_at":"2026-08-07T04:34:08.755303Z","title":"To see is to believe: Prompting GPT-4V for better visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:08.755303Z"},"links":{"cited_paper":"/paper/2311.07574","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:75d15d448c1dc6d8cc2620c86742d91a10d5891e6130e6fef7735c06e5041313","observation_id":"7d658a2d-493f-4c3d-8b4e-c7ef1a5e2a14","resolution":{"observed_at":"2026-08-07T04:34:08.755303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:14.824732Z","title":"OFA: unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework","venue":null,"work_id":"bb11eee6-77c6-46b9-a7e9-91460f0173aa","year":2022},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:08.885112Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:939832a25558048ba1ea4bc0afdeb573c1dfa7626ea76c0313bb459eda7b7f8f","observation_id":"2219aa29-5f93-4240-afd0-11434893aa1e","resolution":{"observed_at":"2026-08-07T04:34:14.903752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:09.022182Z","title":"Image as a foreign language: BEIT pretraining for vision and vision-language tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:09.022182Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:a100e5342a23229b1dfec8e01b88a652a12d893bd281081a20f01b4939e1eeed","observation_id":"7c8c064b-8e4f-45de-9a24-3a57bbc9bddb","resolution":{"observed_at":"2026-08-07T04:34:09.022182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18869","last_updated":"2024-09-27T16:06:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T16:06:11Z","title":"Emu3: Next-Token Prediction is All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.18869","snapshot_observed_at":"2026-08-07T04:34:09.192768Z","title":"Emu3: Next-token prediction is all you need, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:09.192768Z"},"links":{"cited_paper":"/paper/2409.18869","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:7a284c8e420fac87b961ec14cc47ab291a41e6a21d4888a5a5e3618c98819d08","observation_id":"1d99262d-5b9d-4c38-be12-0575e788c3eb","resolution":{"observed_at":"2026-08-07T04:34:09.192768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13848","last_updated":"2024-10-17T17:58:37Z","snapshot_observed_at":"2026-08-03T03:16:45.693966Z","submitted_at":"2024-10-17T17:58:37Z","title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13848","snapshot_observed_at":"2026-08-07T04:34:09.374451Z","title":"Janus: Decoupling visual encoding for unified multimodal understanding and generation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:09.374451Z"},"links":{"cited_paper":"/paper/2410.13848","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:bd70e2353ddc70194b7559e58f477978a9ca84228127e6ec47da89f9ce68e045","observation_id":"974ce6fa-37c5-4c72-883f-1f57d968f342","resolution":{"observed_at":"2026-08-07T04:34:09.374451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.05519","last_updated":"2024-06-25T05:01:09Z","snapshot_observed_at":"2026-08-04T16:34:07.714734Z","submitted_at":"2023-09-11T15:02:25Z","title":"NExT-GPT: Any-to-Any Multimodal LLM","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.05519","snapshot_observed_at":"2026-08-07T04:34:09.517877Z","title":"Next-gpt: Any-to-any multimodal LLM","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:09.517877Z"},"links":{"cited_paper":"/paper/2309.05519","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:1746b346321c40b18cfe396ea9ba1885d6b4e3b7468172dddf1bae0024ab247a","observation_id":"d18c754e-cc81-41f7-95a0-bf11ed74dfb9","resolution":{"observed_at":"2026-08-07T04:34:09.517877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:14.623055Z","title":"Grok 1.5v: The next generation of ai","venue":null,"work_id":"9208a32d-ad12-498f-810c-638720bf9dd9","year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:09.703995Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:8a4790ea6b7ed352702568ba68ac0314863168f8cb979573ac097121a9b16eed","observation_id":"aa67e46c-d63e-481c-8fef-55eceba2b082","resolution":{"observed_at":"2026-08-07T04:34:14.727112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12528","last_updated":"2025-09-08T02:42:57Z","snapshot_observed_at":"2026-07-06T19:04:43.716629Z","submitted_at":"2024-08-22T16:32:32Z","title":"Show-o: One Single Transformer to Unify Multimodal Understanding and Generation","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12528","snapshot_observed_at":"2026-08-07T04:34:09.844460Z","title":"Show-o: One single transformer to unify multimodal understanding and generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:09.844460Z"},"links":{"cited_paper":"/paper/2408.12528","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:073f2ab32ae82fa86e6d4e23122194eba01003e30d4165b5e3eddc8599d7d3c7","observation_id":"ff7c1f67-5593-4f19-ae75-b98806a9f200","resolution":{"observed_at":"2026-08-07T04:34:09.844460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11703","last_updated":"2024-03-18T12:04:11Z","snapshot_observed_at":"2026-08-06T16:32:30.757011Z","submitted_at":"2024-03-18T12:04:11Z","title":"LLaVA-UHD: an LMM Perceiving Any Aspect Ratio and High-Resolution Images","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11703","snapshot_observed_at":"2026-08-07T04:34:09.966497Z","title":"Llava-uhd: an LMM perceiving any aspect ratio and high-resolution images","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:09.966497Z"},"links":{"cited_paper":"/paper/2403.11703","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:6bf38b44d135f996b6c4b4b5bb9bc0ee8b8f60b29cc005f9b417957b9cd75c6b","observation_id":"d30a2810-ca8f-4506-841d-cab57a499d09","resolution":{"observed_at":"2026-08-07T04:34:09.966497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.acl-long.641","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:12.436515Z","title":"Multiinstruct: Improving multi-modal zero-shot learning via instruction tuning","venue":null,"work_id":"fdf534fe-c05b-470c-868b-1042b0083e06","year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:10.109785Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:3d331e28e12a27e29b9978e700011a51b906b85afc6737196432d7bec26634c5","observation_id":"877fdd3f-effc-4341-8f7d-aee0f9d1ff3a","resolution":{"observed_at":"2026-08-07T04:34:12.568692Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11690","last_updated":"2024-02-18T19:38:44Z","snapshot_observed_at":"2026-07-06T17:31:53.996417Z","submitted_at":"2024-02-18T19:38:44Z","title":"Vision-Flan: Scaling Human-Labeled Tasks in Visual Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11690","snapshot_observed_at":"2026-08-07T04:34:10.263657Z","title":"Vision-flan: Scaling human-labeled tasks in visual instruction tuning, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:10.263657Z"},"links":{"cited_paper":"/paper/2402.11690","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:2ee81a76b7e1d0594783b1e036544f53d38379d722323d32f43a9f5ac2359831","observation_id":"60240428-c3d7-42de-b785-8280a413f474","resolution":{"observed_at":"2026-08-07T04:34:10.263657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:14.339155Z","title":"Modality-specialized synergizers for interleaved vision-language generalists","venue":null,"work_id":"3ab2c877-f860-42c9-ae40-e3d207ef020d","year":2025},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:10.391353Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:0ff6d45a52386f3a25522a9b124357567b913e5115caed2bdd981d1216688e99","observation_id":"1cd2c1f0-cfcb-4753-a007-e055acbfb868","resolution":{"observed_at":"2026-08-07T04:34:14.442097Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:14.111713Z","title":"Retrieval-augmented multimodal language modeling","venue":null,"work_id":"8f2ddb06-e1a6-4d2a-b88d-e612e5708fe0","year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:10.582907Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:07b57b692617b11c22f030449ec7ad7240a326a270ff1fa09fa714cbd5d113b9","observation_id":"8f22ff07-7612-4a2a-b947-666c44328754","resolution":{"observed_at":"2026-08-07T04:34:14.215441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-07T04:34:10.749553Z","title":"mplug-owl: Modularization empowers large language models with multimodality","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:10.749553Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:2dca81ad596afbbd199ecc55017d31935fd61ea4bbb8b5a1f0a1fd2208a52c19","observation_id":"e7c6d5cf-9f27-42b2-ad58-643a88ed7b46","resolution":{"observed_at":"2026-08-07T04:34:10.749553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.06687","last_updated":"2023-11-06T07:02:19Z","snapshot_observed_at":"2026-07-06T15:41:16.963249Z","submitted_at":"2023-06-11T14:01:17Z","title":"LAMM: Language-Assisted Multi-Modal Instruction-Tuning Dataset, Framework, and Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.06687","snapshot_observed_at":"2026-08-07T04:34:10.895265Z","title":"LAMM: language-assisted multi-modal instruction-tuning dataset, framework, and benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:10.895265Z"},"links":{"cited_paper":"/paper/2306.06687","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:d92552d6b1d65de01fb100777133192b0fb4ee67a9b8a84189883555bc12210a","observation_id":"eb2dfaed-8279-4d42-9cac-c2fa98baea49","resolution":{"observed_at":"2026-08-07T04:34:10.895265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.02591","last_updated":"2023-09-05T21:27:27Z","snapshot_observed_at":"2026-08-06T19:04:26.186950Z","submitted_at":"2023-09-05T21:27:27Z","title":"Scaling Autoregressive Multi-Modal Models: Pretraining and Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.02591","snapshot_observed_at":"2026-08-07T04:34:11.061508Z","title":"Scaling autoregressive multi-modal models: Pretraining and instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:11.061508Z"},"links":{"cited_paper":"/paper/2309.02591","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:d84471d59955a41562bc4cfa4217e3f1e1ad8c7995927cebe6829435d96ce495","observation_id":"e9b4edd9-af5c-4184-8228-db135bf16a19","resolution":{"observed_at":"2026-08-07T04:34:11.061508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02490","last_updated":"2024-12-01T05:46:03Z","snapshot_observed_at":"2026-08-06T13:03:19.727877Z","submitted_at":"2023-08-04T17:59:47Z","title":"MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.02490","snapshot_observed_at":"2026-08-07T04:34:11.250107Z","title":"Mm-vet: Evaluating large multimodal models for integrated capabilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:11.250107Z"},"links":{"cited_paper":"/paper/2308.02490","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:4a49072c641ce9e9f24c19c16541b29584ed6e1ec2167eb30685adaf4268b0a9","observation_id":"825db274-4934-4ccb-9e22-98aa3e98a425","resolution":{"observed_at":"2026-08-07T04:34:11.250107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:11.405831Z","title":"Mmmu: A massive multi-discipline multimodal understanding and reasoning benchmark for expert agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:11.405831Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:a575fd734fc259a5c6c998b730b4389153590801db21f499f1dcb1a5cebff99f","observation_id":"12e13317-6f53-4ab6-ad1b-3267e3526906","resolution":{"observed_at":"2026-08-07T04:34:11.405831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:34:11.533516Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:11.533516Z"},"links":{"citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:5d814a796a3682274e7e09c42f8bae7ed2c3e31672fbe875c00ca273240a92f0","observation_id":"1f35042e-3a44-45ce-86a2-9093cc06b9a2","resolution":{"observed_at":"2026-08-07T04:34:11.533516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12226","last_updated":"2025-09-08T07:04:17Z","snapshot_observed_at":"2026-07-06T17:32:15.447061Z","submitted_at":"2024-02-19T15:33:10Z","title":"AnyGPT: Unified Multimodal LLM with Discrete Sequence Modeling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12226","snapshot_observed_at":"2026-08-07T04:34:11.720293Z","title":"Anygpt: Unified multimodal LLM with discrete sequence modeling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:11.720293Z"},"links":{"cited_paper":"/paper/2402.12226","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:72c23ebd029a8234d26e0615cf45c66a4e52e4425372e0b3c6b05a9bedc69645","observation_id":"c35cc7c8-52e8-4e1f-897b-b484e3188537","resolution":{"observed_at":"2026-08-07T04:34:11.720293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11039","last_updated":"2024-08-20T17:48:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-20T17:48:20Z","title":"Transfusion: Predict the Next Token and Diffuse Images with One Multi-Modal Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.11039","snapshot_observed_at":"2026-08-07T04:34:11.878440Z","title":"Transfusion: Predict the next token and diffuse images with one multi-modal model, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:11.878440Z"},"links":{"cited_paper":"/paper/2408.11039","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:d8c677c623dc1faae0dda4979d34857347dd78d5b03786813f002fc30b32cca3","observation_id":"1137c489-997b-411f-82db-76314dc056c9","resolution":{"observed_at":"2026-08-07T04:34:11.878440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-07T04:34:12.006826Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:12.006826Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:66b6aa5291302c1dedd944b46961b9190aad7f15d1f82d12c8574906977637d1","observation_id":"df24cd45-8a6c-4865-9926-9dcdf3affe65","resolution":{"observed_at":"2026-08-07T04:34:12.006826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.09251","last_updated":"2023-12-14T18:59:43Z","snapshot_observed_at":"2026-07-06T17:02:05.607732Z","submitted_at":"2023-12-14T18:59:43Z","title":"VL-GPT: A Generative Pre-trained Transformer for Vision and Language Understanding and Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.09251","snapshot_observed_at":"2026-08-07T04:34:12.154340Z","title":"VL-GPT: A generative pre-trained transformer for vision and language understanding and generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation","version":2},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-07T04:34:12.154340Z"},"links":{"cited_paper":"/paper/2312.09251","citing_paper":"/paper/2506.10395"},"observation_digest":"sha256:c07a369ce1ebdc2b86134819c9a75e2959483ac8c6c3106c80e266ba4f9512e7","observation_id":"9c091c59-76f3-4992-bda8-50607a9d87c7","resolution":{"observed_at":"2026-08-07T04:34:12.154340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.10395","last_updated":"2025-07-12T20:42:24Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T04:25:14.940851Z","submitted_at":"2025-06-12T06:37:34Z","title":"Pisces: An Auto-regressive Foundation Model for Image Understanding and Generation"},"reference_resolution":{"displayed":85,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":65,"verified_exact":2,"verified_fuzzy":18},"total_outbound_references":85},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 85 of 85 outbound references and 2 inbound Pith citation observations for arXiv:2506.10395."}