{"as_of":"2026-08-04T08:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:526e38b61d6ba5e9bafd533c88e78f77fec9f8f65ac096e3be468b102f1846cb","coverage":[{"denominator":173,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T05:12:37.339084Z","state":"measured"},{"denominator":115,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":115,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":15,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T04:17:41.402174Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T02:04:26.362870Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2605.25343","last_updated":"2026-05-25T01:57:43Z","snapshot_observed_at":"2026-07-06T23:35:16.647496Z","submitted_at":"2026-05-25T01:57:43Z","title":"Toward Native Multimodal Modeling: A Roadmap","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T22:58:38.610609Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2605.25343"},"observation_digest":"sha256:62999a228e1675af742d034430ece5d96caa53168d7ced0c32441cf812017490","observation_id":"7397f622-e740-4968-ad16-2d500d2353f7","resolution":{"observed_at":"2026-06-29T23:04:01.929840Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2605.27310","last_updated":"2026-05-26T17:20:05Z","snapshot_observed_at":"2026-08-02T03:29:23.760478Z","submitted_at":"2026-05-26T17:20:05Z","title":"How and What to Imagine? Visual Thinking in Unified Multimodal Models for Cross-View Spatial Reasoning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-29T18:19:40.323661Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2605.27310"},"observation_digest":"sha256:d2bea7b7991acb3639c59b7dfea11b475ec7ed3696083e5dfc6e600d936bfcee","observation_id":"78146988-1f12-4e74-b914-132dd2518d34","resolution":{"observed_at":"2026-06-29T18:23:50.686391Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2605.31604","last_updated":"2026-07-03T09:50:38Z","snapshot_observed_at":"2026-08-03T15:39:37.184662Z","submitted_at":"2026-05-29T17:59:55Z","title":"Representation Forcing for Bottleneck-Free Unified Multimodal Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T22:54:10.460872Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2605.31604"},"observation_digest":"sha256:310469e1de811d14e3cc9940c51f4a673511224cef306f675d081712b9598989","observation_id":"ac6bc258-b7d3-4f08-b0bf-94cf4f60932c","resolution":{"observed_at":"2026-07-01T19:16:00.959654Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-12T15:31:57.426559Z","title":"Sensenova-u1: Unifying multimodal understanding and generation with neo-unify architecture.arXiv preprint arXiv:2605.12500, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.31604","last_updated":"2026-07-03T09:50:38Z","snapshot_observed_at":"2026-08-03T15:39:37.184662Z","submitted_at":"2026-05-29T17:59:55Z","title":"Representation Forcing for Bottleneck-Free Unified Multimodal Models","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-12T15:31:57.426559Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2605.31604"},"observation_digest":"sha256:c00bd90f314eed6d2b9cce0edf33193392b962717193ebdb17663b70d9330676","observation_id":"eec1c095-5b55-49ab-8e4e-e111cafe4944","resolution":{"observed_at":"2026-07-12T15:31:57.426559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2606.15236","last_updated":"2026-06-16T13:59:42Z","snapshot_observed_at":"2026-08-03T10:13:54.916430Z","submitted_at":"2026-06-13T10:29:11Z","title":"Show the Signal, Hide the Noise: Spectral Forcing for Pixel-Space Diffusion","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T04:21:43.851963Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2606.15236"},"observation_digest":"sha256:7ab46d28033817780e0b8c6bccf4e558cbcb5eaee3408bf8640baaefa7cf5107","observation_id":"e62b9711-82d3-4ed2-ae0c-cb8bbada6876","resolution":{"observed_at":"2026-07-03T17:18:43.615339Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2606.20100","last_updated":"2026-06-18T11:20:05Z","snapshot_observed_at":"2026-08-02T10:05:29.820038Z","submitted_at":"2026-06-18T11:20:05Z","title":"WeGenBench: A Multidimensional Diagnostic Benchmark towards Text-to-Image Model Optimization","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-26T17:54:09.656061Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2606.20100"},"observation_digest":"sha256:e4a353d0c5a13a092c29f24beedc40000392ab2d7ebcb83e55b3396cb7f9c51f","observation_id":"2875229e-e963-4ef7-ac89-44815a95dc0a","resolution":{"observed_at":"2026-07-04T03:39:29.350086Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2606.31982","last_updated":"2026-06-30T17:20:29Z","snapshot_observed_at":"2026-08-02T20:40:00.392185Z","submitted_at":"2026-06-30T17:20:29Z","title":"ERA: Entropy-Guided Visual Token Pruning with Rectified Attention for Efficient MLLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-01T05:44:12.758733Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2606.31982"},"observation_digest":"sha256:5344c9742469364873c1f1f4955dcd2f600342d30a50446f7d3ab2154e2dedb1","observation_id":"482ae9cd-c5df-4a8d-a343-ee798656f161","resolution":{"observed_at":"2026-07-01T10:15:44.328438Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2606.32039","last_updated":"2026-06-30T17:59:57Z","snapshot_observed_at":"2026-07-07T00:05:41.920778Z","submitted_at":"2026-06-30T17:59:57Z","title":"GEAR: Guided End-to-End AutoRegression for Image Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-01T05:19:21.647714Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2606.32039"},"observation_digest":"sha256:4f74de4748cc8b27a5b278b075978d6964fc88690b8d92121a55d4dedb2f7ecd","observation_id":"3b0384e4-509b-42e6-ad03-e48e446b8c9b","resolution":{"observed_at":"2026-07-01T10:35:42.646547Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2607.02290","last_updated":"2026-07-02T15:07:47Z","snapshot_observed_at":"2026-07-07T00:07:47.719699Z","submitted_at":"2026-07-02T15:07:47Z","title":"DisciplineGen-1M: A Large-Scale Dataset for Multidisciplinary Visual Generation and Editing","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-03T15:38:36.388570Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2607.02290"},"observation_digest":"sha256:d8de1fe1dba8ace29b0a47c167cbb9c4920511fdb40bbabd65b6e258cde8f5f8","observation_id":"2c37936e-c6c7-41b2-bcd1-c810c43c906f","resolution":{"observed_at":"2026-07-03T15:48:35.117418Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":"2605.12500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-08T02:04:26.362870Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","venue":"cs.CV","work_id":"3fc01381-3983-477d-a5cf-193da464f6e3","year":2026},"citing_paper":{"arxiv_id":"2607.06560","last_updated":"2026-07-07T17:58:33Z","snapshot_observed_at":"2026-07-10T23:18:28.655344Z","submitted_at":"2026-07-07T17:58:33Z","title":"Vision as Unified Multimodal Generation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-08T01:54:30.649092Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2607.06560"},"observation_digest":"sha256:a51418ee99be0ee16b1fd8479fcf80621404978735a9ad2dbc05c5a09dc08b73","observation_id":"fc9d60c5-a1f8-4468-a9f1-a5ba17bf8b9e","resolution":{"observed_at":"2026-07-08T02:04:26.364613Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-08-01T23:43:35.072852Z","title":"arXiv preprint arXiv:2605.12500 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15272","last_updated":"2026-07-16T17:58:36Z","snapshot_observed_at":"2026-08-03T23:54:06.403225Z","submitted_at":"2026-07-16T17:58:36Z","title":"SciDiagramEdit: Learning to Edit Scientific Diagrams from Paper Revisions","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-01T23:43:35.072852Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2607.15272"},"observation_digest":"sha256:3081acd1f3d8e1d12ca569523f3c56c5b80a9dd906c128754a2b0aafa87a8638","observation_id":"9cace125-3dd5-41a2-88e9-ce4217fe4fae","resolution":{"observed_at":"2026-08-01T23:43:35.072852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-08-01T19:41:47.438584Z","title":"arXiv preprint arXiv:2605.12500 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16898","last_updated":"2026-08-02T08:43:47Z","snapshot_observed_at":"2026-08-04T07:56:58.476892Z","submitted_at":"2026-07-18T17:35:33Z","title":"Cross-Branch Conflict as a Shield: Safeguarding Facial Identities in Unified Multimodal Image Editing","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-01T19:41:47.438584Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2607.16898"},"observation_digest":"sha256:c63cd783eafce483901d9f646aa8c6819e948966202e6dbb34c64ae95c9d61fe","observation_id":"4579cabf-9355-4505-80c7-f018bb3855eb","resolution":{"observed_at":"2026-08-01T19:41:47.438584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-08-04T04:17:41.402174Z","title":"arXiv preprint arXiv:2605.12500 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16898","last_updated":"2026-08-02T08:43:47Z","snapshot_observed_at":"2026-08-04T07:56:58.476892Z","submitted_at":"2026-07-18T17:35:33Z","title":"Cross-Branch Conflict as a Shield: Safeguarding Facial Identities in Unified Multimodal Image Editing","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-04T04:17:41.402174Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2607.16898"},"observation_digest":"sha256:f8848027c40f3ee2a673cfba17a60af296b09b9a1cacb938956ef5ae2f895962","observation_id":"7cb9eaa3-f7b0-4224-8e16-8cf6f3a294b7","resolution":{"observed_at":"2026-08-04T04:17:41.402174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-08-01T18:56:57.619206Z","title":"Sensenova-u1: Unifying multimodal understanding and generation with neo-unify architecture.arXiv preprint arXiv:2605.12500, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.17140","last_updated":"2026-07-19T08:53:01Z","snapshot_observed_at":"2026-08-01T18:56:55.577266Z","submitted_at":"2026-07-19T08:53:01Z","title":"STBridge: Shared-Target Alignment for Bridging Understanding and Generation in UMMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T18:56:57.619206Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2607.17140"},"observation_digest":"sha256:39031f7c6cbafbe09b23d269db5f767a56b89b348f9e16855a4474ff30c484ce","observation_id":"37a337e9-960c-47a3-aee4-a1388418953c","resolution":{"observed_at":"2026-08-01T18:56:57.619206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12500","snapshot_observed_at":"2026-07-31T14:57:39.507916Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.24439","last_updated":"2026-07-27T13:47:38Z","snapshot_observed_at":"2026-08-03T05:34:53.299119Z","submitted_at":"2026-07-27T13:47:38Z","title":"Unifying Generative Recall and Multi-Objective Ranking in a Single Decoder-Only Sequence","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-31T14:57:39.507916Z"},"links":{"cited_paper":"/paper/2605.12500","citing_paper":"/paper/2607.24439"},"observation_digest":"sha256:49c7bc77c41c6ff5559b47297dd8d6529b4b09849b8e0ec077e702e5f3e33f34","observation_id":"8c2401b0-516e-4693-a4ea-365964f7b622","resolution":{"observed_at":"2026-07-31T14:57:39.507916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.12500/citation-record","integrity":"/paper/2605.12500/integrity","json":"/paper/2605.12500/citation-record.json","paper":"/paper/2605.12500"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"40995f15-58e1-4bdb-8885-4ad729de9a28","year":2022},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:6726517514b5440c797ed7b424a9538c4f1e1176bbadc838be0503fc95dac9a9","observation_id":"1f4acabf-572c-4638-aa06-a34a8129976b","resolution":{"observed_at":"2026-05-13T10:52:39.506554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:26ba493041944b1462fb842266d58e038e06edf41c26e9f732d418f1e4c498ce","observation_id":"104bddc6-2e96-497b-9350-c5dfb4399823","resolution":{"observed_at":"2026-05-13T05:17:18.807782Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2408.07009","doi":"10.48550/arxiv.2408.07009","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Imagen 3","venue":null,"work_id":"a1dd317f-8300-4a79-a1d0-92ddd93fa983","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:7ff243a210db238549793abc06e2a5743c37acfe136e8b926cefde801becf908","observation_id":"e3742a16-86be-4c8e-a868-4d3c247d5541","resolution":{"observed_at":"2026-05-13T05:17:18.480326Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:30.169107+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:30.169107+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing our multimodal models","venue":null,"work_id":"93623b82-30f7-4fff-8732-39e4674d3ff1","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:92a168fd8f69f6b77f55ea27051cf9b061653758c98c907779ba58dd12bfd290","observation_id":"8e998da3-23bf-4739-a366-62998a19074f","resolution":{"observed_at":"2026-05-13T10:52:39.510060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improving image generation with better captions","venue":null,"work_id":"89483649-39dc-4d6c-87c2-37bc6164ce45","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:d91856cc7d7659ee8a746e040f40c3078d1e4044b94ed2a33f591182cee93dda","observation_id":"0a484062-6dee-4232-9ee5-fc2c2cb62350","resolution":{"observed_at":"2026-05-13T10:52:39.543956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Seedream 4.5","venue":null,"work_id":"1eab5521-9e04-4cda-87e0-8ebb56d16cd4","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:baebb7aed7dbc8223d107964bed62c180ebb48502fd220efd5625d283474cc45","observation_id":"fd0f4b83-43db-4a66-8d8e-efa064b3709c","resolution":{"observed_at":"2026-05-13T10:52:39.470808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.22699","last_updated":"2026-07-06T06:19:03Z","snapshot_observed_at":"2026-08-03T19:47:26.577384Z","submitted_at":"2025-11-27T18:52:07Z","title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","version":5},"cited_work":{"arxiv_id":"2511.22699","doi":"10.48550/arxiv.2511.22699","metadata_source":"pith","pith_arxiv_id":"2511.22699","snapshot_observed_at":"2026-07-11T00:07:42.322641Z","title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","venue":"cs.CV","work_id":"f1080a62-48e1-4255-b023-7556be57370d","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2511.22699","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:770e57a7a302e89cad9c87cbd92ca0c29f26b1977d1286a0fd0494d133fb2679","observation_id":"1595dcbd-9599-46a1-a445-a10c6da4c2a8","resolution":{"observed_at":"2026-05-13T05:17:18.556563Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-25T00:53:20.152696+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T00:53:20.152696+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22705","last_updated":"2025-05-28T17:59:15Z","snapshot_observed_at":"2026-07-30T00:45:52.058972Z","submitted_at":"2025-05-28T17:59:15Z","title":"HiDream-I1: A High-Efficient Image Generative Foundation Model with Sparse Diffusion Transformer","version":1},"cited_work":{"arxiv_id":"2505.22705","doi":"10.48550/arxiv.2505.22705","metadata_source":"pith","pith_arxiv_id":"2505.22705","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"HiDream-I1: A High-Efficient Image Generative Foundation Model with Sparse Diffusion Transformer","venue":"cs.CV","work_id":"68d4c0f7-3dfd-438d-a823-6a93fd0a835d","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2505.22705","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:518d8423686dff93d4e3ffa4c6ae167d3152bf19c8496a1935faf7c1af440ef7","observation_id":"19a68527-dd97-4535-9fd3-19439d270e91","resolution":{"observed_at":"2026-05-16T17:13:39.137651Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.13142","doi":"10.48550/arxiv.2508.13142","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Has gpt-5 achieved spatial intelligence? an empirical study","venue":null,"work_id":"b0d24cae-8b82-4d58-9b26-2a803fd083cc","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:0071fab1fa5e81d0ae08282fa3d8ef0d34b5dd58c039693e09334fb288e0914a","observation_id":"0851aadd-5583-42a8-a682-06f9b5b42945","resolution":{"observed_at":"2026-05-13T05:17:18.785885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling spatial intelligence with multimodal foundation models","venue":null,"work_id":"cbf7c0fd-8d96-4984-afa9-2b6b42a2d703","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:c76f70f5f95b0f7424a699ec750c1aa1bcbb35cc69d9d842a5b8de284bfb609f","observation_id":"946f2202-1768-4755-81df-5355968f8f5b","resolution":{"observed_at":"2026-05-13T10:52:39.472623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.23951","last_updated":"2026-06-26T07:53:46Z","snapshot_observed_at":"2026-07-06T22:31:00.843452Z","submitted_at":"2025-09-28T16:14:10Z","title":"HunyuanImage 3.0 Technical Report","version":3},"cited_work":{"arxiv_id":"2509.23951","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.23951","snapshot_observed_at":"2026-07-04T16:59:57.981208Z","title":"HunyuanImage 3.0 Technical Report","venue":"cs.CV","work_id":"36511e33-8360-43dd-9407-dddf867c1d7c","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2509.23951","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:b9a9962fdbfa8ae24c2d2c8c6b5655f0cfa276971ea0b872a92e4b0bde1fcbf4","observation_id":"4aba686b-0fa6-4085-a1b0-ab7f8b4af776","resolution":{"observed_at":"2026-05-16T02:02:32.945705Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07977","last_updated":"2025-06-26T15:47:09Z","snapshot_observed_at":"2026-08-03T22:18:20.423791Z","submitted_at":"2025-06-09T17:50:21Z","title":"OneIG-Bench: Omni-dimensional Nuanced Evaluation for Image Generation","version":3},"cited_work":{"arxiv_id":"2506.07977","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07977","snapshot_observed_at":"2026-07-04T13:39:51.249678Z","title":"Oneig-bench: Omni- dimensional nuanced evaluation for image generation","venue":null,"work_id":"1b87af32-6bef-4951-a841-9ea611693ca6","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2506.07977","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:7935a8127769485146443ca7484d50ceb2b9e0aa5ee2c246bf72925230e32be2","observation_id":"71d5c1ad-1931-4f39-988a-69d7aa632b8f","resolution":{"observed_at":"2026-05-13T05:17:18.581960Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.25162","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T14:33:54.307986Z","title":"arXiv preprint arXiv:2509.25162 (2025) 4","venue":null,"work_id":"28315f59-d5f2-4386-b527-c6735b88b398","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:4c29143ab47d9e49af9a549366aa1342651432fecf28232d3e39a6f09cf69473","observation_id":"d12a3c72-d5b6-4632-b541-f6ae0ca03266","resolution":{"observed_at":"2026-05-13T05:17:18.570432Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09568","last_updated":"2025-05-14T17:11:07Z","snapshot_observed_at":"2026-07-06T21:23:57.084147Z","submitted_at":"2025-05-14T17:11:07Z","title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","version":1},"cited_work":{"arxiv_id":"2505.09568","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.09568","snapshot_observed_at":"2026-07-08T00:04:22.585543Z","title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","venue":"cs.CV","work_id":"86d896d2-592f-4d9b-938e-dfeb11f9388f","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2505.09568","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:6c6e504d33ef765ae2810b871b6cbb3b0c66da102bf0418ffdddc807779d9365","observation_id":"cb6c6fd3-56b9-4952-987d-dc12d54f630a","resolution":{"observed_at":"2026-05-13T05:17:18.725291Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.06521","last_updated":"2026-07-07T15:04:59Z","snapshot_observed_at":"2026-08-03T11:25:34.815323Z","submitted_at":"2026-01-10T10:42:44Z","title":"BabyVision: Visual Reasoning Beyond Language","version":2},"cited_work":{"arxiv_id":"2601.06521","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.06521","snapshot_observed_at":"2026-07-10T09:37:00.824668Z","title":"Babyvision: Visual reasoning beyond language","venue":"cs.CV","work_id":"c3797ec2-73a2-46c7-bc93-8bf33dd8b187","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2601.06521","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:fed688e2205d608cf9582848b90fff641ec995c154ced57872c883a7611e1653","observation_id":"2eecc1b5-7e55-4014-b21c-e4c0d30ef38b","resolution":{"observed_at":"2026-07-08T02:18:40.971166Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Are we on the right way for evaluating large vision-language models? Advances in Neural Information Processing Systems, 37:27056–27087","venue":null,"work_id":"a58e7415-5055-464d-b6c4-2a8acccf7c48","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:99fdca5679485335a086827f077f2f1a5af976b00ad332cec9f3241f731c028a","observation_id":"0ae41b3c-c004-49b2-90eb-e3d964c85286","resolution":{"observed_at":"2026-05-13T10:52:39.511912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07963","last_updated":"2025-04-10T17:59:56Z","snapshot_observed_at":"2026-08-04T06:26:36.498042Z","submitted_at":"2025-04-10T17:59:56Z","title":"PixelFlow: Pixel-Space Generative Models with Flow","version":1},"cited_work":{"arxiv_id":"2504.07963","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.07963","snapshot_observed_at":"2026-07-04T20:50:11.498748Z","title":"arXiv preprint arXiv:2504.07963 (2025)","venue":null,"work_id":"1beb1a44-0887-4a87-93c8-5dca148563bd","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2504.07963","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:7d0444e52c4ab3d7959b6346f5d59df1df43007d8119040aa54b64a05e7886c1","observation_id":"f495c277-ff77-46fc-a33a-c0285f78f7df","resolution":{"observed_at":"2026-05-13T05:17:18.498517Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17811","last_updated":"2025-01-29T18:00:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-29T18:00:19Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","version":1},"cited_work":{"arxiv_id":"2501.17811","doi":"10.48550/arxiv.2501.17811","metadata_source":"pith","pith_arxiv_id":"2501.17811","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","venue":"cs.AI","work_id":"67d9e391-26d1-459e-ab56-07e60511c886","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2501.17811","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:3cc8a273c290ca0b0e3eeda45bfbbe17b3f8c57a361e6dd4b1bc66f100a866f7","observation_id":"b6ce9a3f-84d6-475b-bcd1-3ce4aadb8ac0","resolution":{"observed_at":"2026-05-13T05:17:18.728349Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A single transformer for scalable vision-language modeling.Transactions on Machine Learning Research","venue":null,"work_id":"6f7bc753-5a9c-44eb-9953-bdcd836528a4","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:1178c8e06a7c5ddb5db024416cb835fbb58a0c6ce41e274d041ad9a33b87228e","observation_id":"e05ac1b0-8a00-447e-9527-580c0425a742","resolution":{"observed_at":"2026-05-13T10:52:39.474488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.18822","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T16:38:39.577895Z","title":"arXiv preprint arXiv:2511.18822 (2025)","venue":null,"work_id":"1bcc412b-04bd-477a-87df-54e466aa8616","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:c42142d06fdd99e9b20766e2ea191b392861279490221265e3a672f0d2a6b77b","observation_id":"1a4291a7-35e8-4184-93c7-54ea208fe93f","resolution":{"observed_at":"2026-05-13T05:17:18.701506Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06135","last_updated":"2024-07-08T17:08:02Z","snapshot_observed_at":"2026-07-06T18:43:09.852565Z","submitted_at":"2024-07-08T17:08:02Z","title":"ANOLE: An Open, Autoregressive, Native Large Multimodal Models for Interleaved Image-Text Generation","version":1},"cited_work":{"arxiv_id":"2407.06135","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06135","snapshot_observed_at":"2026-07-04T16:29:56.712695Z","title":"Anole: An open, autoregressive, native large multimodal models for interleaved image-text generation","venue":null,"work_id":"31ce9d99-2071-41a0-9f51-51b8c5e3ba7e","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2407.06135","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:d7726722e44733e0a5e750c8109620723f7866a0b4c5175f0fc668e706f406a5","observation_id":"9f5be935-48d2-433d-8137-702321f38ce5","resolution":{"observed_at":"2026-05-13T05:17:18.811647Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.05595","last_updated":"2025-07-08T02:14:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-08T02:14:10Z","title":"PaddleOCR 3.0 Technical Report","version":1},"cited_work":{"arxiv_id":"2507.05595","doi":"10.48550/arxiv.2507.05595.url:http://arxiv.org/","metadata_source":"pith","pith_arxiv_id":"2507.05595","snapshot_observed_at":"2026-07-11T02:17:46.362300Z","title":"PaddleOCR 3.0 Technical Report","venue":"cs.CV","work_id":"444c549c-1895-4cae-a2d3-c13d7ed5f462","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2507.05595","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:268a2d8519084a00453d507350703caf5b32e8bae5b2f896b1189c697effb75f","observation_id":"1857c167-9b5c-4916-913a-b1ed054df8d6","resolution":{"observed_at":"2026-05-14T23:24:28.568269Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.26583","last_updated":"2025-10-30T15:11:16Z","snapshot_observed_at":"2026-07-31T17:39:49.561332Z","submitted_at":"2025-10-30T15:11:16Z","title":"Emu3.5: Native Multimodal Models are World Learners","version":1},"cited_work":{"arxiv_id":"2510.26583","doi":"10.48550/arxiv.2510.26583","metadata_source":"pith","pith_arxiv_id":"2510.26583","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Emu3.5: Native Multimodal Models are World Learners","venue":"cs.CV","work_id":"518b061e-87a3-45f9-8419-30a14d89b122","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2510.26583","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:8140bae6bf9d57834a7cbe31adc16ac5fea141c637135f39b2f6fb2028f28e9c","observation_id":"15f8e895-b4ee-41ae-abcc-e9c546de505b","resolution":{"observed_at":"2026-05-18T01:12:13.931083Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:49:30.986785+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:49:30.986785+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini 3 pro image model card, Nov 2025","venue":null,"work_id":"818c48df-7ae0-4e93-840f-9634875697c6","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:a3b9a61d657909479c1e7f7fe0f6030e469dc598fa08894d2f8a3bbe586df283","observation_id":"9f12aa6a-6822-43f5-bb1d-4b552b9d01e0","resolution":{"observed_at":"2026-05-13T10:52:39.479601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini 2.0 flash","venue":null,"work_id":"8478eea6-18f6-4b2c-8167-4ee5dcfea3a0","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:696a24fdec2f672ea5d691a333f95391d0906efa8600d19216fa3e8c7dda4b54","observation_id":"4c991b3b-629b-40c6-947a-d66b0b67d4c1","resolution":{"observed_at":"2026-05-13T10:52:39.477915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini 2.5 flash & gemini 2.5 flash image model card","venue":null,"work_id":"76a09a85-0245-4c5d-8a20-2f3bb2b161e0","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:f9d0c812c1047cbe20074ec71fbf559f4fa503bfb9b73e502389dcd18e989c28","observation_id":"659920e2-20a9-40e7-8e91-843b0e633f56","resolution":{"observed_at":"2026-05-13T10:52:39.535110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemma 4: Byte for byte, the most capable open models","venue":null,"work_id":"7650bea2-c61b-4566-ab40-a379d73c214f","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:377d4edb0f48778dfba579c8803b6191c2aee0e78ba48d5770a13a66ef23eaeb","observation_id":"ed3e9c13-99ca-440e-9b0d-b61e561b7be2","resolution":{"observed_at":"2026-05-13T10:52:39.469120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14683","last_updated":"2025-07-27T11:45:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T17:59:30Z","title":"Emerging Properties in Unified Multimodal Pretraining","version":3},"cited_work":{"arxiv_id":"2505.14683","doi":"10.48550/arxiv.2505.14683","metadata_source":"pith","pith_arxiv_id":"2505.14683","snapshot_observed_at":"2026-07-10T13:37:06.841769Z","title":"Emerging Properties in Unified Multimodal Pretraining","venue":"cs.CV","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2505.14683","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:32980a2a3cc55e09c3f8da5cc956b559107b74130ff7ee7efe635dd5a76bd024","observation_id":"f0bf9667-6cbf-43ae-9f64-b1a2c94337cc","resolution":{"observed_at":"2026-05-13T05:17:18.694161Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unveiling encoder-free vision-language models","venue":null,"work_id":"4bdf9fcc-96e7-401e-a3a4-01e30c34ad12","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:ec8268a8aa7222d04a80bdad082393b16b5631086e139635c69508192cfe431f","observation_id":"d29c59fa-dbeb-4925-b0fa-1a06aff5717c","resolution":{"observed_at":"2026-05-13T10:52:39.461578Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.14979","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T07:35:28.718163Z","title":"From pixels to words–towards native vision-language primitives at scale.arXiv preprint arXiv:2510.14979, 2025a","venue":null,"work_id":"73f6fd43-0cd3-49fc-afac-14eda2b1a95f","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:edc188861ac125aa0d7aac5718a111aee94f56886246145bf31f490ce1365778","observation_id":"3ecc11cd-201a-4d42-ad50-bd6746178fbf","resolution":{"observed_at":"2026-05-13T05:17:18.545969Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evev2: Improved baselines for encoder-free vision-language models","venue":null,"work_id":"2c5464ae-f76a-4e29-9655-8520635cef63","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:32868005481404efc2c160abd81ac3fb487126726c5cb4e5f4adbfabed402414","observation_id":"5b4a9577-7806-441f-ad61-b19bc3a98ca1","resolution":{"observed_at":"2026-05-13T10:52:39.465470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.23461","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:39:58.047391Z","title":"Textcrafter: Accurately rendering multiple texts in complex visual scenes","venue":null,"work_id":"4d349026-15b1-433b-ace5-5104109b2bef","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:7c1b2a076d2538789c9fad3f87e4fef8e97b7fc393c67d1783c284b1c649b871","observation_id":"b06e328e-7ad2-44d0-9571-2bbcb51099a7","resolution":{"observed_at":"2026-05-13T05:17:18.690826Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14739","last_updated":"2025-03-28T15:21:44Z","snapshot_observed_at":"2026-07-29T21:41:16.650188Z","submitted_at":"2025-02-20T17:05:58Z","title":"SuperGPQA: Scaling LLM Evaluation across 285 Graduate Disciplines","version":4},"cited_work":{"arxiv_id":"2502.14739","doi":"10.48550/arxiv.2502.14739","metadata_source":"pith","pith_arxiv_id":"2502.14739","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"SuperGPQA: Scaling LLM Evaluation across 285 Graduate Disciplines","venue":"cs.CL","work_id":"58a97b2d-494b-4c1a-bd65-13fa6bb7f8f6","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2502.14739","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:388008d0d6f352e850acc73a44be39e1e8afb565746ff228743e891390177f57","observation_id":"1a555176-9818-49a9-b386-ae0d5eda1945","resolution":{"observed_at":"2026-05-16T00:47:04.366365Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-22T13:52:42.853111+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T13:52:42.853111+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling rectified flow transformers for high-resolution image synthesis","venue":null,"work_id":"c687e77f-00ed-494d-930f-0b48eeaa4a8b","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:9f1c7661a89820cdfb2a90c4435e7070ad6b63ca76786ebc2060a38e5e9ee5d9","observation_id":"6e84b32d-5411-4541-ac03-61fc41da9ab4","resolution":{"observed_at":"2026-05-13T10:52:39.529324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.19693","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T14:28:29.615040Z","title":"The prism hypothesis: Harmonizing semantic and pixel representations via unified autoencoding","venue":null,"work_id":"d77a406c-aae9-4870-aaeb-521e5ee9fa82","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:d3dc62af6772a96bd84b6b32d7a4e7bc030cf6604476cdd311ba10a2f00ff0fa","observation_id":"05cedc28-4d04-47ec-b11e-911515f64702","resolution":{"observed_at":"2026-05-13T05:17:18.745699Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.27684","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T01:07:45.294136Z","title":"Phased dmd: Few-step distribution matching distillation via score matching within subintervals.arXiv preprint arXiv:2510.27684","venue":null,"work_id":"438b7a67-4d0f-48fe-aa35-debbef88f1f7","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:ed769caf0c483a60ffaedf63e357ac42afb67058d814efd289897bb2d0392e5f","observation_id":"04eafcfc-4ee4-451f-866a-27ae18004a14","resolution":{"observed_at":"2026-05-13T05:17:18.712253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-02T18:54:27.250149Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"cited_work":{"arxiv_id":"2501.00321","doi":"10.48550/arxiv.2501.00321","metadata_source":"pith","pith_arxiv_id":"2501.00321","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","venue":"cs.CV","work_id":"58bef901-2b24-42dd-9edc-28c0b2148490","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2501.00321","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:c511e2543155b01e46f94c52df1dd373c571583d44e78e3bf3e5b7d87d7589ff","observation_id":"a11a964b-2432-49fb-8b43-141d99e97590","resolution":{"observed_at":"2026-05-17T20:33:27.157097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11346","last_updated":"2025-06-28T11:46:35Z","snapshot_observed_at":"2026-07-06T21:09:50.780345Z","submitted_at":"2025-04-15T16:19:07Z","title":"Seedream 3.0 Technical Report","version":3},"cited_work":{"arxiv_id":"2504.11346","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.11346","snapshot_observed_at":"2026-07-04T07:29:38.429370Z","title":"Seedream 3.0 Technical Report","venue":"cs.CV","work_id":"013e56d0-7f47-4d0e-bbca-e9540fc0e0cc","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2504.11346","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:16e8d97b260fd8a0ec71fbac98d2e4ccdbbc356b5755d19f909fd9f1f64e5a18","observation_id":"96d34c76-bfef-4ea5-ad22-f3d1a1c4a295","resolution":{"observed_at":"2026-05-13T07:55:38.865721Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01218","last_updated":"2023-10-02T14:03:02Z","snapshot_observed_at":"2026-08-02T03:02:58.907220Z","submitted_at":"2023-10-02T14:03:02Z","title":"Making LLaMA SEE and Draw with SEED Tokenizer","version":1},"cited_work":{"arxiv_id":"2310.01218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01218","snapshot_observed_at":"2026-07-09T20:16:29.588718Z","title":"Making llama see and draw with seed tokenizer","venue":"cs.CV","work_id":"9453feaf-a19d-4603-b9de-da50593c5d43","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2310.01218","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:8153efce3a9bdd1316bef4a3e90e021ae76382e6fb6c2a5d6ee2a193b8de08fb","observation_id":"f43ffe58-d574-4613-868b-02f1b8e1feb9","resolution":{"observed_at":"2026-05-13T05:17:18.821835Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14396","last_updated":"2025-03-02T07:53:44Z","snapshot_observed_at":"2026-07-06T18:03:51.687441Z","submitted_at":"2024-04-22T17:56:09Z","title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","version":2},"cited_work":{"arxiv_id":"2404.14396","doi":"10.48550/arxiv.2404.14396","metadata_source":"pith","pith_arxiv_id":"2404.14396","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","venue":"cs.CV","work_id":"15953092-dd9e-49ae-9f72-e28fc93a6068","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2404.14396","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:d92b7037eab8c278d7d29810f95aa5ca9e6ecc84467784ac5fa87f80ea006d62","observation_id":"ea04040f-fb56-4c54-9a00-b72e7a0b557e","resolution":{"observed_at":"2026-05-15T22:48:36.519178Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"644fb5d5-ccd1-4946-88a1-370bd96c5829","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:1ae0f9473498763399e7b48ec4ca0fd4e8039f84ebab237878f49f023d60c9e7","observation_id":"4c038810-2117-44c8-86ec-2bba18be15cf","resolution":{"observed_at":"2026-05-13T10:52:39.499323Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.22058","last_updated":"2025-07-29T17:59:04Z","snapshot_observed_at":"2026-08-03T01:06:06.514446Z","submitted_at":"2025-07-29T17:59:04Z","title":"X-Omni: Reinforcement Learning Makes Discrete Autoregressive Image Generative Models Great Again","version":1},"cited_work":{"arxiv_id":"2507.22058","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.22058","snapshot_observed_at":"2026-07-04T13:39:51.493261Z","title":"X-omni: Reinforcement learning makes discrete autoregressive image generative models great again","venue":null,"work_id":"3ee0ee57-31b9-49d4-98fc-c12c499a14b9","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2507.22058","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:fbcb808ece8e5c76714784303ce0df7439a36c4bab318c1a6fc0b129c3cb95cf","observation_id":"bc1d692f-d031-4224-899e-ba7e78be7c93","resolution":{"observed_at":"2026-05-13T05:17:18.603130Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Geneval: An object-focused framework for evaluating text-to-image alignment","venue":null,"work_id":"686a1b87-4f87-4e33-a945-717bebf9073a","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:90b28975e9555496568b7cabae53ea4c220e513726ad71dd482249d2133fbb7a","observation_id":"a47bf840-3cc3-467e-9360-54c22cff79a4","resolution":{"observed_at":"2026-05-13T10:52:39.444983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Past-future scheduler for llm serving under sla guarantees","venue":null,"work_id":"7ba69db9-75f6-4211-8134-c8f80a4a4f77","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:716d857f910bf8168fa943dc176862efc718008ab864fca4d8bb5d70cfa517f6","observation_id":"2e82c0ea-db84-4a1c-9282-a6834c7b83ab","resolution":{"observed_at":"2026-05-13T10:52:39.443118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Imagen 4 model card","venue":null,"work_id":"5a79cbc9-51b0-47dd-8648-5aed42f6b815","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:f56c5f7216fb9521354c7fd5fed9ea48f03ef1e009c7693407bf15cdd9cb1c21","observation_id":"d7772b3d-7277-40a4-a837-a5e86e59c660","resolution":{"observed_at":"2026-05-13T10:52:39.446664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nano banana 2: Combining pro capabilities with lightning-fast speed","venue":null,"work_id":"9ce159a0-dcd6-464e-bcf1-39d3f25fd160","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:f3c5bb7aad1e93ae610609cb8175d3ac8cc2d53df4b4cb8081932ccf13384bf4","observation_id":"936dab28-9af6-49cb-8a69-3912342d1ca4","resolution":{"observed_at":"2026-05-13T10:52:39.450217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini 3 Pro Model Card, November 2025","venue":null,"work_id":"5dc28fb6-2027-49de-a8a0-4bc57f332cd9","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:1eae05289df294909d7ea4a7501538fc41a22bed603e2e57ebdf04fecc37f0a6","observation_id":"9938e232-8458-427a-a576-07bb5f55258e","resolution":{"observed_at":"2026-05-13T10:52:39.459739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.27492","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:37:06.844147Z","title":"Thinkmorph: Emergent properties in multimodal interleaved chain-of-thought reasoning","venue":null,"work_id":"03da248b-c710-48de-9cdc-1345fcfa31f4","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:1f54282bf5bc57d2ec3fee13c876c4efcd238b0243f0a680a452504df5f08453","observation_id":"c5f5654f-111c-48d0-8e4e-1f7de304f871","resolution":{"observed_at":"2026-05-13T05:17:18.483141Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hallusionbench: an advanced diagnostic suite for entangled language hallucination and visual illusion in large vision-language models","venue":null,"work_id":"8b982293-4eeb-49a9-82b5-e7df903738fd","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:eddf65e273c700bace4f9c0616842c966c572fe0967464dfd7cb9ff53851df9a","observation_id":"e8470115-ad17-40d5-99a7-4f74c984021f","resolution":{"observed_at":"2026-05-13T10:52:39.458016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Infinity: Scaling bitwise autoregressive modeling for high-resolution image synthesis","venue":null,"work_id":"1f96ffad-723b-4104-a013-d85a3996daf5","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:814e5c33044322b2d533d03cde398a72179645ee635858fd92d716115d26d556","observation_id":"b8ca05a5-5a27-4210-b54b-344b35c956ee","resolution":{"observed_at":"2026-05-13T10:52:39.435630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ai2d-rst","venue":null,"work_id":"6d296643-1b1c-46d9-b2e4-0234e078ba6f","year":2021},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:e119b2ad6239314ef886e43c4885ba3f15f2266771784e864c63fab014616975","observation_id":"9289e161-1dd6-49ca-b077-6e72100fcac2","resolution":{"observed_at":"2026-05-13T10:52:39.437628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01006","last_updated":"2026-01-01T13:07:25Z","snapshot_observed_at":"2026-08-03T18:50:30.558321Z","submitted_at":"2025-07-01T17:55:04Z","title":"GLM-4.5V and GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","version":6},"cited_work":{"arxiv_id":"2507.01006","doi":"10.48550/arxiv.2507.01006","metadata_source":"pith","pith_arxiv_id":"2507.01006","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"GLM-4.5V and GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","venue":"cs.CV","work_id":"366607ba-e4ea-4726-98c3-63356e32351c","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2507.01006","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:9abff6aa249c59d65e4c04c65fff7c4d96e3b01ce9de03b05f9bf2dfabd6a9e7","observation_id":"910d930d-9505-4e90-aa85-34bfabe303fc","resolution":{"observed_at":"2026-05-13T05:17:18.674337Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05135","last_updated":"2024-03-08T08:08:10Z","snapshot_observed_at":"2026-08-02T05:01:21.775493Z","submitted_at":"2024-03-08T08:08:10Z","title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","version":1},"cited_work":{"arxiv_id":"2403.05135","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.05135","snapshot_observed_at":"2026-07-04T16:59:58.040795Z","title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","venue":"cs.CV","work_id":"94248955-4bc5-4517-98a0-66224a36d865","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2403.05135","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:ac75f6dd2564deed3e258e352b921e56e496e0d3a4094b6302d24a8cf59109bb","observation_id":"fb3583f9-1a7b-4a95-a691-e7de0c5e2c8d","resolution":{"observed_at":"2026-05-13T05:17:18.760218Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01934","last_updated":"2025-04-03T16:43:14Z","snapshot_observed_at":"2026-07-06T21:03:11.554376Z","submitted_at":"2025-04-02T17:45:00Z","title":"ILLUME+: Illuminating Unified MLLM with Dual Visual Tokenization and Diffusion Refinement","version":2},"cited_work":{"arxiv_id":"2504.01934","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.01934","snapshot_observed_at":"2026-07-04T16:09:57.571267Z","title":"Illume+: Illuminating unified mllm with dual visual tokenization and diffusion refinement","venue":null,"work_id":"f4320ca7-31dd-4c56-9eba-d0556b06956a","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2504.01934","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:bd09e42bdf35f407ccb5e3176f118886684d39697cdf59ec40a2215a35420da2","observation_id":"f84d59dc-7e92-4d70-942f-7d3ea52f80a0","resolution":{"observed_at":"2026-05-13T05:17:18.739464Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"C-eval: A multi-level multi-discipline chinese evaluation suite for foundation models","venue":null,"work_id":"18ff8c0f-dc6c-4f6e-be42-1dc0c172f21d","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:c9b83336da50d3dc5d94dcbb9ebe365cd4ba06260b178a05f0f6c3fc7936e63a","observation_id":"e5be6e8d-3f39-485d-bae3-06b5f27da7cd","resolution":{"observed_at":"2026-05-13T10:52:39.439459Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:ba29f2a2a497c64ea6e921c012b7696c90fa4a7a616bc8d3926c66314424ba4f","observation_id":"e522aa40-2147-47da-9893-ae08d904c8ad","resolution":{"observed_at":"2026-05-13T05:17:18.756274Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20534","last_updated":"2026-02-03T04:57:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-28T05:35:43Z","title":"Kimi K2: Open Agentic Intelligence","version":2},"cited_work":{"arxiv_id":"2507.20534","doi":"10.1145/3448609","metadata_source":"pith","pith_arxiv_id":"2507.20534","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Kimi K2: Open Agentic Intelligence","venue":"cs.LG","work_id":"7f18284c-12d3-4137-bea1-1da97e8cf3c1","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2507.20534","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:57e3dd9299a1d80ff375d701581120c34c589c0552fab0d377db7fa5a22715e1","observation_id":"6cb9f8bd-c4b9-4930-b335-80cc611e1ac8","resolution":{"observed_at":"2026-05-13T05:17:18.735424Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.6114","last_updated":"2022-12-10T21:04:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2013-12-20T20:58:10Z","title":"Auto-Encoding Variational Bayes","version":11},"cited_work":{"arxiv_id":"1312.6114","doi":"10.2139/ssrn.4269703","metadata_source":"pith","pith_arxiv_id":"1312.6114","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Auto-Encoding Variational Bayes","venue":"stat.ML","work_id":"97d95295-30e1-42b4-bbf6-85f0fa4edb44","year":2013},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/1312.6114","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:8db5f1d1ba3c387038c14085c884a711039d6a027ee9d8b75fd487541f1d4816","observation_id":"6ad436d7-7d8b-4bf4-9533-f7168ad9429d","resolution":{"observed_at":"2026-05-13T05:17:18.532228Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Kolors 2.0","venue":null,"work_id":"c95bcdef-b373-406f-bbb5-1079a4d72f29","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:5a9b69098a9d5c843f6b7cedf9d5b037703dfa05b896403171f51222e7a5a74d","observation_id":"8b786f63-54fe-4772-ac95-5d593950379b","resolution":{"observed_at":"2026-05-13T10:52:39.502836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flux","venue":null,"work_id":"13af8a4c-5373-42c2-a84f-80390fcdeced","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:286b1939f4a5d9a6328abb1cb8c90ed0cb3008a1b261d4a90b7290cdc0aa2d56","observation_id":"98617361-80fd-4e51-a325-9850284f742c","resolution":{"observed_at":"2026-05-13T10:52:39.538644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FLUX.2: Frontier Visual Intelligence","venue":null,"work_id":"5c7ee218-9b40-4210-808f-2ec049983acf","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:b9274bb05d452db6d0eb3d471409dd69b30907e3dc9c5b503587abc7dfd5a730","observation_id":"36c687be-eb70-4217-9214-1d4cde96c3b3","resolution":{"observed_at":"2026-05-13T10:52:39.545613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15742","last_updated":"2025-06-24T05:31:03Z","snapshot_observed_at":"2026-07-06T21:44:22.901650Z","submitted_at":"2025-06-17T20:18:23Z","title":"FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space","version":2},"cited_work":{"arxiv_id":"2506.15742","doi":"10.48550/arxiv.2506.15742","metadata_source":"pith","pith_arxiv_id":"2506.15742","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space","venue":"cs.GR","work_id":"5dfe19d5-3541-4803-8fe9-3c8b9e29b281","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2506.15742","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:b8e36d3d3ceab9dea7b2d76acdbfa587407af38eb11f50129a811eed1b7c1680","observation_id":"315466c3-f697-4cb2-ac1f-b10cafd405c9","resolution":{"observed_at":"2026-05-13T05:17:18.526215Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-25T00:53:20.581238+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T00:53:20.581238+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The scalability of simplicity: Empirical analysis of vision-language learning with a single transformer","venue":null,"work_id":"2862dace-b428-4102-9642-88d35751c6df","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:2555de202e94bac821aa0d6af1971b84efae67fb2b1b1e3cb268f8725f9ab9cf","observation_id":"1d606c1f-2b99-4180-9835-63c350a1e62f","resolution":{"observed_at":"2026-05-13T10:52:39.454021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T18:04:01.061356Z","title":"Repa-e: Unlocking vae for end-to-end tuning of latent diffusion transformers","venue":null,"work_id":"fbd3049c-8814-4ae7-9c1d-e35559ab065e","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:09c64db762fcc5ba45dde0966ea80ed3cf844a2324b92dd21dcd2e971f07dd1d","observation_id":"32e3dd20-cf57-4c66-8626-991e50ba65f3","resolution":{"observed_at":"2026-05-13T10:52:39.456131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.21500","doi":"10.48550/arxiv.2505.21500","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Viewspatial-bench: Evaluating multi-perspective spatial localization in vision-language models.ArXiv, abs/2505.21500","venue":null,"work_id":"4e64344e-47c6-4d04-a5a4-c05266da7d8c","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:5d628fd49302b340b6c76bbe759b7fb05b40a153828277711a73910471fc78de","observation_id":"f2c28161-d97e-4b6e-b4d3-d2b20cbaa561","resolution":{"observed_at":"2026-05-13T05:17:18.537844Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.03498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T14:28:32.074454Z","title":"Onecat: Decoder-only auto-regressive model for unified understanding and generation","venue":null,"work_id":"b201beef-0544-4d04-9644-65dfeacb78bd","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:aa490db370a82b49634bfbb732add6f01bd0346b23cf957a553800027f13bc14","observation_id":"30a5e343-3429-4d2f-bce0-7645e513a7e0","resolution":{"observed_at":"2026-05-13T05:17:18.486368Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T14:43:52.726323Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"6b48634f-b23f-47f9-83bb-e0654c3523d9","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:b50114bb1c6af1ec6637244a6e0d2911648a60bbe505ca6b4ae5154e8ce4db48","observation_id":"1b8e3371-fe0d-4001-9ceb-217e5649f6e9","resolution":{"observed_at":"2026-05-13T10:52:39.433697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.01833","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T02:06:26.917131Z","title":"Tir-bench: A comprehensive benchmark for agentic thinking-with-images reasoning","venue":null,"work_id":"1aef68b6-0ae5-4140-820c-d1c9d9138699","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:86b17a588a7138f2e7f26b44810adec130127a74c6795e9ea2b8d5c4ddd214f3","observation_id":"88daa041-5906-4682-9b74-4d897a824f49","resolution":{"observed_at":"2026-05-13T05:17:18.732293Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.13720","last_updated":"2026-01-07T05:36:57Z","snapshot_observed_at":"2026-07-06T22:36:08.495952Z","submitted_at":"2025-11-17T18:59:57Z","title":"Back to Basics: Let Denoising Generative Models Denoise","version":2},"cited_work":{"arxiv_id":"2511.13720","doi":"10.48550/arxiv.2511.13720","metadata_source":"pith","pith_arxiv_id":"2511.13720","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Back to Basics: Let Denoising Generative Models Denoise","venue":"cs.CV","work_id":"37973de8-a5e6-4d92-897b-a98fa9f7f2f3","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2511.13720","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:021ab20ffe13283427df039502c167d4de4a4b40f2f85866e1a9bd6d1393234d","observation_id":"8a364973-830e-4f19-974a-a05d4a268d0e","resolution":{"observed_at":"2026-05-13T05:17:18.800887Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Breen: bridge data-efficient encoder-free multimodal learning with learnable queries","venue":null,"work_id":"dfc81167-ace5-4775-874a-8c2d3934d6a2","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:e20c39760ca816c62c47bc558034a6d82dd9c43bde448a70cc29db48bd1b600e","observation_id":"7ca87196-2182-475a-9d2d-bdf09c9d34f7","resolution":{"observed_at":"2026-05-13T10:52:39.441282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.25732","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T11:46:55.817071Z","title":"Bizgeneval: A systematic benchmark for commercial visual content generation","venue":null,"work_id":"1cb8692b-ed50-490b-a003-110689c0b1d7","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:6c13686ad848061c572e14f432a5b2811beaeae3749e7cc1082702e4cf4e7f34","observation_id":"ebd8ae8d-a66f-45c3-bcc2-85f99d1c488d","resolution":{"observed_at":"2026-05-13T05:17:18.492515Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.16888","last_updated":"2025-11-04T13:15:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-19T15:38:06Z","title":"Uniworld-V2: Reinforce Image Editing with Diffusion Negative-aware Finetuning and MLLM Implicit Feedback","version":3},"cited_work":{"arxiv_id":"2510.16888","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.16888","snapshot_observed_at":"2026-07-05T16:51:14.250864Z","title":"Uniworld-V2: Reinforce Image Editing with Diffusion Negative-aware Finetuning and MLLM Implicit Feedback","venue":"cs.CV","work_id":"9687bd6c-c4f6-4f49-ba59-ca7e825f0710","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2510.16888","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:597c2c90043750165a12316cbe9076252a6a8f4e65f388013bac980dba8fadcd","observation_id":"7de6334b-e280-48a8-8b50-ae5cdd89c236","resolution":{"observed_at":"2026-05-21T18:01:19.940440Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04996","last_updated":"2025-05-08T01:53:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T18:59:06Z","title":"Mixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models","version":2},"cited_work":{"arxiv_id":"2411.04996","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.04996","snapshot_observed_at":"2026-07-04T11:59:50.940046Z","title":"Mixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models","venue":"cs.CL","work_id":"82eb1f7e-d598-409b-9fec-8a7e82965d26","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2411.04996","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:252ea400dba8bb66cfa4b964948e224f07d32aef13ec68327230caf35a1dcc7a","observation_id":"8490c6c8-47a4-49d3-baea-426d2faccba8","resolution":{"observed_at":"2026-05-18T02:48:45.158778Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05472","last_updated":"2025-05-11T18:47:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-08T17:58:57Z","title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","version":2},"cited_work":{"arxiv_id":"2505.05472","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.05472","snapshot_observed_at":"2026-07-10T07:36:57.986858Z","title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","venue":"cs.CV","work_id":"f2badd0e-c06a-45f9-9d9e-7ceda62176b8","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2505.05472","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:cd95b91e09c17371e80bef0a42408e966938b685dea4786792e8026bc0969289","observation_id":"549d4ce2-198c-4882-9690-89277b67842b","resolution":{"observed_at":"2026-05-17T07:24:05.047503Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03147","last_updated":"2025-06-18T18:00:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-03T17:59:33Z","title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","version":4},"cited_work":{"arxiv_id":"2506.03147","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.03147","snapshot_observed_at":"2026-07-05T16:51:14.197597Z","title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","venue":"cs.CV","work_id":"488a273e-95d8-46f1-87c7-2244068d00d0","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2506.03147","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:385b99b77fa9de5267d800fc3421f20ec18503ffd69fd13903e897f420782587","observation_id":"e902012f-d892-42c9-95a1-1a5d2f5a713a","resolution":{"observed_at":"2026-05-13T05:17:18.517445Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05422","last_updated":"2025-08-15T08:56:27Z","snapshot_observed_at":"2026-07-06T21:21:01.070459Z","submitted_at":"2025-05-08T17:12:19Z","title":"TokLIP: Marry Visual Tokens to CLIP for Multimodal Comprehension and Generation","version":2},"cited_work":{"arxiv_id":"2505.05422","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.05422","snapshot_observed_at":"2026-07-04T16:39:58.252935Z","title":"arXiv preprint arXiv:2505.05422 (2025) 2, 4, 7, 9, 1","venue":null,"work_id":"0a372e74-f91a-42a7-b9c6-ec5dde3054f0","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2505.05422","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:171f822a03de581098ccf2a5aacbe363545204c3da8d29cb915fcc91c5cab268","observation_id":"3a157326-a4d9-4ee4-a2bd-2f5483acdd24","resolution":{"observed_at":"2026-05-13T05:17:18.721952Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"c3f2f893-e737-46e6-b88f-9bbcb2f21249","year":2014},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:bfe345fba5b3149c8bef68ecb3ff67a56026fc8c1732eb7f70532feea6d5d2ce","observation_id":"342bab0a-83bf-49d1-8305-6ecda8da66b5","resolution":{"observed_at":"2026-05-13T10:52:39.463562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21770","last_updated":"2024-08-12T16:20:37Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:46:51Z","title":"MoMa: Efficient Early-Fusion Pre-training with Mixture of Modality-Aware Experts","version":3},"cited_work":{"arxiv_id":"2407.21770","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.21770","snapshot_observed_at":"2026-07-04T19:20:06.471468Z","title":"Moma: Efficientearly-fusionpre-trainingwithmixtureofmodality-awareexperts","venue":null,"work_id":"70d7bf05-49be-45c0-b6e2-e2b66c6bb0b0","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2407.21770","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:4a4d21c63672fd018e72b67aca9f49e3a5b8b7aa22c7e40f5c4abf07e2c93875","observation_id":"57e87d1c-9fd2-41f0-bc2f-3bfdd81001db","resolution":{"observed_at":"2026-05-13T05:17:18.708488Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual instruction tuning","venue":null,"work_id":"20df813a-8d97-4d94-869e-4c051e8cfef8","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:c46124fda9c8d8719085079a53f7b2f9458672ac962df817677a70920a4b2ece","observation_id":"252c684e-3c08-469c-8dce-64d8fe1883f8","resolution":{"observed_at":"2026-05-13T10:52:39.429665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05470","last_updated":"2025-10-27T09:57:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-08T17:58:45Z","title":"Flow-GRPO: Training Flow Matching Models via Online RL","version":5},"cited_work":{"arxiv_id":"2505.05470","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.05470","snapshot_observed_at":"2026-07-09T02:25:55.898592Z","title":"Flow-GRPO: Training Flow Matching Models via Online RL","venue":"cs.CV","work_id":"bf1e8e81-ff31-401a-a5dc-d9c49df168ab","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2505.05470","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:e901688ade9ba974c467e3e1dfd4659a69d741213d11c6b80a4fcd44403b370d","observation_id":"8bd0a807-3759-4777-a36f-8df649eea968","resolution":{"observed_at":"2026-05-13T05:17:18.804504Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17761","last_updated":"2025-07-31T05:05:45Z","snapshot_observed_at":"2026-07-06T21:14:24.007428Z","submitted_at":"2025-04-24T17:25:12Z","title":"Step1X-Edit: A Practical Framework for General Image Editing","version":5},"cited_work":{"arxiv_id":"2504.17761","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.17761","snapshot_observed_at":"2026-07-10T01:46:41.223286Z","title":"Step1X-Edit: A Practical Framework for General Image Editing","venue":"cs.CV","work_id":"3392f2c8-a1cb-4d6c-8c82-2cdccffa33f9","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2504.17761","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:1d75dc4afa37905922bb538106c62a024affa6a22f374926071d016f21f12842","observation_id":"1dfcc8d3-cc39-4dd3-951c-269755acd279","resolution":{"observed_at":"2026-05-13T05:17:18.643765Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mmbench: Is your multi-modal model an all-around player? In Proceedings of the European Conference on Computer Vision, pages 216–233","venue":null,"work_id":"5ee6fbe1-8fba-4542-b3ce-827764860ea7","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:e99909f2b42aca6e7a60519ba121d2c00413b6d17b3c31453eb13c928bc517cb","observation_id":"896535d9-5dd5-463c-a14b-e108a59cfa43","resolution":{"observed_at":"2026-05-13T10:52:39.421664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ocrbench: on the hidden mystery of ocr in large multimodal models","venue":null,"work_id":"d3fccacd-f00a-45b3-8a16-5fdc1766dcf7","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:d8e9fc9c59f9971d783ca86753fcb203cf4a61a82d96bc86f4179dfaadf69fbe","observation_id":"9967e967-951c-48c3-932c-ed9d649f6ad0","resolution":{"observed_at":"2026-05-13T10:52:39.425818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.02014","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:37:06.868333Z","title":"Tuna: Taming unified visual representations for native unified multimodal models","venue":null,"work_id":"caa100a4-03f6-4ad6-8461-4d3a89233486","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:4beec09694b415754977d8e256e24b1a3b0c1c136fd7b4884118b59e7bbf8a85","observation_id":"50b46e35-daa3-462c-8d04-c1f577a878e2","resolution":{"observed_at":"2026-05-13T05:17:18.567494Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24763","last_updated":"2026-05-18T04:20:18Z","snapshot_observed_at":"2026-07-06T23:10:42.926677Z","submitted_at":"2026-04-27T17:59:56Z","title":"Tuna-2: Pixel Embeddings Beat Vision Encoders for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2604.24763","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.24763","snapshot_observed_at":"2026-07-10T13:37:06.881254Z","title":"Tuna-2: Pixel Embeddings Beat Vision Encoders for Multimodal Understanding and Generation","venue":"cs.CV","work_id":"6d1815d2-46f1-4217-a874-f25ae24be85c","year":2026},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2604.24763","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:38c64961d3107202b01195bd5233bc30b59c00b16c822cf873acede6c5cf0ac9","observation_id":"7e87a0de-826d-44e1-b9fd-4df531930344","resolution":{"observed_at":"2026-05-13T05:17:18.656071Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Stable diffusion 3: Research paper","venue":null,"work_id":"5d4cc6e7-d92e-48ac-a08b-99e2a235be91","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:446fcc95bf327668cd666071674797ceacbfefbecba9b2b40eeef4a6b6053f6d","observation_id":"1d2cdf16-c3a8-4b6a-a9d3-58525c4974ee","resolution":{"observed_at":"2026-05-13T10:52:39.431536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02255","last_updated":"2024-01-21T03:47:06Z","snapshot_observed_at":"2026-07-06T16:27:15.027202Z","submitted_at":"2023-10-03T17:57:24Z","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","version":3},"cited_work":{"arxiv_id":"2310.02255","doi":"10.1109/cvpr52734.2025.01245","metadata_source":"pith","pith_arxiv_id":"2310.02255","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","venue":"cs.CV","work_id":"e22c3789-9e71-4242-b6ea-3e60e06e2b66","year":2023},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2310.02255","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:4dd3c904d876baa3acd2930a5a74b3cc2b626be460f17bd5a120c17f83bca866","observation_id":"41bbb502-84f4-4c01-95dd-807771edde84","resolution":{"observed_at":"2026-05-13T05:17:18.764085Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.12566","last_updated":"2025-07-16T18:31:23Z","snapshot_observed_at":"2026-07-06T21:58:18.473542Z","submitted_at":"2025-07-16T18:31:23Z","title":"Mono-InternVL-1.5: Towards Cheaper and Faster Monolithic Multimodal Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.12566","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.12566","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mono-internvl-1.5: Towards cheaper and faster monolithic multimodal large language models","venue":null,"work_id":"c454767e-76df-46b9-af2d-e54cf575b057","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2507.12566","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:6187fcb037ca8538a3a943b7012800df951521426ba88e04bfdd662042de9619","observation_id":"85ce1315-55db-4859-9f18-b28e63813223","resolution":{"observed_at":"2026-05-13T05:17:18.752838Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mono-internvl: Pushing the boundaries of monolithic multimodal large language models with endogenous visual pre-training","venue":null,"work_id":"24320ea1-953e-42e3-b781-fc6a44112309","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:38d2fd4ba0648d79d06db51f30672866067c0a339bbfe8918b467d221bf54bb2","observation_id":"b7113530-4fdf-4d9f-b603-0d3e3ad84bdc","resolution":{"observed_at":"2026-05-13T10:52:39.417939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.20321","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T13:09:50.827602Z","title":"Unitok: A unified tokenizer for visual generation and understanding.arXiv preprint arXiv:2502.20321, 2025a","venue":null,"work_id":"31d5c278-398d-4fee-8130-a864fb6b3717","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:b670250c43cf94cb230d76a9079f55ac2ffff87667ec1e2c1a92459fdc169f97","observation_id":"99796e2b-f2c4-491c-93be-267ba69516dc","resolution":{"observed_at":"2026-05-13T05:17:18.678621Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2412.07825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T03:29:31.552020Z","title":"3dsrbench: A comprehensive 3d spatial reasoning benchmark","venue":null,"work_id":"3d100652-7ff3-4c05-a9cb-9f7383953f74","year":2024},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:38570fac7156b5f0e4234386522f60d608f20278ec2f2eefff35ffaf745dc141","observation_id":"2af374c7-54e3-4a15-81a0-0d793229f689","resolution":{"observed_at":"2026-05-13T05:17:18.609205Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Janusflow: Harmonizing autoregression and rectified flow for unified multimodal understanding and generation","venue":null,"work_id":"393b2f4f-3dcd-421e-b9db-f49d0f907625","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:91fa47cdd8f3cd49e0f127011fb294cd2495519df7bad0dbfcf63c0a1545b660","observation_id":"5aba762f-585f-4f07-8ff7-2eb50c825c9d","resolution":{"observed_at":"2026-05-13T10:52:39.419733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T03:05:55.721547Z","title":"Hpsv3: Towards wide-spectrum human preference score","venue":null,"work_id":"625faa9b-3cb8-4c7f-b9ca-37a9ed6daaee","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:a39c04acc0f44fb33d784e7eb703c44f7e7096c1091c9478a0289089e0e3c2b3","observation_id":"05195b58-2b82-49fc-92f2-6e959aa97f48","resolution":{"observed_at":"2026-05-13T10:52:39.427730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T15:02:40.742244Z","title":"Infographicvqa","venue":null,"work_id":"db9d73a2-78d6-4aeb-b480-16bde495dad9","year":2022},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:41a7cd0edf17feba8673a143170b0f43404a022651c02dc5bd5f8e3b79856f6c","observation_id":"90d7109a-3a0c-450c-afbe-acb5f96e6381","resolution":{"observed_at":"2026-05-13T10:52:39.467321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Midjourney v7","venue":null,"work_id":"98e6acc5-aeaa-45a8-9c1d-6cad35d7f8b3","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:bb0f9e3572f04d99a5c0d937a667238be347d765d2ee79153051c20b354d513e","observation_id":"30fe1491-7edd-4187-a310-0a6563afafb4","resolution":{"observed_at":"2026-05-13T10:52:39.540340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Image-01","venue":null,"work_id":"edecb9ba-dfaa-4e4b-86fa-ac2632d8b659","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:415233bc0cc9a2424397ce42767e8d2e19c41c08d9e4b9ae36c181bd7f6d62e5","observation_id":"08fc4245-b99a-4b20-8e68-1840c915859a","resolution":{"observed_at":"2026-05-13T10:52:39.523988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LightLLM: A python-based llm inference and serving framework","venue":null,"work_id":"0643f9b7-53a9-445f-a68b-003343fa74d7","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:659cd11afaec2e1d6bb6fba44f9183927d655d8eb3efd2df016648f2f2dd004f","observation_id":"4fcd9f5c-e310-4b12-a4e7-f9d7fed13764","resolution":{"observed_at":"2026-05-13T10:52:39.508356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LightX2V: A lightweight video and image generation inference framework","venue":null,"work_id":"1b1378cd-cfa5-4446-a4f8-8cab5ce61325","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:8e57f14439d0f7130da632bb9313693022f309cf6ae10d2da4bdea1e38f0a824","observation_id":"59ceda2b-ced0-4a3e-990c-277750e74d3b","resolution":{"observed_at":"2026-05-13T10:52:39.476260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.07265","last_updated":"2026-06-02T17:11:50Z","snapshot_observed_at":"2026-07-06T20:49:51.505079Z","submitted_at":"2025-03-10T12:47:53Z","title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","version":4},"cited_work":{"arxiv_id":"2503.07265","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.07265","snapshot_observed_at":"2026-07-04T13:19:50.710913Z","title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","venue":"cs.CV","work_id":"604c1ae0-368d-49bf-8117-d688ac59f305","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"cited_paper":"/paper/2503.07265","citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:ac635c76b26ef5350f6a1d8a6af499ab6f3e830c5b751b04c8cecc1749252ca9","observation_id":"84c3964e-28d5-42c8-b513-7e1841cf4f37","resolution":{"observed_at":"2026-05-15T16:24:27.819650Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-image-1","venue":null,"work_id":"d968c0fd-0f76-49f7-a089-653570d76248","year":2025},"citing_paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-05-13T05:12:37.339084Z"},"links":{"citing_paper":"/paper/2605.12500"},"observation_digest":"sha256:799daf3de4a518237854e172e114b4a121642b35fd8b0c6d5f819fe0b575ca79","observation_id":"ea26a3db-4287-4eac-b9ea-8b6e0c69d1fc","resolution":{"observed_at":"2026-05-13T10:52:39.515485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.12500","last_updated":"2026-05-12T17:59:58Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:24:13.851504Z","submitted_at":"2026-05-12T17:59:58Z","title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":1,"verified_exact":51,"verified_fuzzy":44},"total_outbound_references":173},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 100 of 173 outbound references and 15 inbound Pith citation observations for arXiv:2605.12500."}