{"as_of":"2026-08-04T12:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1e28ed017e87de2cb7ef6453e4f1d9f4d240cf4f23fde9fb3d987f84eeedefa4","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T11:29:16.883675Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2501.17811","last_updated":"2025-01-29T18:00:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-29T18:00:19Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-11T08:14:52.890145Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2501.17811"},"observation_digest":"sha256:e6e5aa1857e200ca90d658adcce84fe4a1abf4f1c3a8df0ff948ae63119c22da","observation_id":"5b9e76d7-d684-4499-8d69-a1284bc0d394","resolution":{"observed_at":"2026-05-11T08:14:53.158653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2503.14324","last_updated":"2026-04-20T17:55:12Z","snapshot_observed_at":"2026-08-02T12:34:37.994590Z","submitted_at":"2025-03-18T14:56:46Z","title":"DualToken: Towards Unifying Visual Understanding and Generation with Dual Visual Vocabularies","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-22T23:51:43.934329Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2503.14324"},"observation_digest":"sha256:d6cce708495c25c3a320d63ec1432df498c48ba60d8ad9ebea4be849f2a51080","observation_id":"f55f4173-db02-4d93-a5e4-96030d315cea","resolution":{"observed_at":"2026-05-22T23:52:16.721256Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2505.05472","last_updated":"2025-05-11T18:47:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-08T17:58:57Z","title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-17T07:24:04.460276Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2505.05472"},"observation_digest":"sha256:f442de87dfcd79843f4085cdff527a7935c5ec997ada3dfe8bce3c6c0a0a9d54","observation_id":"a9e53c24-575a-4fc3-b371-025ed7011778","resolution":{"observed_at":"2026-05-17T07:24:04.603862Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2505.09568","last_updated":"2025-05-14T17:11:07Z","snapshot_observed_at":"2026-07-06T21:23:57.084147Z","submitted_at":"2025-05-14T17:11:07Z","title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-11T23:34:26.878354Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2505.09568"},"observation_digest":"sha256:5106292d0d29b4f3c403df51f1e915847c9c11b943bce9ed528f5177dba41adc","observation_id":"6869a934-ada0-4a3d-a060-0bf823273652","resolution":{"observed_at":"2026-05-11T23:34:27.042984Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2505.14683","last_updated":"2025-07-27T11:45:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T17:59:30Z","title":"Emerging Properties in Unified Multimodal Pretraining","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T16:23:41.854132Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2505.14683"},"observation_digest":"sha256:605d58ababf592f1aa88c00ce53b61c03f45f0686bd45d9902936218bbfddec7","observation_id":"98fed8a3-a44b-49fa-8069-5a61c92b004b","resolution":{"observed_at":"2026-05-10T16:23:42.211623Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2506.15564","last_updated":"2025-09-22T01:24:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-18T15:39:15Z","title":"Show-o2: Improved Native Unified Multimodal Models","version":3},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-12T18:51:15.428692Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2506.15564"},"observation_digest":"sha256:423017a24a729f9326539d31dc3bd70289d62211732d72dc33bac91a42bfb118","observation_id":"fb7f0dab-7b70-4b83-b7d9-fbfcbae00a30","resolution":{"observed_at":"2026-05-12T18:51:15.542475Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2601.15507","last_updated":"2026-05-08T04:00:14Z","snapshot_observed_at":"2026-07-06T22:42:36.961014Z","submitted_at":"2026-01-21T22:29:33Z","title":"A Unified and Controllable Framework for Layered Image Generation with Visual Effects","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-16T11:54:46.989748Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2601.15507"},"observation_digest":"sha256:894a6b7c1d0b0dbe39da5d661d4da762b6d8692eb7501bb3d0eca15161640318","observation_id":"b88afb4d-7579-4fad-ae74-0661592f29e0","resolution":{"observed_at":"2026-05-16T11:57:50.252224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2604.03893","last_updated":"2026-06-01T03:09:36Z","snapshot_observed_at":"2026-07-13T12:10:48.358603Z","submitted_at":"2026-04-04T23:18:58Z","title":"FeynmanBench: Benchmarking Multimodal LLMs on Diagrammatic Physics Reasoning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T16:51:48.705876Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2604.03893"},"observation_digest":"sha256:359d4ed59c239a7106dd46367a3c9fd58345139eb57f5780b0d92ba95232c651","observation_id":"609a9b36-99f1-4287-a95c-8be0c11ae5cb","resolution":{"observed_at":"2026-05-13T16:52:59.470222Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-13T12:10:53.720348Z","title":"Du, Zehuan Yuan, and Xinglong Wu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.03893","last_updated":"2026-06-01T03:09:36Z","snapshot_observed_at":"2026-07-13T12:10:48.358603Z","submitted_at":"2026-04-04T23:18:58Z","title":"FeynmanBench: Benchmarking Multimodal LLMs on Diagrammatic Physics Reasoning","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-13T12:10:53.720348Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2604.03893"},"observation_digest":"sha256:d65c61591316df2d63e9875ab944251a49736ef31f2e081928224c16e4070639","observation_id":"f7bec7b5-3764-4502-9b7e-e1511547bb45","resolution":{"observed_at":"2026-07-13T12:10:53.720348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2604.04746","last_updated":"2026-04-08T01:34:51Z","snapshot_observed_at":"2026-07-06T22:53:37.357996Z","submitted_at":"2026-04-06T15:11:57Z","title":"Think in Strokes, Not Pixels: Process-Driven Image Generation via Interleaved Reasoning","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T19:16:58.323955Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2604.04746"},"observation_digest":"sha256:1335bfffa13fd0bf370f938e95ea8a75cbc58ebf95d12e7a7dd8d76ddb7572b2","observation_id":"211c994b-f3f4-45f2-a9ff-72e6f011aac3","resolution":{"observed_at":"2026-05-10T23:10:53.714152Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:5d00a941ddea8cfd2c888208b9a1d1ebc9a24495c4adb9574bc5156cb2fb72cc","observation_id":"85b178fd-4ad0-47f0-974e-080507de5924","resolution":{"observed_at":"2026-05-10T14:15:28.979944Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2604.24625","last_updated":"2026-04-27T15:52:48Z","snapshot_observed_at":"2026-07-06T23:10:38.548416Z","submitted_at":"2026-04-27T15:52:48Z","title":"Meta-CoT: Enhancing Granularity and Generalization in Image Editing","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-08T04:30:28.636915Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2604.24625"},"observation_digest":"sha256:c141a4017760c61801217ce67745aaac31c32e0761f5fbe591514d22b7e3680d","observation_id":"fec4d6af-7265-4a0e-a833-957939712b01","resolution":{"observed_at":"2026-05-11T21:41:19.367534Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2604.28185","last_updated":"2026-04-30T17:59:02Z","snapshot_observed_at":"2026-07-06T23:13:29.310140Z","submitted_at":"2026-04-30T17:59:02Z","title":"Visual Generation in the New Era: An Evolution from Atomic Mapping to Agentic World Modeling","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-07T06:38:04.459129Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2604.28185"},"observation_digest":"sha256:c194c64132418e640bdce9b66930d56425b107391eab11b9e6e47c406910899b","observation_id":"226722f2-0983-4fef-883d-c2374909187d","resolution":{"observed_at":"2026-05-12T10:16:29.041921Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2605.07915","last_updated":"2026-05-08T15:52:51Z","snapshot_observed_at":"2026-08-02T05:37:32.685801Z","submitted_at":"2026-05-08T15:52:51Z","title":"What Matters for Diffusion-Friendly Latent Manifold? Prior-Aligned Autoencoders for Latent Diffusion","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-11T01:57:24.033068Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2605.07915"},"observation_digest":"sha256:8fad9dc5ae08e27c9e14323ebb917e1484bbd0972d50aca730db8af7dabba9d0","observation_id":"b7e1ded3-b160-45dd-bb73-6d9cfdeb56cc","resolution":{"observed_at":"2026-05-11T04:05:57.477353Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2605.18390","last_updated":"2026-05-18T13:38:43Z","snapshot_observed_at":"2026-07-06T23:29:14.916114Z","submitted_at":"2026-05-18T13:38:43Z","title":"Vision Foundation Models as Generalist Tokenizers for Image Generation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-20T11:01:24.738195Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2605.18390"},"observation_digest":"sha256:4c4e953921b7d97ee5cc5cb2542230f2669d4cbc5946f7ddd5de7e471b4d503b","observation_id":"ab1bc623-546d-41bc-9581-ddde19cc7ab6","resolution":{"observed_at":"2026-05-20T11:03:13.455436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":null,"work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2606.31683","last_updated":"2026-06-30T13:58:56Z","snapshot_observed_at":"2026-08-01T20:35:23.922416Z","submitted_at":"2026-06-30T13:58:56Z","title":"Histogram-constrained Image Generation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-01T05:51:19.756153Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2606.31683"},"observation_digest":"sha256:8b01a6ea7db3f6fa280dcbe9a898eff3e788e8c10f9fc168b124b4a144f6cc60","observation_id":"179e42ab-e52d-411d-96ca-81da6fff4ca9","resolution":{"observed_at":"2026-07-01T10:05:41.186922Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-08-01T11:29:16.883675Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19895","last_updated":"2026-07-22T08:29:30Z","snapshot_observed_at":"2026-08-03T12:38:59.702401Z","submitted_at":"2026-07-22T08:29:30Z","title":"OSVE: One Step Video Editing with One Step Diffusion Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T11:29:16.883675Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2607.19895"},"observation_digest":"sha256:48d7774e9f44ae77fa91c6eec620f5479ae1df3e4e30bb40c25a2878dee25752","observation_id":"c07c4d84-88bb-4bb0-bf4c-91baa3e94539","resolution":{"observed_at":"2026-08-01T11:29:16.883675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.03069/citation-record","integrity":"/paper/2412.03069/integrity","json":"/paper/2412.03069/citation-record.json","paper":"/paper/2412.03069"},"outbound":[],"paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2412.03069."}