{"as_of":"2026-08-07T17:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a4417a95a8ccd1bed79b231a92670784557f73722fd45a00b0895931518f9a49","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":34,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:34:53.137002Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T16:39:58.175257Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"reference_index":234,"source":"pdf_text","source_observed_at":"2026-05-12T08:40:40.910461Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2503.09567"},"observation_digest":"sha256:17f8045a61adee68867bc8fbcbd2702a8cb48185b954fe388c842be3b35dea64","observation_id":"48f17002-9e99-4caf-acf4-ae4711e6c14c","resolution":{"observed_at":"2026-05-12T08:40:41.448367Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2503.12605","last_updated":"2025-03-23T13:47:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-16T18:39:13Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-15T17:18:52.996467Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2503.12605"},"observation_digest":"sha256:77573bce0f9d4b4270457ac978f17ede543fc8810d60b51d740730e4c5c07a5d","observation_id":"fba5499c-c290-4ca8-a6e6-37db20ab6514","resolution":{"observed_at":"2026-05-15T17:18:53.651274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2505.07818","last_updated":"2025-08-28T17:19:45Z","snapshot_observed_at":"2026-07-31T14:51:03.625964Z","submitted_at":"2025-05-12T17:59:34Z","title":"DanceGRPO: Unleashing GRPO on Visual Generation","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T22:28:24.929046Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.07818"},"observation_digest":"sha256:3c884e550d5adffd00a3ab502a2f0cde2e3774a3addc9466b637127244db3f4f","observation_id":"c6f9f267-9f74-4ea7-9ee7-83052f813b8b","resolution":{"observed_at":"2026-05-11T22:28:26.031397Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T15:34:53.137002Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step.arXiv:2501.13926, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14682","last_updated":"2025-05-20T17:59:26Z","snapshot_observed_at":"2026-08-07T15:27:34.288941Z","submitted_at":"2025-05-20T17:59:26Z","title":"UniGen: Enhanced Training & Test-Time Strategies for Unified Multimodal Understanding and Generation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:53.137002Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.14682"},"observation_digest":"sha256:6e152970ce972f3d098078f6036a4832c5cd59cbc7db1e17b1a96e951124f23c","observation_id":"1fc057b5-f3db-4e60-b51b-77bf28e69907","resolution":{"observed_at":"2026-08-07T15:34:53.137002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T14:55:51.520817Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17017","last_updated":"2025-06-10T13:46:37Z","snapshot_observed_at":"2026-08-07T14:49:20.251890Z","submitted_at":"2025-05-22T17:59:49Z","title":"Delving into RL for Image Generation with CoT: A Study on DPO vs. GRPO","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:55:51.520817Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.17017"},"observation_digest":"sha256:57d2ad6bea7212eedfcba5377e9de134cc70762c1316a7005405e998fd281c98","observation_id":"052310a6-09eb-4896-a7dd-72062cfa63af","resolution":{"observed_at":"2026-08-07T14:55:51.520817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T14:49:03.001071Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17540","last_updated":"2025-05-23T06:44:26Z","snapshot_observed_at":"2026-08-07T14:42:55.028404Z","submitted_at":"2025-05-23T06:44:26Z","title":"RePrompt: Reasoning-Augmented Reprompting for Text-to-Image Generation via Reinforcement Learning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:03.001071Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.17540"},"observation_digest":"sha256:45119aacfcb2c58ab79dc019400ddd2b921aa1425fb679a24d77aa1002257980","observation_id":"d2ea6fd7-5eff-4b53-b889-422edfe5faf5","resolution":{"observed_at":"2026-08-07T14:49:03.001071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T14:31:13.310672Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18536","last_updated":"2025-05-24T06:01:48Z","snapshot_observed_at":"2026-08-07T14:27:39.562174Z","submitted_at":"2025-05-24T06:01:48Z","title":"Reinforcement Fine-Tuning Powers Reasoning Capability of Multimodal Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:31:13.310672Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.18536"},"observation_digest":"sha256:f5aadc39b5e966d71ebe433c24def4108244f088e3e1aab20e7792694122a10b","observation_id":"051c8f17-2752-4dd2-86b8-fe78be8377d2","resolution":{"observed_at":"2026-08-07T14:31:13.310672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T13:14:36.335219Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22407","last_updated":"2025-05-28T14:37:21Z","snapshot_observed_at":"2026-08-07T13:05:30.933128Z","submitted_at":"2025-05-28T14:37:21Z","title":"Self-Reflective Reinforcement Learning for Diffusion-based Image Reasoning Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T13:14:36.335219Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.22407"},"observation_digest":"sha256:6a42cd1443ec1888d844196f7725d2d1b21ba496585b4283d38aec5ddc458f77","observation_id":"5c976fdc-86e1-4356-bcef-5e81b52d8b8e","resolution":{"observed_at":"2026-08-07T13:14:36.335219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T12:53:22.895103Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step.CoRR, abs/2501.13926, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23380","last_updated":"2025-05-29T12:00:15Z","snapshot_observed_at":"2026-08-07T12:44:53.093273Z","submitted_at":"2025-05-29T12:00:15Z","title":"UniRL: Self-Improving Unified Multimodal Models via Supervised and Reinforcement Learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:53:22.895103Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.23380"},"observation_digest":"sha256:e3212b5e829cb4b8a6952760a2410f52c54dbc06919fdcfd451b0342e8d711ee","observation_id":"3e8053ea-4c2d-445e-aebf-e7b2b1ee8a09","resolution":{"observed_at":"2026-08-07T12:53:22.895103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T12:48:51.044539Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23493","last_updated":"2025-05-29T14:43:46Z","snapshot_observed_at":"2026-08-07T12:42:36.882299Z","submitted_at":"2025-05-29T14:43:46Z","title":"R2I-Bench: Benchmarking Reasoning-Driven Text-to-Image Generation","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T12:48:51.044539Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.23493"},"observation_digest":"sha256:727efb411c992fba753ee31cd88ab9b486dbc43de946ae1b2ae8b4dcd48ce095","observation_id":"c30ed327-5349-46e7-b27a-bb7d9de92dad","resolution":{"observed_at":"2026-08-07T12:48:51.044539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T12:45:45.740034Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23693","last_updated":"2025-05-29T17:31:13Z","snapshot_observed_at":"2026-08-07T14:01:56.505702Z","submitted_at":"2025-05-29T17:31:13Z","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:45.740034Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.23693"},"observation_digest":"sha256:2d2f509e9e0088c4ba4e4d2c3a8ec7757de02054be990d79a3ae12ee3faf2985","observation_id":"8a0adae8-f90e-48c0-a83b-bca2917ef604","resolution":{"observed_at":"2026-08-07T12:45:45.740034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T12:18:32.622198Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24787","last_updated":"2025-05-30T16:48:14Z","snapshot_observed_at":"2026-08-07T15:30:21.145991Z","submitted_at":"2025-05-30T16:48:14Z","title":"Draw ALL Your Imagine: A Holistic Benchmark and Agent Framework for Complex Instruction-based Image Generation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:18:32.622198Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.24787"},"observation_digest":"sha256:ff270e4f00c8328198fe01d70ad5c22c1dc52eab212947c25691e13b7cb69c04","observation_id":"b3f0c90b-2bef-45b1-972f-8a68e39d3793","resolution":{"observed_at":"2026-08-07T12:18:32.622198Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T12:19:21.683953Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step.arXiv preprint arXiv:2501.13926, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24875","last_updated":"2025-06-05T17:51:58Z","snapshot_observed_at":"2026-08-07T12:10:14.143340Z","submitted_at":"2025-05-30T17:59:48Z","title":"ReasonGen-R1: CoT for Autoregressive Image generation models through SFT and RL","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:19:21.683953Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2505.24875"},"observation_digest":"sha256:8df4f3237b63ac6aa0edce0494a89e7e079aca576c18a00c48cb5d8ccf27d37e","observation_id":"081cfe8c-5871-4cda-a8be-8a5bfe6693bf","resolution":{"observed_at":"2026-08-07T12:19:21.683953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T11:33:20.782033Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step.arXiv preprint arXiv:2501.13926, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02161","last_updated":"2026-07-12T08:53:18Z","snapshot_observed_at":"2026-08-07T11:27:10.636069Z","submitted_at":"2025-06-02T18:44:07Z","title":"TIIF-Bench: How Does Your T2I Model Follow Your Instructions?","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:33:20.782033Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.02161"},"observation_digest":"sha256:e8bbae5460c16d4b6ddb0c6ee6258b17ae9bf96542c2dc197e74573965613efa","observation_id":"553d2b69-b567-4d94-9e80-6517d6fa6cf0","resolution":{"observed_at":"2026-08-07T11:33:20.782033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2506.03530","last_updated":"2026-05-22T06:55:17Z","snapshot_observed_at":"2026-08-01T13:49:24.957131Z","submitted_at":"2025-06-04T03:22:44Z","title":"How Far Are We from Generating Missing Modalities with Foundation Models?","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-25T08:15:12.947854Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.03530"},"observation_digest":"sha256:7754f79d99ad61936fdc55ae10d7ca895a10c0b4f90e79a5c6087477062e6061","observation_id":"f932e840-318a-45e1-9f6f-875b17f5d0a0","resolution":{"observed_at":"2026-05-25T08:15:33.584502Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T10:28:48.738793Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step.arXiv preprint arXiv:2501.13926, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05331","last_updated":"2025-06-05T17:59:02Z","snapshot_observed_at":"2026-08-07T10:19:01.005683Z","submitted_at":"2025-06-05T17:59:02Z","title":"MINT-CoT: Enabling Interleaved Visual Tokens in Mathematical Chain-of-Thought Reasoning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:48.738793Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.05331"},"observation_digest":"sha256:4fbef39784363d5e0e30443d8a3fd5f7875bef3e841bfc56a79d25324672719d","observation_id":"d4c71618-6fcb-43ad-bb39-3eb3b7ef5313","resolution":{"observed_at":"2026-08-07T10:28:48.738793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T11:22:44.051500Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05384","last_updated":"2025-06-12T16:38:10Z","snapshot_observed_at":"2026-08-07T11:15:19.658520Z","submitted_at":"2025-06-03T10:11:51Z","title":"Q-Ponder: A Unified Training Pipeline for Reasoning-based Visual Quality Assessment","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:22:44.051500Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.05384"},"observation_digest":"sha256:673e873eee3fd1345ef9c5e24b3729edac6b074457cfdc6c7e0488082a3b721f","observation_id":"23dca6e3-ebf9-46de-8494-cb092f44b9b5","resolution":{"observed_at":"2026-08-07T11:22:44.051500Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T10:23:12.511095Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05501","last_updated":"2025-06-05T18:36:33Z","snapshot_observed_at":"2026-08-07T10:17:55.989653Z","submitted_at":"2025-06-05T18:36:33Z","title":"FocusDiff: Advancing Fine-Grained Text-Image Alignment for Autoregressive Visual Generation through RL","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T10:23:12.511095Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.05501"},"observation_digest":"sha256:baa0eaffcf74d4282fdd2664bdfb49f3e162be1ca1de6ef336f8587a844673cc","observation_id":"96b75dea-f716-4ad6-8b6c-2b525eb43572","resolution":{"observed_at":"2026-08-07T10:23:12.511095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-07T00:40:20.308003Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12830","last_updated":"2025-06-15T12:22:55Z","snapshot_observed_at":"2026-08-07T15:29:39.049060Z","submitted_at":"2025-06-15T12:22:55Z","title":"ComplexBench-Edit: Benchmarking Complex Instruction-Driven Image Editing via Compositional Dependencies","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:40:20.308003Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.12830"},"observation_digest":"sha256:f21fa814935efe475a95cb1da1fa95bfcc2edb0e9e01c45d2123441e1a62fd78","observation_id":"294919e4-d9e3-4c01-9070-ad012af279f8","resolution":{"observed_at":"2026-08-07T00:40:20.308003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2506.16796","last_updated":"2026-04-13T02:17:17Z","snapshot_observed_at":"2026-07-30T08:45:44.946546Z","submitted_at":"2025-06-20T07:21:21Z","title":"RealSR-R1: Reinforcement Learning for Real-World Image Super-Resolution with Vision-Language Chain-of-Thought","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T08:32:20.566798Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.16796"},"observation_digest":"sha256:89b3def377c5f5e4c19807f4b89130846bc9b827443e0762686b77b65abf2664","observation_id":"26c2409b-046a-4e3e-87f2-a4338ad6850b","resolution":{"observed_at":"2026-05-19T08:33:02.188477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2506.18871","last_updated":"2026-04-21T17:32:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-23T17:38:54Z","title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T07:47:34.464711Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2506.18871"},"observation_digest":"sha256:d47e72c55d1667daf44ae0b1164e1fb93cf5505e166576bf21d6148188d2df66","observation_id":"b8fadb8b-3881-4257-9666-e71df3074dc1","resolution":{"observed_at":"2026-05-19T07:52:10.744905Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-06T16:42:42.906185Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12761","last_updated":"2025-07-17T03:33:46Z","snapshot_observed_at":"2026-08-06T16:36:45.445367Z","submitted_at":"2025-07-17T03:33:46Z","title":"Think-Before-Draw: Decomposing Emotion Semantics & Fine-Grained Controllable Expressive Talking Head Generation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T16:42:42.906185Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2507.12761"},"observation_digest":"sha256:07b39fe0dea76f8b504b5b1a7fa1c9d8a0f74f30615abdf6089a3d66737304b4","observation_id":"1255e509-c867-452e-a7d0-a6c4e527b6b2","resolution":{"observed_at":"2026-08-06T16:42:42.906185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-05T22:32:51.804152Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06905","last_updated":"2025-08-26T12:18:14Z","snapshot_observed_at":"2026-08-07T03:28:03.679250Z","submitted_at":"2025-08-09T09:36:21Z","title":"MultiRef: Controllable Image Generation with Multiple Visual References","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T22:32:51.804152Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2508.06905"},"observation_digest":"sha256:53bdb1e75e9b0bf31f576b7aa4069798670d96c915da356e3c4d0ba1015110d1","observation_id":"e73c98f9-504a-442d-99f6-92ef727bdfd2","resolution":{"observed_at":"2026-08-05T22:32:51.804152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-05T20:44:12.432203Z","title":"Can we generate images with cot? let's verify and reinforce image generation step by step","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09987","last_updated":"2025-08-13T17:59:28Z","snapshot_observed_at":"2026-08-06T06:04:10.094208Z","submitted_at":"2025-08-13T17:59:28Z","title":"Echo-4o: Harnessing the Power of GPT-4o Synthetic Images for Improved Image Generation","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T20:44:12.432203Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2508.09987"},"observation_digest":"sha256:a912913996f43e84b2252d9bea8a2e0a6ee28cacf58423828418c742dfb81b5f","observation_id":"1c41b22a-8423-4dd8-9ff9-b14b3a0b11d8","resolution":{"observed_at":"2026-08-05T20:44:12.432203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-04T22:36:07.963130Z","title":"Can we generate images with cot? let's verify and reinforce image generation step by step","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07295","last_updated":"2026-06-25T06:17:40Z","snapshot_observed_at":"2026-08-04T22:36:03.033298Z","submitted_at":"2025-09-08T23:59:32Z","title":"Reconstruction Alignment Improves Unified Multimodal Models","version":4},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-04T22:36:07.963130Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2509.07295"},"observation_digest":"sha256:187f5a3485a1f8c90e93269e5235c13b26d271374e6bf3ef33a864516e023ab9","observation_id":"c7d72a8c-5a3b-47d9-8018-1bc7d72d3564","resolution":{"observed_at":"2026-08-04T22:36:07.963130Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2512.07348","last_updated":"2026-04-28T10:02:14Z","snapshot_observed_at":"2026-08-06T12:32:11.849043Z","submitted_at":"2025-12-08T09:40:11Z","title":"MICo-150K: A Comprehensive Dataset Advancing Multi-Image Composition","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T00:20:58.483350Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2512.07348"},"observation_digest":"sha256:07bfd11d792e0a8d2d46c22f1b485c3f54879896dbe891df2cbccc10487e6b14","observation_id":"23d37186-de7a-4292-9018-bf899153e531","resolution":{"observed_at":"2026-05-17T00:21:23.343091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-13T23:27:11.006580Z","title":"arXiv preprint arXiv:2501.13926 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.16870","last_updated":"2026-07-31T08:55:01Z","snapshot_observed_at":"2026-08-05T23:10:24.488750Z","submitted_at":"2026-03-17T17:59:55Z","title":"Demystifying Video Reasoning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-13T23:27:11.006580Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2603.16870"},"observation_digest":"sha256:2cab191310ce6137c647f218250f4b3345475a94d0ff5e64bdd80ca5b70e9041","observation_id":"d3c4c922-b22d-4177-b0c6-590127eb25af","resolution":{"observed_at":"2026-07-13T23:27:11.006580Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-08-03T02:33:54.197765Z","title":"arXiv preprint arXiv:2501.13926 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.16870","last_updated":"2026-07-31T08:55:01Z","snapshot_observed_at":"2026-08-05T23:10:24.488750Z","submitted_at":"2026-03-17T17:59:55Z","title":"Demystifying Video Reasoning","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T02:33:54.197765Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2603.16870"},"observation_digest":"sha256:c7e5044ee33e2a60cedb23af40dd0fd96d7077a08d175871c3df5df60144d1e1","observation_id":"17c05c7b-cb86-4bb6-a657-19a2bcbeb692","resolution":{"observed_at":"2026-08-03T02:33:54.197765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2604.02355","last_updated":"2026-03-12T12:49:26Z","snapshot_observed_at":"2026-07-06T22:51:44.663367Z","submitted_at":"2026-03-12T12:49:26Z","title":"From Broad Exploration to Stable Synthesis: Entropy-Guided Optimization for Autoregressive Image Generation","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-15T12:50:13.764159Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2604.02355"},"observation_digest":"sha256:6f454350bff2d9150ced8b4c7e446fa3f4ced5d4346cf8cb4fbe87e4a7da29a7","observation_id":"53b29446-6ab5-4120-bea8-1451dcd9a279","resolution":{"observed_at":"2026-05-15T12:50:37.069714Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2606.04264","last_updated":"2026-06-02T22:30:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-02T22:30:46Z","title":"UniCanvas: A Diffusion-base Unified Model for Text-in-Image Joint Generation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T10:23:43.501656Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2606.04264"},"observation_digest":"sha256:cb9d4839f6f6d7c2e45c6974a44ba435444b2b8e0f0602191eaebff85d7e7c4c","observation_id":"aafb7341-2c9b-4aa3-a2c7-b6e392123764","resolution":{"observed_at":"2026-07-02T03:06:29.520773Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2606.08231","last_updated":"2026-06-06T15:39:29Z","snapshot_observed_at":"2026-08-07T16:58:05.550639Z","submitted_at":"2026-06-06T15:39:29Z","title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-27T19:36:57.231932Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2606.08231"},"observation_digest":"sha256:f3f40d3c5f52470b6cd56d9540dca65dcc83579af4608ff14711ed614c9d9db8","observation_id":"1ea53590-1d6f-4963-b40a-5219ead2ee88","resolution":{"observed_at":"2026-07-02T21:37:25.295363Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2606.17888","last_updated":"2026-06-16T13:09:32Z","snapshot_observed_at":"2026-08-04T07:49:41.211792Z","submitted_at":"2026-06-16T13:09:32Z","title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","version":1},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-06-27T01:23:40.564561Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2606.17888"},"observation_digest":"sha256:4840a9df9e05803ed74f09da755366e26c74145456f6d3f7037b2ed63da1a2cd","observation_id":"e823c830-b45d-49d7-92a8-b36f8676a353","resolution":{"observed_at":"2026-07-03T20:18:57.837081Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":"2501.13926","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-04T16:39:58.175257Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step","venue":null,"work_id":"42bc4633-5505-4914-a32c-2e3d11cc6e03","year":2025},"citing_paper":{"arxiv_id":"2606.24849","last_updated":"2026-06-23T17:28:00Z","snapshot_observed_at":"2026-08-03T11:30:36.402988Z","submitted_at":"2026-06-23T17:28:00Z","title":"IV-CoT: Implicit Visual Chain-of-Thought for Structure-Aware Text-to-Image Generation","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-06-26T00:19:49.071495Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2606.24849"},"observation_digest":"sha256:f023bd61151b7b96f2f49aa3e50d27e2a04dd90de4e8afb3ada83f654be43898","observation_id":"fdf1b1a6-3013-4842-8d7f-e210294a46c2","resolution":{"observed_at":"2026-07-04T16:39:58.177090Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13926","snapshot_observed_at":"2026-07-14T01:07:06.861298Z","title":"Can we generate images with cot? let’s verify and reinforce image generation step by step.arXiv preprint arXiv:2501.13926, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.10004","last_updated":"2026-07-10T22:03:01Z","snapshot_observed_at":"2026-08-03T02:26:02.235143Z","submitted_at":"2026-07-10T22:03:01Z","title":"Model Guides You How to Draw: Adaptive Visual Gating for Unified Multimodal Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-14T01:07:06.861298Z"},"links":{"cited_paper":"/paper/2501.13926","citing_paper":"/paper/2607.10004"},"observation_digest":"sha256:c38b347da7534def91a58e006415f1622184e223301864a55e6765070214c08c","observation_id":"1730a604-eb8c-49ea-9e3f-50043610d767","resolution":{"observed_at":"2026-07-14T01:07:06.861298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.13926/citation-record","integrity":"/paper/2501.13926/integrity","json":"/paper/2501.13926/citation-record.json","paper":"/paper/2501.13926"},"outbound":[],"paper":{"arxiv_id":"2501.13926","last_updated":"2025-07-23T16:09:10Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:59:43Z","title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 34 inbound Pith citation observations for arXiv:2501.13926."}