{"as_of":"2026-08-06T12:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:848ba14039baad2a696d5c1cd71f78fec47d125eb8f8e44c5b46450ab4205572","coverage":[{"denominator":18,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T14:15:02.723774Z","state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2604.13540/citation-record","integrity":"/paper/2604.13540/integrity","json":"/paper/2604.13540/citation-record.json","paper":"/paper/2604.13540"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.17811","last_updated":"2025-01-29T18:00:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-29T18:00:19Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","version":1},"cited_work":{"arxiv_id":"2501.17811","doi":"10.48550/arxiv.2501.17811","metadata_source":"pith","pith_arxiv_id":"2501.17811","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","venue":"cs.AI","work_id":"67d9e391-26d1-459e-ab56-07e60511c886","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2501.17811","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:7ffb251428ea67e38a7db8ffca143f9966e06096724ef692341e599496cdbede","observation_id":"24737616-9a47-473b-88bd-6601e0d62702","resolution":{"observed_at":"2026-05-11T08:14:53.293790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.14687","last_updated":"2024-05-20T04:23:45Z","snapshot_observed_at":"2026-08-03T03:59:22.374270Z","submitted_at":"2022-09-29T11:12:27Z","title":"Diffusion Posterior Sampling for General Noisy Inverse Problems","version":4},"cited_work":{"arxiv_id":"2209.14687","doi":"10.1101/2025.01.08","metadata_source":"pith","pith_arxiv_id":"2209.14687","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Diffusion Posterior Sampling for General Noisy Inverse Problems","venue":"stat.ML","work_id":"083ab9fb-05f0-41e1-9628-982017eac344","year":2022},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2209.14687","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:006aba222fe181b2a199f005a1d3694f451b337b8a04cd4075392e3eb4944d69","observation_id":"6cf1abac-1d10-464f-8a97-e7fc89eb6710","resolution":{"observed_at":"2026-05-13T00:54:40.489660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14683","last_updated":"2025-07-27T11:45:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T17:59:30Z","title":"Emerging Properties in Unified Multimodal Pretraining","version":3},"cited_work":{"arxiv_id":"2505.14683","doi":"10.48550/arxiv.2505.14683","metadata_source":"pith","pith_arxiv_id":"2505.14683","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emerging Properties in Unified Multimodal Pretraining","venue":"cs.CV","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2505.14683","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:738c71fcf3cb34ed1181bd39a3dbbf4f206fc2931e1af5314bd10e0447e1eace","observation_id":"5ecaf5bd-edb1-457d-90b4-bf7544875243","resolution":{"observed_at":"2026-05-10T16:23:42.412235Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11513","last_updated":"2023-10-17T18:20:03Z","snapshot_observed_at":"2026-07-06T16:34:42.650324Z","submitted_at":"2023-10-17T18:20:03Z","title":"GenEval: An Object-Focused Framework for Evaluating Text-to-Image Alignment","version":1},"cited_work":{"arxiv_id":"2310.11513","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.11513","snapshot_observed_at":"2026-07-04T13:19:50.698021Z","title":"GenEval: An Object-Focused Framework for Evaluating Text-to-Image Alignment","venue":null,"work_id":"39e1cc77-c682-4e20-bc57-ee5498213f92","year":2023},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2310.11513","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:0466411a372ba1cfe3ec9b04ab200a7667f4b216847732ce590909ae9a9fc5db","observation_id":"01332f9f-cd8b-4bb8-b7c3-553d3e57960d","resolution":{"observed_at":"2026-05-10T14:15:28.974537Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:9b932a99ec6ca8fed68c18f347af50735353cf7ed2f2d982c5f0bb6247fda945","observation_id":"dabe7962-fcd2-4b9f-8c9a-f390a018f5b4","resolution":{"observed_at":"2026-05-10T14:15:29.017261Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.12598","last_updated":"2022-07-26T01:42:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-07-26T01:42:07Z","title":"Classifier-Free Diffusion Guidance","version":1},"cited_work":{"arxiv_id":"2207.12598","doi":"10.1109/cvpr52733.2024.02494","metadata_source":"pith","pith_arxiv_id":"2207.12598","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Classifier-Free Diffusion Guidance","venue":"cs.LG","work_id":"acf2c588-c088-4a6c-938e-150ad7c666d7","year":2022},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2207.12598","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:b1c476a658a2f80b230fb757c1dd9818cf2da415481d7b827a34864836c31c87","observation_id":"8c2d6a0a-44bf-4c92-9a5a-6c8cd116b812","resolution":{"observed_at":"2026-05-10T15:00:27.982267Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05135","last_updated":"2024-03-08T08:08:10Z","snapshot_observed_at":"2026-08-04T23:51:18.339338Z","submitted_at":"2024-03-08T08:08:10Z","title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","version":1},"cited_work":{"arxiv_id":"2403.05135","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.05135","snapshot_observed_at":"2026-07-04T16:59:58.040795Z","title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","venue":"cs.CV","work_id":"94248955-4bc5-4517-98a0-66224a36d865","year":2024},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2403.05135","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:1e6653cb4b826d9a393535048079a36aa70f2ee70c01c7272515f99d9df94433","observation_id":"03c50dd1-1e34-492c-a3f7-cc4d94827690","resolution":{"observed_at":"2026-05-11T19:43:03.755490Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03147","last_updated":"2025-06-18T18:00:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-03T17:59:33Z","title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","version":4},"cited_work":{"arxiv_id":"2506.03147","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.03147","snapshot_observed_at":"2026-07-05T16:51:14.197597Z","title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","venue":"cs.CV","work_id":"488a273e-95d8-46f1-87c7-2244068d00d0","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2506.03147","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:d98d154beb4bc848ad61d7b15f3d4b09599cd0e590e3b27a94e13732fd250009","observation_id":"6cae96b6-9cd4-4ce6-8ef1-3fa5e551dabe","resolution":{"observed_at":"2026-05-12T17:34:27.304691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02747","last_updated":"2023-02-08T15:46:05Z","snapshot_observed_at":"2026-08-02T18:24:58.914589Z","submitted_at":"2022-10-06T08:32:20Z","title":"Flow Matching for Generative Modeling","version":2},"cited_work":{"arxiv_id":"2210.02747","doi":"10.1038/s41467-024-47656-z","metadata_source":"pith","pith_arxiv_id":"2210.02747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Flow Matching for Generative Modeling","venue":"cs.LG","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","year":2022},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2210.02747","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:abbedb346d26978414268bc97230c81cc464ac5c0e2c7ae50c85372040e32b85","observation_id":"9aa61bb8-5042-4e6a-ab2e-7f96f3a8d085","resolution":{"observed_at":"2026-05-10T14:15:29.000052Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"crossref_status_cache"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.05606","doi":"10.48550/arxiv.2508.05606","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Uni-cot: Towards unified chain-of-thought reasoning across text and vision.arXiv preprint arXiv:2508.05606","venue":"arXiv (Cornell University)","work_id":"2d7c3351-90a7-45ce-a374-890c71fbea6e","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:e2e9fcec5abc1f12d45ce9a06d1b7a95aece91fdeccfcea2dafb8bf5fa392d44","observation_id":"544b655d-27ec-4060-964e-bfa509037882","resolution":{"observed_at":"2026-05-10T14:15:28.985902Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03069","last_updated":"2025-08-07T09:08:33Z","snapshot_observed_at":"2026-07-06T20:01:23.373458Z","submitted_at":"2024-12-04T06:46:55Z","title":"TokenFlow: Unified Image Tokenizer for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2412.03069","doi":"10.48550/arxiv.2412.03069","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03069","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tokenflow: Unified image tokenizer for multimodal understanding and generation","venue":"arXiv (Cornell University)","work_id":"6a964215-761a-438d-b579-a2699f32913f","year":2024},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2412.03069","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:ec9a2cde7a7f2986d6fd364e21e6c8119b05678fae2610e96e0613c04c199bc5","observation_id":"85b178fd-4ad0-47f0-974e-080507de5924","resolution":{"observed_at":"2026-05-10T14:15:28.979944Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15188","last_updated":"2025-02-05T02:26:38Z","snapshot_observed_at":"2026-08-02T19:18:33.010410Z","submitted_at":"2024-12-19T18:56:24Z","title":"LMFusion: Adapting Pretrained Language Models for Multimodal Generation","version":4},"cited_work":{"arxiv_id":"2412.15188","doi":"10.48550/arxiv.2412.15188","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.15188","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2025.doi:10.48550/arXiv.2412.15188","venue":"arXiv (Cornell University)","work_id":"7dc74790-08b8-4917-aa5a-f9da5f5b3edf","year":2024},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2412.15188","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:1486e967b9c95a4c9c708ffdfdf6585d75ddafbedbb442c793699bf61fb59f59","observation_id":"b0800302-88df-4745-bf7b-dd720fc3f116","resolution":{"observed_at":"2026-05-10T14:15:28.953805Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.11807","last_updated":"2020-06-21T14:10:47Z","snapshot_observed_at":"2026-07-06T09:31:08.556302Z","submitted_at":"2020-06-21T14:10:47Z","title":"Improving Image Captioning with Better Use of Captions","version":1},"cited_work":{"arxiv_id":"2006.11807","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2006.11807","snapshot_observed_at":"2026-07-04T12:49:52.887579Z","title":"Improving image captioning with better use of captions","venue":null,"work_id":"7d761619-9f65-478d-be0f-a8feb0119886","year":2006},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2006.11807","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:f93bbd38339ab7aa208213ef5237bab1af0976a23b3368d42d290b3fca8289b4","observation_id":"76583375-fdc1-4add-9d4e-7f223ad8d26a","resolution":{"observed_at":"2026-05-10T14:15:28.958910Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09818","last_updated":"2025-03-21T05:54:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-16T05:23:41Z","title":"Chameleon: Mixed-Modal Early-Fusion Foundation Models","version":2},"cited_work":{"arxiv_id":"2405.09818","doi":"10.48550/arxiv.2405.09818","metadata_source":"pith","pith_arxiv_id":"2405.09818","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chameleon: Mixed-Modal Early-Fusion Foundation Models","venue":"cs.CL","work_id":"2661b9a6-25cc-41a1-8100-612d2b801289","year":2024},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2405.09818","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:cdd7b768737ce4b6efa060b6f307678358a8fe64a981b67caf370e54a2227cf5","observation_id":"0001630c-4456-4ef4-b068-4a49be2a21ba","resolution":{"observed_at":"2026-05-11T10:03:28.218044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T17:49:22.013335+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T17:49:22.013335+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":"2501.12599","doi":"10.48550/arxiv.2501.12599","metadata_source":"pith","pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","venue":"cs.AI","work_id":"bff96ab1-bd6a-4585-be23-74fdb51969c7","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:c7b8a0e34b8ee4a5cb6831702ca17d0a62e4e0c2845b0c224a542f95554f1b01","observation_id":"322f73f2-315e-47a4-b66d-991648ca7902","resolution":{"observed_at":"2026-05-10T17:58:27.800589Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:38.585868+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:38.585868+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18869","last_updated":"2024-09-27T16:06:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T16:06:11Z","title":"Emu3: Next-Token Prediction is All You Need","version":1},"cited_work":{"arxiv_id":"2409.18869","doi":"10.48550/arxiv.2409.18869","metadata_source":"pith","pith_arxiv_id":"2409.18869","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emu3: Next-Token Prediction is All You Need","venue":"cs.CV","work_id":"720d288e-fac0-464c-9929-19efd9a52afc","year":2024},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"cited_paper":"/paper/2409.18869","citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:7e394fab95be620de19cda2365babe9a491510cf661cffa737f93922b33a4a27","observation_id":"b143072c-af30-4f5a-b92a-4ef2688af517","resolution":{"observed_at":"2026-05-11T10:56:09.993852Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.22946","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T18:35:00.342868Z","title":"arXiv:2510.22946 (2025) 3","venue":null,"work_id":"9373ea7b-3f55-41f8-bc09-f286cecd7130","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:986c99c81e101efb78dde74b229cb8ae1e1a726ec22ea2018311bdd949e05b24","observation_id":"5c9b118e-708b-4639-8572-a17aa07ce5a2","resolution":{"observed_at":"2026-05-10T14:15:28.946140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.02567","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T07:36:58.031623Z","title":"Unified multimodal understanding and generation models: Advances, challenges, and opportunities.arXiv preprint arXiv:2505.02567","venue":null,"work_id":"b2e70f99-2155-4021-82c3-540574140c6a","year":2025},"citing_paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T14:15:02.723774Z"},"links":{"citing_paper":"/paper/2604.13540"},"observation_digest":"sha256:73051ad8febea320498d297e7951c4e90d07d737de1496bc3b978199f478fc2d","observation_id":"f16f6d0d-1bb4-4c3b-a7ef-7a96ce2d136e","resolution":{"observed_at":"2026-05-10T14:15:28.950010Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.13540","last_updated":"2026-04-15T06:41:56Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:41:56Z","title":"Free Lunch for Unified Multimodal Models: Enhancing Generation via Reflective Rectification with Inherent Understanding"},"reference_resolution":{"displayed":18,"state_counts":{"malformed_identifier":0,"metadata_mismatch":6,"parse_uncertain":0,"unresolved":0,"verified_exact":12,"verified_fuzzy":0},"total_outbound_references":18},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 18 of 18 outbound references and 0 inbound Pith citation observations for arXiv:2604.13540."}