{"as_of":"2026-08-23T21:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:91218973a2876a7240fc3dc9dd95e62168e8388fa5d61fe901800232920f2b4c","coverage":[{"denominator":85,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":85,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:24:10.973352Z","state":"measured"},{"denominator":85,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":85,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.13195/citation-record","integrity":"/paper/2412.13195/integrity","json":"/paper/2412.13195/citation-record.json","paper":"/paper/2412.13195"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.448204Z","title":"Hrs-bench: Holistic, reliable and scalable benchmark for text-to-image models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.448204Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:200aa5831985f38909edaa2b506f1d03be1055ed5e2bec571111a26143935a30","observation_id":"af09a962-410c-4633-baf7-fa361a82a668","resolution":{"observed_at":"2026-08-11T13:24:10.448204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.454506Z","title":"Improving Image Genera- tion with Better Captions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.454506Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:ffbb3fb2a83f0f28fd4e56d01a1d8805ca91f1af44308637da9f2a6cd84928d3","observation_id":"8c18e47f-eaf8-408d-a348-aee5213e7c4a","resolution":{"observed_at":"2026-08-11T13:24:10.454506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.460598Z","title":"Training diffusion models with reinforce- ment learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.460598Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:463b5a3eef039cb8779824caabb9010cf6046ed9721797812406c2d4a46f1412","observation_id":"0de1cc4c-a751-4bb9-8d41-f9b0bc750704","resolution":{"observed_at":"2026-08-11T13:24:10.460598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.466530Z","title":"FLUX.1-dev","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.466530Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:00c0229394b14e3c1814ceba9076a954ec97e41d8734e7ddf14ce5789f373da4","observation_id":"3d84197d-9f93-4fdf-a324-4bcea850789f","resolution":{"observed_at":"2026-08-11T13:24:10.466530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.472366Z","title":"Conceptual 12m: Pushing web-scale image-text pre- training to recognize long-tail visual concepts","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.472366Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:391a82121e3e1f87f9342176f24720798813ca90b3e9947529196b4560f0ee01","observation_id":"cb2c6247-805d-40ac-9042-dfc3950a54fc","resolution":{"observed_at":"2026-08-11T13:24:10.472366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.791052Z","title":"Getting it right: Improving spatial consis- tency in text-to-image models","venue":null,"work_id":"30cfef8e-7734-41d0-ad54-406e27e3d9c9","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.477951Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:1b7282e786eee49619ffae3614167abefd2fee62d2a79e0887bd80177ee23314","observation_id":"3363fdc0-0f7c-416f-b1ef-9ca6e21ecde5","resolution":{"observed_at":"2026-08-11T13:24:12.796356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.772846Z","title":"Attend-and-excite: Attention-based se- mantic guidance for text-to-image diffusion models","venue":null,"work_id":"8f2ca7ec-6973-4d35-bb5b-c82574d643e4","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.483179Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:faf55646a9082fcba4c63e267307c2c66f3b5b02f9ed841e5e7cf983b9802d89","observation_id":"b9fa0cb9-321a-4865-8d00-b8c45627b3c6","resolution":{"observed_at":"2026-08-11T13:24:12.778961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.753666Z","title":"Cohn, Dayou Liu, Sheng-Sheng Wang, Jihong Ouyang, and Qiangyuan Yu","venue":null,"work_id":"3bb0d4b1-7051-4bdb-a053-e79fa9fc3843","year":2015},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.488978Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:9ba6983d807e96b72bec7a285d196f497b4c83e8ffd082402084766c69978ae5","observation_id":"102c9a39-a694-47f6-a416-0473d30b262a","resolution":{"observed_at":"2026-08-11T13:24:12.760862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.732969Z","title":"Cohn, Dayou Liu, Sheng-Sheng Wang, Jihong Ouyang, and Qiangyuan Yu","venue":null,"work_id":"f8ad0cec-a3a7-4461-8473-b7399b83a5f0","year":2015},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.494138Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:e8689cec82aa5c21294e381d8ef3d8ad1c7bd34a7eaf1ea465e2c3be76628899","observation_id":"8da81bf0-c142-4669-a306-17dce0b2cba5","resolution":{"observed_at":"2026-08-11T13:24:12.738552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.713456Z","title":"Pixart- Σ: Weak-to-strong training of dif- fusion transformer for 4k text-to-image generation","venue":null,"work_id":"4ec33ddd-c1e5-4ce1-b372-8b1887128b99","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.498844Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:d48386e8523046fbb02b01cd34666ebbde7080071cff41df6bf98b7dbb87ab12","observation_id":"bca47ade-3f40-4842-acb0-e5282a6a1d3c","resolution":{"observed_at":"2026-08-11T13:24:12.719647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05252","last_updated":"2024-01-10T16:27:38Z","snapshot_observed_at":"2026-08-16T14:28:28.987831Z","submitted_at":"2024-01-10T16:27:38Z","title":"PIXART-{\\delta}: Fast and Controllable Image Generation with Latent Consistency Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05252","snapshot_observed_at":"2026-08-11T13:24:10.503518Z","title":"Pixart- δ: Fast and controllable image generation with latent consistency mod- els","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.503518Z"},"links":{"cited_paper":"/paper/2401.05252","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:e6ae8c950d0a53485fc598b1e39af99327aabc930a8e95f5c82cc8620051e816","observation_id":"7a101917-882d-4623-85e1-d04b43433a7d","resolution":{"observed_at":"2026-08-11T13:24:10.503518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.684751Z","title":"Kwok, Ping Luo, Huchuan Lu, and Zhenguo Li","venue":null,"work_id":"f018d64b-2b80-4bf9-bca4-f35543eb6a52","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.509234Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:1031522da8bca20c3d1c45b1dfa4772c5c61b90765e1ffbd8567903d46312c26","observation_id":"4d672181-2995-4cbc-94dc-4420a8875e78","resolution":{"observed_at":"2026-08-11T13:24:12.695268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.661142Z","title":"Training-free layout control with cross-attention guidance","venue":null,"work_id":"2f50c24a-14e7-4803-b752-968c56abe7b6","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.515304Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:9403f6a0d3117999849b5557b917f7f5728b1ba7ffafa45d4bfc14a904983066","observation_id":"d42d775a-fb74-4ae0-9514-4c1a903ad14f","resolution":{"observed_at":"2026-08-11T13:24:12.666729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.640005Z","title":"Reproducible scaling laws for contrastive language-image learning","venue":null,"work_id":"6fc7be8f-3ddc-42fe-b0da-6cdaca4ad1ca","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.520846Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:57a67ef22ad66156750df9ef8f205c1480039d173755cfd69b9d4408bfcf83eb","observation_id":"96f2c72e-714b-4ec6-ab8a-9f2773381b35","resolution":{"observed_at":"2026-08-11T13:24:12.648255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.619339Z","title":"Visual pro- gramming for step-by-step text-to-image generation and evaluation","venue":null,"work_id":"9cf592c7-ea5a-4786-8673-f3fb326bb20b","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.525705Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:2f28013b5e0df572ef5f9028d69ebc445471b368d572070da4be2ce4a4852184","observation_id":"091e9028-2ce8-483b-886b-04b61b8c425a","resolution":{"observed_at":"2026-08-11T13:24:12.625385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.600603Z","title":"Coventry, Merc `e Prat-Sala, and Lynn Richards","venue":null,"work_id":"1e0f522c-bf42-4986-8dbd-f454d9665c2c","year":2001},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.532228Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:bce96544a965dcc558c2e036167a362ec0211aa92d1a432345cc7a65b3e92bb3","observation_id":"c93a0399-66fd-49be-9fef-7ed72e56cea7","resolution":{"observed_at":"2026-08-11T13:24:12.606074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.581477Z","title":"Dall·e mini","venue":null,"work_id":"c0fe18b4-592f-40ab-9f84-479f5b008a0e","year":2021},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.540376Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:03c4f8df44400285d5a308453b932fbbe5dcb48393271d20cb2901d718cdb32a","observation_id":"63e64da8-b7bf-4218-90ac-21cbe38ba662","resolution":{"observed_at":"2026-08-11T13:24:12.588078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.546909Z","title":"Diffu- sion models beat gans on image synthesis","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.546909Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:5408960d82f1c93c61e8ebdabe74869ffe30862cc9a41d969f469400f50e6fa1","observation_id":"132aeb3f-f4b2-45d5-80ce-8ce3e606c3ad","resolution":{"observed_at":"2026-08-11T13:24:10.546909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.546441Z","title":"Cogview2: Faster and better text-to-image generation via hierarchical transformers","venue":null,"work_id":"f322923e-869b-4d27-acaf-de419c84e88c","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.552714Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:5030f8467eb3aaea5c78459cdae0bc9962581b4ccf6f52e785fd826533215db3","observation_id":"26028663-c670-4f62-82b9-c724ffb288d9","resolution":{"observed_at":"2026-08-11T13:24:12.552414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.522735Z","title":"Scaling rec- tified flow transformers for high-resolution image synthesis","venue":null,"work_id":"1d0690e6-80c3-4170-acac-ab0c76a16c14","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.559790Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:d84df8249a600beba53ff32ec017f7a42c979a813dae0a4c7ba2da6eaa5ed14b","observation_id":"f16bbd7e-d132-4e80-a5c7-f744889a73dc","resolution":{"observed_at":"2026-08-11T13:24:12.531405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.498746Z","title":"Re- inforcement learning for fine-tuning text-to-image diffusion models","venue":null,"work_id":"aa058520-88d1-4829-a1fe-41082c605f24","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.566012Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:e87c8a77069631b4503321bbb40262a6ef943c4f4e6d449d8bee28aaf47d9680","observation_id":"3fc00ec8-d18d-44b1-a426-64a9359772f3","resolution":{"observed_at":"2026-08-11T13:24:12.505806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.469133Z","title":"Akula, Pradyumna Narayana, Sugato Basu, Xin Eric Wang, and William Yang Wang","venue":null,"work_id":"868d55e6-e8f2-43d5-ac08-6074dae5f473","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.572203Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:279cddf7bccf6e8e272f7c20290dacb2e3407c126f458b207c3beb13a8001db0","observation_id":"15d09b38-ff1a-49f3-bb54-645892f8cabe","resolution":{"observed_at":"2026-08-11T13:24:12.478478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.445975Z","title":"Akula, Xuehai He, Sugato Basu, Xin Eric Wang, and William Yang Wang","venue":null,"work_id":"ad6bb026-cdce-47b6-a77b-5478d08ff231","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.577769Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:c8eb264e2940a077a68919a21731c486ca528b7186978873badf8c67e2cf1316","observation_id":"8c3e2e20-3aff-46e8-8794-f14fb1036e13","resolution":{"observed_at":"2026-08-11T13:24:12.452703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08656","last_updated":"2024-06-12T21:41:32Z","snapshot_observed_at":"2026-08-21T16:09:45.055261Z","submitted_at":"2024-06-12T21:41:32Z","title":"TC-Bench: Benchmarking Temporal Compositionality in Text-to-Video and Image-to-Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08656","snapshot_observed_at":"2026-08-11T13:24:10.583658Z","title":"Tc-bench: Benchmark- ing temporal compositionality in text-to-video and image-to- video generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.583658Z"},"links":{"cited_paper":"/paper/2406.08656","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:9550d5fd162bf75cea0d77963d150393bbbf250931a76604157b9b611e336f66","observation_id":"91ef17fe-07c7-4694-8578-06be7ef9ddfa","resolution":{"observed_at":"2026-08-11T13:24:10.583658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.422863Z","title":"Geneval: An object-focused framework for evaluating text- to-image alignment","venue":null,"work_id":"14d62012-7a3c-4014-ac1f-21fe1b30a0f4","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.590061Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:c93e9b7e0b8f6b401e6b62de177ed6ad0012fd3853a5e85ef917b58ac35ad59c","observation_id":"267c75fe-1f7f-4b97-9340-31f22bde8b18","resolution":{"observed_at":"2026-08-11T13:24:12.428912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10015","last_updated":"2023-10-27T17:24:04Z","snapshot_observed_at":"2026-08-16T16:07:26.772862Z","submitted_at":"2022-12-20T06:03:51Z","title":"Benchmarking Spatial Relationships in Text-to-Image Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10015","snapshot_observed_at":"2026-08-11T13:24:10.596080Z","title":"Benchmarking spatial relationships in text-to-image generation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.596080Z"},"links":{"cited_paper":"/paper/2212.10015","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:8779cbf3b358f92c42a0c8b6108aa51715849b34230d45ea1a44db8d8eddd49c","observation_id":"dfd366b9-6386-4ffd-bffe-58f836eda018","resolution":{"observed_at":"2026-08-11T13:24:10.596080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06382","last_updated":"2024-06-10T15:42:03Z","snapshot_observed_at":"2026-08-16T13:44:30.615649Z","submitted_at":"2024-06-10T15:42:03Z","title":"Diffusion-RPO: Aligning Diffusion Models through Relative Preference Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06382","snapshot_observed_at":"2026-08-11T13:24:10.602422Z","title":"Diffusion-rpo: Aligning diffusion mod- els through relative preference optimization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.602422Z"},"links":{"cited_paper":"/paper/2406.06382","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:c342544c9ac8dddf0aeb3d632d6dd89b7ebce78c61b49de87193f6fa23ba23d1","observation_id":"03b32048-f7ac-474a-9738-c5fb36383d8c","resolution":{"observed_at":"2026-08-11T13:24:10.602422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.400195Z","title":null,"venue":null,"work_id":"32dcafb3-3b90-4460-9d95-20ab574bec5a","year":2016},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.613575Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:1d175dbc0cf07c55b57b42404ed6fdb38ab3fcabce5510fd610c3517b4e9491b","observation_id":"950d7731-c9b4-4de6-8665-0e2eef36c52b","resolution":{"observed_at":"2026-08-11T13:24:12.406299Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.377735Z","title":"Ganspace: Discovering interpretable GAN controls","venue":null,"work_id":"fd4e4c69-1818-4ee8-bcc0-5877d032e528","year":2020},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.621527Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:f50db18f0c0037b3cecf98b8128bd6d16b2e6d668ead480f527ed0f15e1679fb","observation_id":"aac30f3d-c1d3-40c7-901d-ff32e305d6b4","resolution":{"observed_at":"2026-08-11T13:24:12.386207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.354644Z","title":"Prompt-to-prompt image editing with cross-attention control","venue":null,"work_id":"534eff98-966b-4277-9f99-9d33b4700be2","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.628033Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:d20c0484a2e6bd460e18c4b554a2282b2e5720bb325cd7865d891d2898988ca3","observation_id":"b9a90481-b722-45f0-9f08-c19164f43025","resolution":{"observed_at":"2026-08-11T13:24:12.360600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.332723Z","title":"Gans trained by a two time-scale update rule converge to a local nash equilib- rium","venue":null,"work_id":"fe65f745-2db3-4d8b-a133-06f8ed821b20","year":2017},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.633696Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:00d33f9cfd4951a97b7c6fbd778630c9789003aaa1e4101673cff8f2be961818","observation_id":"620e772f-7014-41ef-a31a-07780e57dd31","resolution":{"observed_at":"2026-08-11T13:24:12.338897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.639534Z","title":"Denoising dif- fusion probabilistic models","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.639534Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:155aa189dd0f308f34827b17ef3b90cb2b816eec7bec5d1d07532a5340c02222","observation_id":"ec42c18d-9b43-4b46-b951-62d2d7ec4e26","resolution":{"observed_at":"2026-08-11T13:24:10.639534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05135","last_updated":"2024-03-08T08:08:10Z","snapshot_observed_at":"2026-08-12T17:58:56.652054Z","submitted_at":"2024-03-08T08:08:10Z","title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05135","snapshot_observed_at":"2026-08-11T13:24:10.644541Z","title":"ELLA: equip diffusion models with LLM for en- hanced semantic alignment","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.644541Z"},"links":{"cited_paper":"/paper/2403.05135","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:51ac383d786926313c0db8ed6d310ca01e98ce827dd101e442d5a6866170782f","observation_id":"cf04431c-2031-4c3d-a723-48a21ca52776","resolution":{"observed_at":"2026-08-11T13:24:10.644541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.299444Z","title":null,"venue":null,"work_id":"8372851c-2e0d-43b9-9e38-42c7cc8c8164","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.649920Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:eff8650a4390eb6f0e7377c87286e6396c1653262965d0b36e8d7104592e8c1d","observation_id":"92168039-98be-4c48-8a39-d198b6f5546f","resolution":{"observed_at":"2026-08-11T13:24:12.305667Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.278025Z","title":"T2i-compbench: A comprehensive benchmark for open-world compositional text-to-image generation","venue":null,"work_id":"7cb4044b-8b44-48f2-9e55-b03b6fa53028","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.654641Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:6f29aa717787bf630c40a14d8975929d699aad749ff7ca9e66bf0a29c8435bc1","observation_id":"59756ed9-5ff3-4488-929d-139381a3cf15","resolution":{"observed_at":"2026-08-11T13:24:12.285816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.250780Z","title":"Re- thinking FID: towards a better evaluation metric for image generation","venue":null,"work_id":"0de87759-f201-476e-ac18-17183840af45","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.661068Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:718cb49a2f7b51c564bb6ee89b80ebcdce6652964eaf942f8affd261d60bee35","observation_id":"c254fee4-ec18-482b-9b2d-f093ee7f6e6e","resolution":{"observed_at":"2026-08-11T13:24:12.258315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.219767Z","title":"Comat: Aligning text-to-image diffusion model with image- to-text concept matching","venue":null,"work_id":"e25fd8e2-635f-41ca-bf13-6f9b4d06227a","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.665565Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:3e08e1867763632921738ebeaa9bca19aee708be855551c8f3a24c62ee60d2c7","observation_id":"beeb2cfe-e62b-4b66-b014-2070001f75ce","resolution":{"observed_at":"2026-08-11T13:24:12.226180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.18013","last_updated":"2024-10-30T13:40:01Z","snapshot_observed_at":"2026-08-21T16:43:37.503570Z","submitted_at":"2024-10-23T16:42:56Z","title":"Scalable Ranked Preference Optimization for Text-to-Image Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.18013","snapshot_observed_at":"2026-08-11T13:24:10.670874Z","title":"Scalable ranked preference optimization for text-to-image generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.670874Z"},"links":{"cited_paper":"/paper/2410.18013","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:4f44dc1a9957af794d7cf634d2b6f80dff0f6e809a31929315c588a17fafc8a8","observation_id":"095f94bf-d488-4cb7-abd2-fba298eefc5a","resolution":{"observed_at":"2026-08-11T13:24:10.670874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.201492Z","title":"Evaluating and improving composi- tional text-to-visual generation","venue":null,"work_id":"c9b6f15c-5374-406a-a94c-3fb2c52b2a75","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.676906Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:ef2ac074acfcbc0626fd8e7ff36c91611f8c126a0b6a72e426cce7b88d88902f","observation_id":"69d327ea-1d7e-4a90-bc86-9a689801b250","resolution":{"observed_at":"2026-08-11T13:24:12.207257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04461","last_updated":"2023-12-07T17:32:29Z","snapshot_observed_at":"2026-08-16T14:36:46.011897Z","submitted_at":"2023-12-07T17:32:29Z","title":"PhotoMaker: Customizing Realistic Human Photos via Stacked ID Embedding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04461","snapshot_observed_at":"2026-08-11T13:24:10.681938Z","title":"Photomaker: Customizing realistic human photos via stacked ID embedding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.681938Z"},"links":{"cited_paper":"/paper/2312.04461","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:7f3e800f4945009b887ae77453a95d21e7ae954e8c48c94c9fc47fc6d7239ec0","observation_id":"f90974a1-72a5-4783-99a5-3f2b4038da5a","resolution":{"observed_at":"2026-08-11T13:24:10.681938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.183560Z","title":"Llm- grounded diffusion: Enhancing prompt understanding of text-to-image diffusion models with large language models","venue":null,"work_id":"7d709fec-71ee-4524-a9d0-bff8adf6f732","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.689503Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:6f5bec25ef0b9c6742d060d5c8271b040dcf48a1feb71eba94652f7bfec2d6f4","observation_id":"6b2af98c-1698-4016-a66d-b7bf95c046dc","resolution":{"observed_at":"2026-08-11T13:24:12.189468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.159400Z","title":"Collins, Yiwen Luo, Yang Li, Kai J","venue":null,"work_id":"e6fe22ef-b56e-4ed3-8fdf-e370e23fdbb0","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.695624Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:ef0cdc305b4dbaeabf5da0ba2c2157a45250b04ce649b9702e8a432be3dd3677","observation_id":"c9aca626-6551-44c8-a593-4bf3ac87b3f7","resolution":{"observed_at":"2026-08-11T13:24:12.169740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.138047Z","title":"Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll ´ar, and C","venue":null,"work_id":"04bfff71-b42a-4285-9fe4-2d9b8e307ec0","year":2014},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.701011Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:3d5a53a06a4cf2d048a459cea560054d2f7cdfcce26cd35be9c7d52c0af13797","observation_id":"765c9c1e-fcd3-4715-9282-cd6e1e1300a2","resolution":{"observed_at":"2026-08-11T13:24:12.144736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.118212Z","title":"Tenenbaum","venue":null,"work_id":"370593a3-6cb2-4233-8921-73a0973596fe","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.707640Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:80fa7282c5edf03c6f5556e00c27ee40f746e6f67aef3e34f214ecf67dcf1ee1","observation_id":"54e9693f-1402-4ca4-8c74-b0f2a6506ee2","resolution":{"observed_at":"2026-08-11T13:24:12.123852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.099155Z","title":"Repaint: Inpainting using denoising diffusion probabilistic models","venue":null,"work_id":"787a1a08-61fe-4346-9b32-2aa57a680bd3","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.713183Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:0b22c0829799f8851d90ab35b4706e7b0e14409c1178707a3640764e06425888","observation_id":"1f5aaf0f-4191-40e3-a279-e9a4f943f8a8","resolution":{"observed_at":"2026-08-11T13:24:12.104822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.080029Z","title":"Pick-and-draw: Training-free semantic guidance for text-to-image person- alization","venue":null,"work_id":"cf6f24e2-9602-4e0b-b065-744a0f286b8e","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.718984Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:0d86d90ac4a1b47a027f65e051b5ebb6a9f18f27947ccaedbd7c8c383e3b76fe","observation_id":"d0320800-6ae1-466a-adfc-0c3e2a2713ca","resolution":{"observed_at":"2026-08-11T13:24:12.086178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.055503Z","title":"MidJourney","venue":null,"work_id":"c2d5b360-3c20-41f1-8a53-5ba47fc2eb1c","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.724530Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:9afbafc30c62d0077a64f6bc496e44ca709e07ea918669d7c1fcea504899f9f8","observation_id":"8ef50172-4d70-4481-ad23-6c70ca93dae9","resolution":{"observed_at":"2026-08-11T13:24:12.063572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.032800Z","title":"T2i-adapter: Learning adapters to dig out more controllable ability for text-to-image diffusion models","venue":null,"work_id":"b697648d-77e5-4d07-b3d4-a5968dcb1566","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.730516Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:fb6ffa3ca299fe5b0fe30ce5c7f2bb6c84d59f6d65f67151bf7ef623489e5362","observation_id":"29ea6c02-6f9a-469c-90b0-022c3e0b4f4d","resolution":{"observed_at":"2026-08-11T13:24:12.041000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:12.009307Z","title":"GLIDE: towards photorealis- tic image generation and editing with text-guided diffusion models","venue":null,"work_id":"2c83b559-ca21-4c4e-b2a3-9ae80832ac90","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.735735Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:075d0f694a90540fc3cc215dfafda11a7129f8196759fd50b4ab080283786e72","observation_id":"e86a4a1c-254c-4a14-bb37-eb7655bbd2d8","resolution":{"observed_at":"2026-08-11T13:24:12.017373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.987254Z","title":"Drag your GAN: interactive point-based manipulation on the generative image manifold","venue":null,"work_id":"84981999-64f7-4c70-8791-920c0727eee8","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.741274Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:b6984a4ba2905d66b78fd97b6b23e300e6b73c31fcfa95917b830f46c48cc6be","observation_id":"926a0225-76de-4d83-8fbf-e0947af5978c","resolution":{"observed_at":"2026-08-11T13:24:11.994288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.967839Z","title":"Scalable diffusion models with transformers","venue":null,"work_id":"564173de-dc6a-48e1-8cc5-4741e563de56","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.747315Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:a3dfe151a5d867135790430d983fdeca54b0e4002a7a559afd9120723aec9791","observation_id":"e23bb6c6-2188-407b-89b3-ffa53ff711c9","resolution":{"observed_at":"2026-08-11T13:24:11.973689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.949843Z","title":"Grounded text-to-image synthesis with attention refocusing","venue":null,"work_id":"ae123b66-5b44-45c4-8394-8a1f2e141699","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.754972Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:0d0a9f90b2cb3d830a8242b02a58047fb6a029eea827c92d32996f3278fbdc27","observation_id":"aa52c4ff-ddd0-4596-ac1b-3b74d8be1610","resolution":{"observed_at":"2026-08-11T13:24:11.956466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.928858Z","title":"SDXL: improving latent diffusion models for high-resolution image synthesis","venue":null,"work_id":"b55d4b6c-5a75-4954-a962-76dd922184fd","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.761651Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:63e22f4e69518edd2b56873fc4dfff9e1c891144fb8eafe412b98c7001b54bb9","observation_id":"e01d5566-530d-4474-a642-06b447fecaa6","resolution":{"observed_at":"2026-08-11T13:24:11.934740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.910432Z","title":"Barron, and Ben Milden- hall","venue":null,"work_id":"ca677853-1156-4797-92df-056251a32d46","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.767976Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:d3905eb9029318e20e8eed50cb25239db563369d31222d14ec013e0f089f5515","observation_id":"7080e902-5def-4ce3-a9ca-78d6b19aec07","resolution":{"observed_at":"2026-08-11T13:24:11.916358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.773083Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.773083Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:eea048d82a9d80e47b733f0d2cee705316030ab6e5b55dca00c6ba1ac2a52e80","observation_id":"0d0c5842-a78b-4d3c-850b-4fdda161ffb6","resolution":{"observed_at":"2026-08-11T13:24:10.773083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.881069Z","title":null,"venue":null,"work_id":"99027703-39a8-4756-a4e1-5be4b9883551","year":2020},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.778341Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:336c70206a0bdbc54c11f4330e06f4b23cc3a8871c9a19df5fd476a70e840f72","observation_id":"dd5e34d9-b678-4588-b97c-a933d210b9d9","resolution":{"observed_at":"2026-08-11T13:24:11.886403Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.06125","last_updated":"2022-04-13T01:10:33Z","snapshot_observed_at":"2026-08-15T12:50:58.405488Z","submitted_at":"2022-04-13T01:10:33Z","title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.06125","snapshot_observed_at":"2026-08-11T13:24:10.783583Z","title":"Hierarchical text-conditional image gener- ation with CLIP latents","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.783583Z"},"links":{"cited_paper":"/paper/2204.06125","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:9430cdb6b672ea727459f7c60f2714767d355e2b1d492db8d7a7228d2cb6e390","observation_id":"2984c77a-f6cf-4c2e-8e3e-3e6545ad89cb","resolution":{"observed_at":"2026-08-11T13:24:10.783583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.862329Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":"2b82de8e-c63c-411e-812e-a54a523a44f8","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.789097Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:602c11fbd4bd80be6fdf3523043fa4c0ea4ef6c664e78155351573e0520fa8bd","observation_id":"7687f769-764f-4579-9e80-72d47737ed45","resolution":{"observed_at":"2026-08-11T13:24:11.868151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.794549Z","title":"Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.794549Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:a28705658ef462d35a69b3bcc1740a4dd4f870cd0523650078adb019a0fac115","observation_id":"b7e4b89f-92ce-4ee5-bffa-374d6ac74fd1","resolution":{"observed_at":"2026-08-11T13:24:10.794549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.828762Z","title":"Runway AI","venue":null,"work_id":"d781a535-0aae-41ec-8bb4-4217fb6454d9","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.800035Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:bb2a3f0b1441a12ccc6d41124aea4b5545cfe6dcff9a614a088aa6916d279137","observation_id":"10611063-b259-48da-8249-1026876cdb59","resolution":{"observed_at":"2026-08-11T13:24:11.835898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.06023","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.297516Z","title":"Dual caption preference optimization for diffusion models","venue":null,"work_id":"7c67fda5-842c-4e5e-ae14-672cad35eeca","year":2025},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.805071Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:194d5a44cc6fdeac2da66e7ce0f5eac485eeebf85767e4da35c4cd8d8be446d6","observation_id":"c9594d61-30ab-482b-a943-4fca15c1330b","resolution":{"observed_at":"2026-08-11T13:24:11.309326Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.810778Z","title":"Denton, Seyed Kamyar Seyed Ghasemipour, Raphael Gontijo Lopes, Burcu Karagol Ayan, Tim Salimans, Jonathan Ho, David J","venue":null,"work_id":"b8679689-da57-475f-8756-472d0a512741","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.810558Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:141eb2d411b5dfd30c265e873550726dcef22d3a2f180e551ea727a420b66bc9","observation_id":"49d2fb5e-11e9-428b-b0fb-1e47a34d58a5","resolution":{"observed_at":"2026-08-11T13:24:11.816335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.793742Z","title":"Scribbler: Controlling deep image synthesis with sketch and color","venue":null,"work_id":"498995cb-915c-4fd3-b69e-f211c7826b0b","year":2017},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.816766Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:dbad46bc354e479d8b688b2a61b4d0ef024a7e8aa7a827fecac5766e8a1a8271","observation_id":"638370b8-2bd4-4f66-8c89-162c47bd72d5","resolution":{"observed_at":"2026-08-11T13:24:11.799628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02114","last_updated":"2021-11-03T10:16:39Z","snapshot_observed_at":"2026-08-02T08:12:49.547570Z","submitted_at":"2021-11-03T10:16:39Z","title":"LAION-400M: Open Dataset of CLIP-Filtered 400 Million Image-Text Pairs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.02114","snapshot_observed_at":"2026-08-11T13:24:10.822470Z","title":"LAION- 400M: open dataset of clip-filtered 400 million image-text pairs","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.822470Z"},"links":{"cited_paper":"/paper/2111.02114","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:d84bc46dc2c5b67b596f05b68965f8bdc5d315d45c22e89325c5e457d4d0d372","observation_id":"445fbd8f-510d-44c1-b4ed-c9510c3693fd","resolution":{"observed_at":"2026-08-11T13:24:10.822470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.776157Z","title":"LAION-5B: an open large-scale dataset for training next generation image-text models","venue":null,"work_id":"6d0761fa-4a39-4595-94c2-f543939a53f7","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.828253Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:84216761ff54fb8e620331cbfa8298adf5c90494ea6980f47e787a044c5b7622","observation_id":"37a46d67-7e22-41ae-ba4c-e2feb7305527","resolution":{"observed_at":"2026-08-11T13:24:11.781666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16656","last_updated":"2023-10-25T14:10:08Z","snapshot_observed_at":"2026-08-19T03:01:39.732108Z","submitted_at":"2023-10-25T14:10:08Z","title":"A Picture is Worth a Thousand Words: Principled Recaptioning Improves Image Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16656","snapshot_observed_at":"2026-08-11T13:24:10.841427Z","title":"A picture is worth a thousand words: Principled recaptioning improves image generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.841427Z"},"links":{"cited_paper":"/paper/2310.16656","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:bf2c8d1aad177928c2d3a06347c1e808470dd4cefa698b430685e07798a6d004","observation_id":"ab00d0d8-669d-4b10-8593-ba5be5ffd414","resolution":{"observed_at":"2026-08-11T13:24:10.841427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17910","last_updated":"2024-02-27T21:51:32Z","snapshot_observed_at":"2026-08-16T14:14:44.996930Z","submitted_at":"2024-02-27T21:51:32Z","title":"Box It to Bind It: Unified Layout Control and Attribute Binding in T2I Diffusion Models","version":1},"cited_work":{"arxiv_id":"2402.17910","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.17910","snapshot_observed_at":"2026-08-11T13:24:11.108391Z","title":"Box It to Bind It: Unified Layout Control and Attribute Binding in T2I Diffusion Models","venue":"cs.CV","work_id":"cce631b6-d31a-4eab-afd6-34fd1779683d","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.856007Z"},"links":{"cited_paper":"/paper/2402.17910","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:3b4373f3222d6899671c4b8afe264ee6bcf3465754deed97b709ef084f043794","observation_id":"48833b0b-4ae8-44d3-bfc6-8257ab996257","resolution":{"observed_at":"2026-08-11T13:24:11.115958Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.757871Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"6466d347-8085-49aa-81b0-8751819c21b9","year":2017},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.862045Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:0ed7f3500c646fbd46ec7fd7f07aebc888658012f28bf3e8a6a4185baa975f13","observation_id":"c4077533-ba0d-4185-9691-e1d9fb5b37e2","resolution":{"observed_at":"2026-08-11T13:24:11.763704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.738626Z","title":"Diffusion model align- ment using direct preference optimization","venue":null,"work_id":"1d0fb249-1d82-45e3-bca5-d9159985b05a","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.867393Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:f3f612f2d1fe6d6aee4c664a023f0c49201c979f53f94a6210d56ea707ad8bc8","observation_id":"2af58fe0-a2bb-49ff-a225-1ebb4ae18b1e","resolution":{"observed_at":"2026-08-11T13:24:11.744672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07519","last_updated":"2024-02-02T16:15:22Z","snapshot_observed_at":"2026-08-16T06:18:35.139211Z","submitted_at":"2024-01-15T07:50:18Z","title":"InstantID: Zero-shot Identity-Preserving Generation in Seconds","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07519","snapshot_observed_at":"2026-08-11T13:24:10.872998Z","title":"Instantid: Zero-shot identity-preserving gener- ation in seconds","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.872998Z"},"links":{"cited_paper":"/paper/2401.07519","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:2ee872264ba4c2fc9df03578007ea8447e9158ed07d10ce3bb2990c5382d4b85","observation_id":"c18779da-4b1e-4b41-8dc5-041ad58897d8","resolution":{"observed_at":"2026-08-11T13:24:10.872998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.720945Z","title":"Tokencompose: Text-to-image diffusion with token-level supervision","venue":null,"work_id":"0c412934-29a5-4d78-9df5-3544957a62fa","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.879033Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:ce7f8cedf20330c8b474a5242ce6ab9402441647c8edcc387ce20888338245d8","observation_id":"b06c5731-26e4-485c-b211-79679011c61a","resolution":{"observed_at":"2026-08-11T13:24:11.726553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.700528Z","title":"Wang, Evan Montoya, David Munechika, Haoyang Yang, Benjamin Hoover, and Duen Horng Chau","venue":null,"work_id":"539bf76d-e1ea-47a6-9822-2fc1bc7e32a2","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.885079Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:bf4a089f7caf2efae5cb8908cb7a0f2d1cb50060fafed0042d6866a160f425af","observation_id":"86825bcd-31b4-4bb2-bab5-29cee8c5d78e","resolution":{"observed_at":"2026-08-11T13:24:11.707263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.675006Z","title":"Seesr: Towards semantics-aware real-world image super-resolution","venue":null,"work_id":"659bab4e-aee1-4975-9485-e0392b60d030","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.892075Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:95e0c4af7ca8e247a2b6c50eceff33e3410303ec8fc4c4fe4af17272d0567f70","observation_id":"38b1eb26-daf8-4186-8483-866e0cc81d01","resolution":{"observed_at":"2026-08-11T13:24:11.683919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.14284","last_updated":"2025-05-06T16:45:30Z","snapshot_observed_at":"2026-08-18T10:40:42.309079Z","submitted_at":"2023-11-24T05:17:01Z","title":"Paragraph-to-Image Generation with Information-Enriched Diffusion Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.14284","snapshot_observed_at":"2026-08-11T13:24:10.898243Z","title":"Paragraph-to-image gener- ation with information-enriched diffusion model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.898243Z"},"links":{"cited_paper":"/paper/2311.14284","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:fc3a41930ff74640c39aeba0612e633c8ea7f96c84fed72065f15813b66114e0","observation_id":"9e03bc59-2d8d-4e12-9b38-77f82f6d8d2a","resolution":{"observed_at":"2026-08-11T13:24:10.898243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09341","last_updated":"2023-09-25T08:19:23Z","snapshot_observed_at":"2026-08-14T02:46:25.655503Z","submitted_at":"2023-06-15T17:59:31Z","title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.09341","snapshot_observed_at":"2026-08-11T13:24:10.904255Z","title":"Human preference score v2: A solid benchmark for evaluating human preferences of text-to-image synthesis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.904255Z"},"links":{"cited_paper":"/paper/2306.09341","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:a29b4eb529cbac934b3934f48bfa118d1c4e40ff3a524f6be9292cd59fc1e11e","observation_id":"2252543c-929d-4168-9b12-723ad639e611","resolution":{"observed_at":"2026-08-11T13:24:10.904255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.652538Z","title":"Human preference score: Better aligning text-to- image models with human preference","venue":null,"work_id":"8e30f61f-b043-43ee-bc33-9ca3d1bf78ae","year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.912130Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:ae144188290162bf3acc53c4b8b47bd54b838c137d90dca096300687e94be6e2","observation_id":"4a59abcc-7d48-4683-af46-379e42f8f5b9","resolution":{"observed_at":"2026-08-11T13:24:11.658991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.629121Z","title":"Stylespace analysis: Disentangled controls for stylegan image genera- tion","venue":null,"work_id":"713851ac-45f4-44b9-af74-7adef171ed16","year":2021},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.923913Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:767481f38a82ed4dd5914189ff9d3a1dd07d8fdaf2529d6f4829a3a2ec88c1e7","observation_id":"83b53798-2f3a-4c30-8aee-6fc1295ada32","resolution":{"observed_at":"2026-08-11T13:24:11.637957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.611578Z","title":"Freeman, Fr ´edo Durand, and Song Han","venue":null,"work_id":"d5565994-bd73-4254-a477-31a5bafe65f7","year":2025},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.932289Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:a07c2cb9f5ecdf18d47c49ead90840dc0076d4e881276838b270c4d1e24263e6","observation_id":"afce0546-38c0-430b-bb9f-d0089fb09d6a","resolution":{"observed_at":"2026-08-11T13:24:11.617144Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.591611Z","title":"R&b: Region and boundary aware zero-shot grounded text-to-image generation","venue":null,"work_id":"d3e95c8d-3cef-4c97-8742-af7dbf97d2e1","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.938638Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:e8e56c8ffd0a7f54bcc3fa17b1ee826d57e2ebcaf06f096f86df5f3ac3baf505","observation_id":"d8b95086-b201-40a2-b605-d57702f4fcad","resolution":{"observed_at":"2026-08-11T13:24:11.598441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.944286Z","title":"Imagere- ward: Learning and evaluating human preferences for text- to-image generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.944286Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:8778f27b8034685ed164d7e98072fa3f7058e2982fac7ec1c365270ca13e1765","observation_id":"f5fb7fd4-67d3-4ae2-9623-e723dd1b9320","resolution":{"observed_at":"2026-08-11T13:24:10.944286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06721","last_updated":"2023-08-13T08:34:51Z","snapshot_observed_at":"2026-07-06T16:05:39.158819Z","submitted_at":"2023-08-13T08:34:51Z","title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06721","snapshot_observed_at":"2026-08-11T13:24:10.951063Z","title":"Ip- adapter: Text compatible image prompt adapter for text-to- image diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.951063Z"},"links":{"cited_paper":"/paper/2308.06721","citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:27192a7c3981df7dea99376b62a8d29220ff6b1c74e828c5c7742ce5c9cc7797","observation_id":"fc44424b-9431-4c8a-952b-f6c06be816fd","resolution":{"observed_at":"2026-08-11T13:24:10.951063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.557093Z","title":"Scaling up to excellence: Practicing model scaling for photo- realistic image restoration in the wild","venue":null,"work_id":"0c4eca99-63ef-4a79-a5d4-6f900ef3d7c3","year":2024},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.957503Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:fd5c2ed17d832197c9906788334b7fbab5fda5a88ada50876543b2679c5c9f8e","observation_id":"cd07df9d-f452-454e-bd4e-39cf1b910cbb","resolution":{"observed_at":"2026-08-11T13:24:11.563663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.534553Z","title":"Scaling autoregressive models for content-rich text-to-image generation","venue":null,"work_id":"05dc338b-4129-4552-81ae-97b4aca2429b","year":2022},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.962644Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:ab5f503d86c0302d7aff9666b19639b4a6887dc19ca2194b03840c1ab3e6bdec","observation_id":"b441700e-f494-4f96-b689-c2cb32f0e77e","resolution":{"observed_at":"2026-08-11T13:24:11.541812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:10.968045Z","title":"Adding conditional control to text-to-image diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.968045Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:3379eb3cebe4a97c84b11f5239df5557f86ca09947db0db0b40d9fe187aaf5fc","observation_id":"21779ad1-bf30-4bcf-b2da-782c78f30706","resolution":{"observed_at":"2026-08-11T13:24:10.968045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:24:11.503542Z","title":"A horse to the left of a bottle","venue":null,"work_id":"ced563a5-ab46-415c-86ed-fbab17c69f1b","year":2019},"citing_paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-11T13:24:10.973352Z"},"links":{"citing_paper":"/paper/2412.13195"},"observation_digest":"sha256:31b53964cff05fde41b4cebcc67cb61b39540b582cde38cbb9f766576e91b35f","observation_id":"9835b348-4bca-4b18-80fa-b8d68909d851","resolution":{"observed_at":"2026-08-11T13:24:11.509598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.13195","last_updated":"2025-08-25T17:59:59Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-21T16:44:01.787390Z","submitted_at":"2024-12-17T18:59:50Z","title":"CoMPaSS: Enhancing Spatial Understanding in Text-to-Image Diffusion Models"},"reference_resolution":{"displayed":85,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":28,"verified_exact":2,"verified_fuzzy":55},"total_outbound_references":85},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 85 of 85 outbound references and 0 inbound Pith citation observations for arXiv:2412.13195."}