{"as_of":"2026-08-07T05:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:901589a6d8fc163ebfddf98102e87e013910ab2b47b7621b4ebcd1972c6e18b6","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-21T17:40:25.779794Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2512.12598/citation-record","integrity":"/paper/2512.12598/integrity","json":"/paper/2512.12598/citation-record.json","paper":"/paper/2512.12598"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blended latent diffusion.ACM transactions on graphics (TOG), 42 (4):1–11","venue":null,"work_id":"d16a8046-cb24-42f7-8af9-915564c1da5d","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:cba72e86ea07ea796f5e0d5e6841656ac8dee34eab8392c2327217f6721c3065","observation_id":"2844178c-3cee-4f70-a0fe-721f8fcd312a","resolution":{"observed_at":"2026-05-21T17:44:17.765437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In- structpix2pix: Learning to follow image editing instructions","venue":null,"work_id":"273977e6-dd4b-4c1d-9f32-f13a08c04d96","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:f7f80a9721727f3e775fd845d97328628ced0b3663cbd378e52ea57aee811af1","observation_id":"3364098a-256e-482c-b447-f32ee2f8b51c","resolution":{"observed_at":"2026-05-21T17:44:17.770393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:d4457b2f07426284468aa6fc597b7741535c6d4dc825c191e4f16e810fd4e709","observation_id":"2803aa8c-fdd6-4a77-9472-ebf5efe062ec","resolution":{"observed_at":"2026-05-21T17:44:17.322134Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09344","last_updated":"2023-12-08T21:02:07Z","snapshot_observed_at":"2026-07-06T15:43:07.989730Z","submitted_at":"2023-06-15T17:59:50Z","title":"DreamSim: Learning New Dimensions of Human Visual Similarity using Synthetic Data","version":3},"cited_work":{"arxiv_id":"2306.09344","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.09344","snapshot_observed_at":"2026-07-04T16:29:56.758273Z","title":"Dreamsim: Learning new dimensions of human visual similarity using synthetic data","venue":"cs.CV","work_id":"7e705ece-244e-42aa-9d5e-eae3d77be181","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2306.09344","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:366cf560fc8fbd8fdab72cb49d49f0a1a59101811039ee3bece9402a40906bf7","observation_id":"c035275d-37b9-4f89-a01f-685f6ab1cd83","resolution":{"observed_at":"2026-05-27T02:04:59.423634Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.01618","last_updated":"2022-08-02T17:50:36Z","snapshot_observed_at":"2026-08-02T23:40:32.342515Z","submitted_at":"2022-08-02T17:50:36Z","title":"An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion","version":1},"cited_work":{"arxiv_id":"2208.01618","doi":"10.48550/arxiv.2208.01618","metadata_source":"pith","pith_arxiv_id":"2208.01618","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion","venue":"cs.CV","work_id":"ca618c21-3ba6-448e-bd86-bcecff3cdeb5","year":2022},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2208.01618","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:1ef8ea4561e1ad7cf7e1724acdd63607076b4137f7607fa1ca1f56e4b78f08c6","observation_id":"f8dc622f-c2d0-4f23-91a1-f73d8d8d23ac","resolution":{"observed_at":"2026-05-21T17:44:17.254045Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.04714","last_updated":"2021-05-10T23:51:14Z","snapshot_observed_at":"2026-07-06T11:08:10.746485Z","submitted_at":"2021-05-10T23:51:14Z","title":"Sample and Computation Redistribution for Efficient Face Detection","version":1},"cited_work":{"arxiv_id":"2105.04714","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2105.04714","snapshot_observed_at":"2026-07-04T03:19:31.625145Z","title":"Sample and computation redistribution for effi- cient face detection","venue":null,"work_id":"77e21ba3-b09c-4e54-9517-c3e970070311","year":2021},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2105.04714","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:caf60211b69ea14bc95dac672c6cb087e1005386ce1a124b3d7b155eeff7e11a","observation_id":"639a2cd5-34ff-49e8-a34f-bd4f58569c6c","resolution":{"observed_at":"2026-05-21T17:44:17.273074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02101","last_updated":"2025-03-13T18:35:06Z","snapshot_observed_at":"2026-07-06T17:54:39.685251Z","submitted_at":"2024-04-02T16:52:41Z","title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","version":2},"cited_work":{"arxiv_id":"2404.02101","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.02101","snapshot_observed_at":"2026-07-10T01:46:41.167891Z","title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","venue":"cs.CV","work_id":"1c05c278-c023-4ef0-a359-25a41f1065eb","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2404.02101","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:ee5db0e6070a4e51ea295579efb0168b75324a92075d1973cd810ee6a6cf1d69","observation_id":"a1f1f424-73e5-49b9-b62e-c732a363e8b7","resolution":{"observed_at":"2026-05-21T17:44:17.230172Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01006","last_updated":"2026-01-01T13:07:25Z","snapshot_observed_at":"2026-08-03T18:50:30.558321Z","submitted_at":"2025-07-01T17:55:04Z","title":"GLM-4.5V and GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","version":6},"cited_work":{"arxiv_id":"2507.01006","doi":"10.48550/arxiv.2507.01006","metadata_source":"pith","pith_arxiv_id":"2507.01006","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLM-4.5V and GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","venue":"cs.CV","work_id":"366607ba-e4ea-4726-98c3-63356e32351c","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2507.01006","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:ab9ce99ba38f2192b53d14d3c5cfc79998fda955489b894a5b823cc26691b1ae","observation_id":"e4860a6b-921b-4252-b950-e8b27b77849b","resolution":{"observed_at":"2026-05-21T17:44:17.309434Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Magicfight: Personalized martial arts combat video generation","venue":null,"work_id":"17887771-bb1c-4143-894d-f64fb2e4b7d3","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:38169289e7128fbc9f7aac872d161ffc13c09a91dfe9edd59c78ee13ece67660","observation_id":"590ac162-0ee6-480c-8f2f-68d4e0d7c64d","resolution":{"observed_at":"2026-05-21T17:44:17.744291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dual-schedule inver- sion: Training-and tuning-free inversion for real image edit- ing","venue":null,"work_id":"f221e2f9-503c-407e-8956-75f3c302f76c","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:5e5a691d9176f0e46dab2f7a5cd77915b28248b7a023510c48d0c08726416655","observation_id":"10b5969a-2a63-4ab0-9095-38db36b4dd4e","resolution":{"observed_at":"2026-05-21T17:44:17.783642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10915","last_updated":"2026-07-10T04:56:00Z","snapshot_observed_at":"2026-08-07T04:11:40.231309Z","submitted_at":"2025-06-12T17:29:40Z","title":"M4V: Multimodal Mamba for Efficient Text-to-Video Generation","version":2},"cited_work":{"arxiv_id":"2506.10915","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.10915","snapshot_observed_at":"2026-07-08T14:44:59.794530Z","title":"M4v: Multi-modal mamba for text-to-video generation","venue":"cs.CV","work_id":"42351ceb-2a18-4578-ba0d-b855394cb673","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2506.10915","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:6c3e893e7509a47c08420ea0b116bc41f3bea86bd7ff297091351554a01a763d","observation_id":"d5ca2794-f90d-48c7-bbc9-01ec5835e62d","resolution":{"observed_at":"2026-05-21T17:44:17.287078Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gen- erative photography: a systematic, constructive approach","venue":null,"work_id":"479fa179-3ab2-43ab-b81f-e674a5987620","year":1986},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:3ae590d484a7af397c0a4f6fd2e9ca24e5cb8107517d843b165895efc7aede0c","observation_id":"0c31571f-3db4-4970-8ff4-da88bc5e68ee","resolution":{"observed_at":"2026-05-21T17:44:17.769801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Brushnet: A plug-and-play image inpainting model with decomposed dual-branch diffusion","venue":null,"work_id":"be85f541-a905-4994-810d-1463b880e957","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:bedc038cef10c67431ca2918f0c16ae78515e2975f5f2ccdb25f49d61d9f7504","observation_id":"19644532-0b24-476e-befa-e0403aa474a0","resolution":{"observed_at":"2026-05-21T17:44:17.788880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"interactive sto- rytelling","venue":null,"work_id":"6343fd0a-0674-4a79-bf1c-0e027c3cd798","year":2012},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:8a44cac85e85eaf97207245bfb4778dcf55da5672289a3711af1eb5b18cc3a23","observation_id":"8abcd290-c764-402f-ad01-c89cb4e6c805","resolution":{"observed_at":"2026-05-21T17:44:17.786130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03603","last_updated":"2025-03-11T08:14:25Z","snapshot_observed_at":"2026-08-03T00:44:01.942521Z","submitted_at":"2024-12-03T23:52:37Z","title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","version":6},"cited_work":{"arxiv_id":"2412.03603","doi":"10.48550/arxiv.2412.03603","metadata_source":"pith","pith_arxiv_id":"2412.03603","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","venue":"cs.CV","work_id":"881efa7e-7e73-4c66-9cc3-2803e551061c","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2412.03603","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:ca2b0d4aca682e14fe49f5ac9763fabe649e96007a7dd2b172d7725c98ec1ab7","observation_id":"31881e81-69eb-4538-ae80-d1fe6afc964d","resolution":{"observed_at":"2026-05-21T17:44:17.307598Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T21:18:49.433177+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T21:18:49.433177+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15742","last_updated":"2025-06-24T05:31:03Z","snapshot_observed_at":"2026-07-06T21:44:22.901650Z","submitted_at":"2025-06-17T20:18:23Z","title":"FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space","version":2},"cited_work":{"arxiv_id":"2506.15742","doi":"10.48550/arxiv.2506.15742","metadata_source":"pith","pith_arxiv_id":"2506.15742","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space","venue":"cs.GR","work_id":"5dfe19d5-3541-4803-8fe9-3c8b9e29b281","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2506.15742","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:ff017c270c5c6722040acdfbfd6b4a850470a384f24e145f5c52ce2747e9d1e8","observation_id":"e7ddcf0b-6cd8-4564-b092-5fee30770c12","resolution":{"observed_at":"2026-05-21T17:44:17.300854Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-25T00:53:20.581238+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T00:53:20.581238+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Control-nerf: Editable feature volumes for scene rendering and manipulation","venue":null,"work_id":"ac0e7a87-f0e6-4ca3-8f26-262a8060a390","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:072de97051c62578308ad230acf2a25c199744ddc117b056b3c795b5d853d3b1","observation_id":"6d38be05-6e83-4c50-b083-5ed8096805f6","resolution":{"observed_at":"2026-05-21T17:44:17.783240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mat: Mask-aware transformer for large hole im- age inpainting","venue":null,"work_id":"baa1339d-e22e-4b68-afbe-eb7e3b4175e7","year":2022},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:3436b421232bc222909dd9b1e36179d34d9cc5d97e5aee5e20f205e1a5d0d04c","observation_id":"98d4c47c-a654-4eb9-844b-086c226d4ac7","resolution":{"observed_at":"2026-05-21T17:44:17.741692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Storygan: A sequential conditional gan for story vi- sualization","venue":null,"work_id":"a54e3fee-ebc3-468f-a548-fc74af250710","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:1af778cad39f69e45c07de5b20941ffc60ef87be04abad48e9adb8c4754f60c7","observation_id":"aed01b9a-b1cd-45b9-a615-d6a59c1f116a","resolution":{"observed_at":"2026-05-21T17:44:17.772917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Photomaker: Customizing re- alistic human photos via stacked id embedding","venue":null,"work_id":"ed0bc3ad-15be-473c-9c8e-e2702fe6dd0a","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:7aabffaec8a8f83c25b6c54ae48033799b93d88779c18be7177aa89d0c9b494d","observation_id":"3ea5290d-c31f-4f54-85b0-3073d548e776","resolution":{"observed_at":"2026-05-21T17:44:17.729986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.16888","last_updated":"2025-11-04T13:15:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-19T15:38:06Z","title":"Uniworld-V2: Reinforce Image Editing with Diffusion Negative-aware Finetuning and MLLM Implicit Feedback","version":3},"cited_work":{"arxiv_id":"2510.16888","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.16888","snapshot_observed_at":"2026-07-05T16:51:14.250864Z","title":"Uniworld-V2: Reinforce Image Editing with Diffusion Negative-aware Finetuning and MLLM Implicit Feedback","venue":"cs.CV","work_id":"9687bd6c-c4f6-4f49-ba59-ca7e825f0710","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2510.16888","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:2657e0a556695eb6b01da6b7c1a1c79c2a63a21de4cba471fbe6901a664c2093","observation_id":"842264af-e7b6-4e5e-a05f-bf771f7ee1b6","resolution":{"observed_at":"2026-05-21T18:01:19.940440Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dl3dv-10k: A large-scale scene dataset for deep learning-based 3d vision","venue":null,"work_id":"47b9d076-e3ba-4b83-9215-c4efbf0aa880","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:406624aa95cd2c0de7cc56c5b74e13d3195ee787ae1c1e7fe131c5208d3666f9","observation_id":"7097b422-ffb6-4bf8-882f-4556a9741dea","resolution":{"observed_at":"2026-05-21T17:44:17.767051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":"86cd8a5e-b7c1-4e2d-b984-5fd52ef49c3c","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:c29a81a9a7cad6cb8a81203f0cd6965a5b013c7295920291f23397fea3182ec9","observation_id":"15c90e28-f2ba-4603-93d6-841558ab69de","resolution":{"observed_at":"2026-05-21T17:44:17.776273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2410.06244","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Story-adapter: A training-free iterative framework for long story visualization","venue":null,"work_id":"5b137cec-538a-4374-a20e-c46d55dbee8d","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:a3c4620c1196ed86bb9eda9a334defe99bce87033bcaa31b790804f02a9ef768","observation_id":"66f18dc7-0205-4019-abec-de79b85843e5","resolution":{"observed_at":"2026-05-21T17:44:17.238550Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Synthesizing coherent story with auto-regressive la- tent diffusion models","venue":null,"work_id":"c9102408-6f13-4e5a-a2ce-903c68367e8e","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:a1efa7a3fb9830de0cd27ea25c927c47ee4e65cb0e0cbe5da251e4611c28e416","observation_id":"8d868800-42b8-4844-ac60-22cd681e4935","resolution":{"observed_at":"2026-05-21T17:44:17.740081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Make-a-story: Visual memory conditioned consistent story generation","venue":null,"work_id":"7d547389-8aa3-4689-a86d-d8db8071d86c","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:9d5f26168831e9bbece49d0043129b46742d6852a9f3a8e2decc96705f12c130","observation_id":"b06f6723-4f7e-4727-a808-28db816a4670","resolution":{"observed_at":"2026-05-21T17:44:17.761412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":"2408.00714","doi":"10.1038/s41598-025-97590-3","metadata_source":"pith","pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SAM 2: Segment Anything in Images and Videos","venue":"cs.CV","work_id":"acc13f66-d814-44f9-9688-375688bf2d4a","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:33cb44a5ea1bbcfd804b9c223d8ea794e4f26dabb5e13e7a6e5690de37209295","observation_id":"52fac083-52f7-4f69-9ea1-d2cdba25d432","resolution":{"observed_at":"2026-05-21T17:44:17.296344Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minima: Modality invariant im- age matching","venue":null,"work_id":"98c6b611-b5e8-4bd6-b097-d15d219fd6d0","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:2feef6b0dc76ac50137205ae53e843b680567836325d83ebf154fc321cec8910","observation_id":"0f488faf-3791-47e2-837f-bfcc08d49bdc","resolution":{"observed_at":"2026-05-21T17:44:17.791206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.20427","last_updated":"2025-12-10T16:37:54Z","snapshot_observed_at":"2026-07-06T22:30:41.141182Z","submitted_at":"2025-09-24T17:59:04Z","title":"Seedream 4.0: Toward Next-generation Multimodal Image Generation","version":3},"cited_work":{"arxiv_id":"2509.20427","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.20427","snapshot_observed_at":"2026-07-11T00:07:42.259448Z","title":"Seedream 4.0: Toward Next-generation Multimodal Image Generation","venue":"cs.CV","work_id":"15c839a0-48a3-4218-82b6-cac5b7f66e13","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2509.20427","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:c177acdd052eeb2f3f64e254052293a9fc7f2e7642c26b02059bebaf5fa074d2","observation_id":"73bf747c-a7dd-4497-81a9-04ee421d4353","resolution":{"observed_at":"2026-05-21T17:44:17.295648Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2410.20084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T11:56:55.111891Z","title":"Univst: A unified framework for training-free localized video style transfer","venue":null,"work_id":"fa3724a6-6154-4087-9354-ba7cb593b53a","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:c7202e5d11fd260259766c3156c1f5bcd5aefa8b74257feee3854f7e441b1677","observation_id":"575b06ba-c52c-4ca8-8c2c-ee13f50d0be6","resolution":{"observed_at":"2026-05-21T17:44:17.313773Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.22994","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T21:00:08.929397Z","title":"Scenedecorator: Towards scene-oriented story generation with scene planning and scene consistency","venue":null,"work_id":"ed482ccc-78af-485d-8ca7-34e1a7bdcd1e","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:b73f5cfd5843bffed761669ba3a07e525fe5a4bc9b2acc5dcc3bfbf6fa5bbed6","observation_id":"d1ecbc1f-a54f-40d6-876f-e69a19975c00","resolution":{"observed_at":"2026-05-21T17:44:17.263091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20314","last_updated":"2025-04-19T02:22:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T08:25:43Z","title":"Wan: Open and Advanced Large-Scale Video Generative Models","version":2},"cited_work":{"arxiv_id":"2503.20314","doi":"10.1109/19.492748","metadata_source":"pith","pith_arxiv_id":"2503.20314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Wan: Open and Advanced Large-Scale Video Generative Models","venue":"cs.CV","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2503.20314","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:7759d59529f22fd14499c7583a2fa2af76e2f16d58656dd45cb297c179813ffc","observation_id":"40e085fd-de90-4e5b-bb83-2f80c7408c6d","resolution":{"observed_at":"2026-05-21T17:44:17.304699Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vistadream: Sampling multiview consistent images for single-view scene reconstruction","venue":null,"work_id":"2c71e446-6cc9-4985-9d0f-b54990583c07","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:0d55c6caea33e213a3d63ba180d94b089157d8343aebd39d9b07f510fae4dacc","observation_id":"fcee3dba-e2a2-4070-bc13-31d83bf5ff90","resolution":{"observed_at":"2026-05-21T17:44:17.781464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07519","last_updated":"2024-02-02T16:15:22Z","snapshot_observed_at":"2026-08-06T23:08:20.817133Z","submitted_at":"2024-01-15T07:50:18Z","title":"InstantID: Zero-shot Identity-Preserving Generation in Seconds","version":2},"cited_work":{"arxiv_id":"2401.07519","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.07519","snapshot_observed_at":"2026-07-04T10:19:46.873019Z","title":"InstantID: Zero-shot Identity-Preserving Generation in Seconds","venue":"cs.CV","work_id":"85490b0d-f13f-4217-a587-51a62742c242","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2401.07519","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:a9cc3bafeab377272f2e1615c59c28f1bead6c58db200f5571564f1581979f6d","observation_id":"4a6deffd-c08a-4aad-ada4-d769ce313d79","resolution":{"observed_at":"2026-05-21T17:44:17.303492Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.01770","last_updated":"2024-10-30T17:05:17Z","snapshot_observed_at":"2026-07-06T16:14:09.494600Z","submitted_at":"2023-09-04T19:16:46Z","title":"StyleAdapter: A Unified Stylized Image Generation Model","version":2},"cited_work":{"arxiv_id":"2309.01770","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.01770","snapshot_observed_at":"2026-07-04T00:19:12.753148Z","title":"Styleadapter: A single-pass lora-free model for stylized image generation","venue":null,"work_id":"a8e26ccf-6859-49ad-9ec5-abbf1676f114","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2309.01770","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:d20c79c46f6de08e162f9f0cb2dba894e3b5feffd0744a26432075dce5c85064","observation_id":"73a8ebd4-e4e9-4a54-8ae7-e6d9f5933f83","resolution":{"observed_at":"2026-05-21T17:44:17.277647Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omniedit: Building image edit- ing generalist models through specialist supervision","venue":null,"work_id":"21afb06e-5fbc-417a-8a13-e5a0e6aa277a","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:bfd79f9f88032ff4f5218c1b982059befc59682b95e34a1a93a90a0b88a484e6","observation_id":"2582e4a0-46ba-4846-9ff8-f45a23341e86","resolution":{"observed_at":"2026-05-21T17:44:17.730162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.02324","last_updated":"2025-08-04T11:49:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-04T11:49:20Z","title":"Qwen-Image Technical Report","version":1},"cited_work":{"arxiv_id":"2508.02324","doi":"10.48550/arxiv.2508.02324","metadata_source":"pith","pith_arxiv_id":"2508.02324","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-Image Technical Report","venue":"cs.CV","work_id":"d06d7ecc-7579-4f89-a60b-4278a0f3c562","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2508.02324","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:580437071cc01af3445ed7575d269aa2f2b13689f40a4a00bad9d5a59c8c9d2a","observation_id":"4e5e305c-f6e1-4e1c-8dd9-bd9181758744","resolution":{"observed_at":"2026-05-21T17:44:17.299328Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-21T15:23:05.016381+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T15:23:05.016381+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.06679","doi":"10.48550/arxiv.2510.06679","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Dreamomni2: Multimodal instruction-based editing and generation","venue":"ArXiv.org","work_id":"caa956fb-4bc6-4cba-ab91-329773bba8a1","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:c23ffe2d48186cdf31cd9bd731c3c101e31fdefa26c54bebb2294510813a8b84","observation_id":"bf524a49-a8da-44b0-959e-b453241e66c5","resolution":{"observed_at":"2026-05-21T17:44:17.272745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fastcomposer: Tuning-free multi- subject image generation with localized attention.Interna- tional Journal of Computer Vision, 133(3):1175–1194","venue":null,"work_id":"998eff4b-7d1a-4f0d-9025-70b1ef22ca80","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:ca3378e377963eb0e934042e5585091ec0f8b19c432655fa02711eb2e286aa77","observation_id":"4e845e5c-996a-4193-87fa-1711cdc680b8","resolution":{"observed_at":"2026-05-21T17:44:17.746047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Smartbrush: Text and shape guided object inpainting with diffusion model","venue":null,"work_id":"eb680884-911e-4fe1-8d7f-e0521547a869","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:e94ba3206467da6872de373b35f7b3f4e8ea820ad735c00b89e0b5edb8eae273","observation_id":"6f842eac-a479-4d7d-8772-569316e94ea3","resolution":{"observed_at":"2026-05-21T17:44:17.752896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Paint by example: Exemplar-based image editing with diffusion mod- els","venue":null,"work_id":"9a36d696-ba2f-482c-8296-a7f35bd24cf7","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:1af64757053cd8c2637478908005cde1296f33c560fcdeb972c91c695935a074","observation_id":"da4fe5ae-57e4-401c-9490-152adbc31aa9","resolution":{"observed_at":"2026-05-21T17:44:17.764646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Seed-story: Multi- modal long story generation with large language model","venue":null,"work_id":"ff2f416e-6e8d-4a6a-af9d-934e3377cdad","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:15191783f7022b0e7497d2461842f8bb237dbebdf1dd81e27e55104649a00c93","observation_id":"a00ac5bc-9c3e-4466-94d2-42db3726b2ff","resolution":{"observed_at":"2026-05-21T17:44:17.778944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.01215","last_updated":"2025-08-02T06:17:23Z","snapshot_observed_at":"2026-08-06T05:47:30.553688Z","submitted_at":"2025-08-02T06:17:23Z","title":"StyDeco: Unsupervised Style Transfer with Distilling Priors and Semantic Decoupling","version":1},"cited_work":{"arxiv_id":"2508.01215","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.01215","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Stydeco: Unsupervised style transfer with distilling priors and semantic decoupling","venue":null,"work_id":"11ae2348-a7f3-49dc-942e-9dba3bf6b3ac","year":2025},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2508.01215","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:7d62f9f315a3661d2cd1c6457c40727edbe02567c9c57dafe24fc7c356468df4","observation_id":"2826e154-e4a0-46e7-a598-f91ff5931322","resolution":{"observed_at":"2026-05-21T17:44:17.229086Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.06072","last_updated":"2025-03-26T08:33:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-12T11:47:11Z","title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","version":3},"cited_work":{"arxiv_id":"2408.06072","doi":"10.48550/arxiv.2408.06072","metadata_source":"pith","pith_arxiv_id":"2408.06072","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","venue":"cs.CV","work_id":"f38fc088-12aa-4bf4-9ecd-08d3e797ccb7","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2408.06072","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:47c8a408758d223a5f0c1a7304670f361d217784b5cd7b25ef61c0e1477523bd","observation_id":"8cf86e30-ea07-4320-ae70-47f65bc33c2d","resolution":{"observed_at":"2026-05-21T17:44:17.258090Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06721","last_updated":"2023-08-13T08:34:51Z","snapshot_observed_at":"2026-07-06T16:05:39.158819Z","submitted_at":"2023-08-13T08:34:51Z","title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","version":1},"cited_work":{"arxiv_id":"2308.06721","doi":"10.48550/arxiv.2308.06721","metadata_source":"pith","pith_arxiv_id":"2308.06721","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","venue":"cs.CV","work_id":"98e51b10-54bd-4251-8a2d-f79bd6215c19","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2308.06721","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:5b9203c521fa8d20d04960b4852b085b39dbd6c52aa16a62608954e6c4444f56","observation_id":"44ccc016-c7b9-4a2e-8825-55f1bacf5356","resolution":{"observed_at":"2026-05-21T17:44:17.317961Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:20:08.139662+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:20:08.139662+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Adding conditional control to text-to-image diffusion models","venue":null,"work_id":"599bfa67-f24d-4f2f-898e-e990d1185484","year":2023},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:8dcf66e778bf2a46bf52c7cfd7ccc5a45026d1248fd00d067a4626483598d46b","observation_id":"fddcb322-f1ab-4db3-a656-b2b8bddb9432","resolution":{"observed_at":"2026-05-21T17:44:17.767669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Places: A 10 million image database for scene recognition.IEEE Transactions on Pattern Analy- sis and Machine Intelligence","venue":null,"work_id":"055d3b12-dfd9-4d90-a0b0-e1c7c35a2e5d","year":2017},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:b6eb04d7a0d580d68bd4680b24a944c39ce19dfc079641c977cdcb1ab7a40d55","observation_id":"7118e286-56f0-4687-930b-df42c32a035d","resolution":{"observed_at":"2026-05-21T17:44:17.733231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13370","last_updated":"2025-05-21T08:18:59Z","snapshot_observed_at":"2026-07-06T19:35:08.846184Z","submitted_at":"2024-10-17T09:22:53Z","title":"MagicTailor: Component-Controllable Personalization in Text-to-Image Diffusion Models","version":3},"cited_work":{"arxiv_id":"2410.13370","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.13370","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Magictailor: Component-controllable person- alization in text-to-image diffusion models","venue":null,"work_id":"049448a3-06e5-4317-9126-a970db810777","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"cited_paper":"/paper/2410.13370","citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:74181bd94298f1e9f994ab244a299071bdd4244f36c0920710e5ee1b7e26ccc0","observation_id":"7fc60c39-ea15-4488-9f55-8cdf754ddc24","resolution":{"observed_at":"2026-05-21T17:44:17.258824Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Storydiffusion: Consistent self- attention for long-range image and video generation.Ad- vances in Neural Information Processing Systems, 37: 110315–110340","venue":null,"work_id":"4560a0b7-da8f-4edd-a604-8d6af8bb87de","year":2024},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:343d61ad88126d736594f862257def2c8301f43c6313f0d4ad15552d56892219","observation_id":"771a5c95-6867-4118-881b-8355346d50e2","resolution":{"observed_at":"2026-05-21T17:44:17.773338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Text-Image Alignment Metric Selection For evaluating text–image alignment, we compare two met- rics:CLIP-T[5] andGemini 2.5 Flash Text–Image Alignment (G2.5F-TIA)[3]","venue":null,"work_id":"af3128c6-28d5-4441-aee0-17244fc111c8","year":1987},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:2e44b92e970022f32631a6f904243050bde7e8a18f743bb5345a5d0e192ce6d5","observation_id":"177b6957-b887-41a9-abc5-866496e535d9","resolution":{"observed_at":"2026-05-21T17:44:17.775927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5d9097bc-b4fb-42e8-8798-d6f12d9416e9","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:9a3a5915d1c5e97d08e5fb6eca6a1ac9131a2fc2e5fa37687e397b7e85f93f6c","observation_id":"bda63fbf-be93-4162-afb9-06fc53c6e0b8","resolution":{"observed_at":"2026-05-21T17:44:17.743029Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e345f7b3-f09a-4185-870b-2de4ad6bca26","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:e37c16fe48cfdb15f729a8261f6043b0243b107894f4723792af87abc672826d","observation_id":"09643a59-d257-4487-9166-14b10ed2c5c7","resolution":{"observed_at":"2026-05-21T17:44:17.714839Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We explicitly describe the Gemini 2.5 Flash prompts used for automatic scoring and the annotation interface shown to hu- man raters","venue":null,"work_id":"5aa6f416-d232-4e53-b688-b7b9de0dc232","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:ea556dfce93f194136b2e2f4db504f140e653a5d96b655620641b13d5aa08719","observation_id":"70691daa-4901-49e6-b932-948e2f3d5a96","resolution":{"observed_at":"2026-05-21T17:44:17.712257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"These videos were synthesized using the Kling image-to-video model, utilizing keyframes produced by our method","venue":null,"work_id":"1a3e20b4-78a9-4738-9db6-3884c7d051e3","year":null},"citing_paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-21T17:40:25.779794Z"},"links":{"citing_paper":"/paper/2512.12598"},"observation_digest":"sha256:24b600ad3c6f392ce2182173c1b69a76b1908b397863b94399587377cdd0a5b2","observation_id":"1ce9a379-2b50-4616-b02b-34c5e603b35d","resolution":{"observed_at":"2026-05-21T17:44:17.717715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2512.12598","last_updated":"2026-05-18T03:43:51Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T05:49:44.908815Z","submitted_at":"2025-12-14T08:35:04Z","title":"Setting the Stage: Text-Driven Scene-Consistent Image Generation"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":2,"verified_exact":24,"verified_fuzzy":28},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 0 inbound Pith citation observations for arXiv:2512.12598."}