{"as_of":"2026-08-08T07:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c1725cacc2c9a72f829c4bf571b7a1bc3c2a5127c16da0bbe0754406488ee339","coverage":[{"denominator":62,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":62,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T04:37:17.479761Z","state":"measured"},{"denominator":67,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":67,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T22:36:08.191484Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T18:35:00.337951Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.03320","snapshot_observed_at":"2026-08-04T22:36:08.191484Z","title":"Skywork unipic: Unified autoregressive modeling for visual understanding and generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07295","last_updated":"2026-06-25T06:17:40Z","snapshot_observed_at":"2026-08-04T22:36:03.033298Z","submitted_at":"2025-09-08T23:59:32Z","title":"Reconstruction Alignment Improves Unified Multimodal Models","version":4},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-04T22:36:08.191484Z"},"links":{"cited_paper":"/paper/2508.03320","citing_paper":"/paper/2509.07295"},"observation_digest":"sha256:76df4c65e8cd2f332d0661a8b4c609409b6e0551b8be8f44e23d1cf25cc9b99f","observation_id":"7f283a41-9c8b-416a-965e-d63576a62cf8","resolution":{"observed_at":"2026-08-04T22:36:08.191484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"cited_work":{"arxiv_id":"2508.03320","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.03320","snapshot_observed_at":"2026-06-30T18:35:00.337951Z","title":"arXiv:2508.03320 (2025) 2","venue":null,"work_id":"7f1ecc8a-1a26-4630-80e1-31a91ad72f41","year":2025},"citing_paper":{"arxiv_id":"2605.12305","last_updated":"2026-05-12T15:54:49Z","snapshot_observed_at":"2026-07-06T23:24:03.984179Z","submitted_at":"2026-05-12T15:54:49Z","title":"Images in Sentences: Scaling Interleaved Instructions for Unified Visual Generation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T05:48:04.997796Z"},"links":{"cited_paper":"/paper/2508.03320","citing_paper":"/paper/2605.12305"},"observation_digest":"sha256:b0720f52c66b2009aec175dabdac2956298ce55a0fd275eb587957bcb344fa29","observation_id":"1a4ee54e-4b6d-44b7-b286-383a13866abd","resolution":{"observed_at":"2026-05-13T05:52:22.732596Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"cited_work":{"arxiv_id":"2508.03320","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.03320","snapshot_observed_at":"2026-06-30T18:35:00.337951Z","title":"arXiv:2508.03320 (2025) 2","venue":null,"work_id":"7f1ecc8a-1a26-4630-80e1-31a91ad72f41","year":2025},"citing_paper":{"arxiv_id":"2605.18714","last_updated":"2026-06-25T13:57:35Z","snapshot_observed_at":"2026-07-06T23:29:33.702647Z","submitted_at":"2026-05-18T17:46:46Z","title":"Semantic Generative Tuning for Unified Multimodal Models","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-20T11:32:24.007847Z"},"links":{"cited_paper":"/paper/2508.03320","citing_paper":"/paper/2605.18714"},"observation_digest":"sha256:aa2a9ee87ae8aaa827ac5ae4dbba3309ba7abbe66966f13d729c8ae76b159425","observation_id":"bb7c23c2-9f76-481c-af3d-8be051a7936a","resolution":{"observed_at":"2026-05-20T11:33:14.193044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"cited_work":{"arxiv_id":"2508.03320","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.03320","snapshot_observed_at":"2026-06-30T18:35:00.337951Z","title":"arXiv:2508.03320 (2025) 2","venue":null,"work_id":"7f1ecc8a-1a26-4630-80e1-31a91ad72f41","year":2025},"citing_paper":{"arxiv_id":"2605.18714","last_updated":"2026-06-25T13:57:35Z","snapshot_observed_at":"2026-07-06T23:29:33.702647Z","submitted_at":"2026-05-18T17:46:46Z","title":"Semantic Generative Tuning for Unified Multimodal Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-30T18:31:10.578558Z"},"links":{"cited_paper":"/paper/2508.03320","citing_paper":"/paper/2605.18714"},"observation_digest":"sha256:f57660da769714e04df50e190e516196d01a254c4ff34925ab0de630240993bf","observation_id":"76a3f3d9-c181-4765-9b82-cd61480e4c65","resolution":{"observed_at":"2026-06-30T18:35:00.339365Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.03320","snapshot_observed_at":"2026-07-14T06:20:23.251121Z","title":"arXiv preprint arXiv:2508.03320 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11199","last_updated":"2026-07-13T07:50:54Z","snapshot_observed_at":"2026-07-16T23:19:15.382857Z","submitted_at":"2026-07-13T07:50:54Z","title":"DynEval: Holistic Evaluations of T2I Generative Models in the Wild","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-07-14T06:20:23.251121Z"},"links":{"cited_paper":"/paper/2508.03320","citing_paper":"/paper/2607.11199"},"observation_digest":"sha256:700e5bd088345af8b6825714de0520f91a716aac7b4833c98e12e2834b7e7700","observation_id":"cc4a3f82-49ea-47fb-b55a-ea0362cdefa8","resolution":{"observed_at":"2026-07-14T06:20:23.251121Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.03320/citation-record","integrity":"/paper/2508.03320/integrity","json":"/paper/2508.03320/citation-record.json","paper":"/paper/2508.03320"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.624054Z","title":"Stable diffusion 3 medium: Multimodal diffusion transformer for photorealistic text-to-image generation","venue":null,"work_id":"2d61f721-3e3c-47f0-96f3-10979b2857ed","year":null},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:12.901562Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:56477ab47c965b1b16488bc1bb56fe2515169b3d66192fd8e89d01f257c6bb5b","observation_id":"b515eb6d-4114-4fd1-bfdb-87df133c1b3b","resolution":{"observed_at":"2026-08-06T04:37:23.627086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.616699Z","title":"Deep- speed inference: Enabling efficient inference of transformer models at unprecedented scale,","venue":null,"work_id":"22feb624-12b1-4be7-9f65-8c5bea1a70a5","year":null},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:12.983546Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:92bc09216bba35c0efc9cf7fecd35b12db049d20d63b8b9d95cd511f3ee07411","observation_id":"01e15f5e-6172-4d88-891a-acdb111450a2","resolution":{"observed_at":"2026-08-06T04:37:23.619238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.609367Z","title":"Humanedit: A high-quality human-rewarded dataset for instruction-based image editing,","venue":null,"work_id":"7e3faedd-02e1-49c1-8962-881a6305ac75","year":null},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.048249Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:2683ace02f3ca6a943199e311b7f371253a3e8d143a8ae62846fafaca1ad5ee6","observation_id":"95c4855d-c359-4a88-bac7-9fa6828dfbf0","resolution":{"observed_at":"2026-08-06T04:37:23.611809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.601570Z","title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond, 2023","venue":null,"work_id":"12fdb7d1-40af-4c4c-8f53-7b95d854f7a9","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.097005Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:80ec8822d18de033027961ed2ad1c44bdc8cfd043221f58dfbb33ef3bb0c60e5","observation_id":"6ae2157d-2c25-41c0-a229-bc638c6c16d9","resolution":{"observed_at":"2026-08-06T04:37:23.604363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.595760Z","title":null,"venue":null,"work_id":"268d286a-c0f0-490a-9111-21e51010fe8e","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.126370Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:265b4e6f99ec3f8a4d2176353f2f252bf6c3ddda1b854025bf1cc3cba39187c0","observation_id":"3dbe9763-d573-4966-a22f-00fd03d180f4","resolution":{"observed_at":"2026-08-06T04:37:23.597772Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.589387Z","title":"Blip3-o: A family of fully open unified multimodal models-architecture, training and dataset, 2025","venue":null,"work_id":"a455b45d-7846-4225-99ca-7821b3370e23","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.189222Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:43887a70232c8c69dc0f276175073a6d4ecc135dbdbb2cba33be1666d69ec546","observation_id":"fbdd849f-aedd-476d-bb32-946998d71609","resolution":{"observed_at":"2026-08-06T04:37:23.591896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.582120Z","title":"Pixart- � : Fast training of diffusion transformer for photorealistic text-to-image synthesis, 2023","venue":null,"work_id":"390bde8f-5fff-40da-9f55-8da33120715c","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.259464Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:79b22fec4d080e84a1664548db0249ef20f1e848b815f71d2d662db622bc4e8c","observation_id":"ba5a460c-df6a-4335-8586-d3f98d36e231","resolution":{"observed_at":"2026-08-06T04:37:23.585150Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.575703Z","title":"Janus-pro: Unified multimodal understanding and generation with data and model scaling, 2025","venue":null,"work_id":"70bf7479-70bf-4585-8c39-1944ab5ed43a","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.353616Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:c92725fb2adc1c1179113946230b253f378a513d3ea524055da2666fdc499385","observation_id":"6002f7f2-1aae-49f7-8e26-d9573ebd0ba5","resolution":{"observed_at":"2026-08-06T04:37:23.578062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.568481Z","title":"Emerging properties in unified multimodal pretraining, 2025","venue":null,"work_id":"1ffebab7-c87b-4ad6-acb7-d1c79d35cbeb","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.410153Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:b1bec16781fbef898a388352b36390153678482eb23dc161d62239ceec584fcf","observation_id":"f22a5152-ee28-4678-9cc5-d1bc09089fd0","resolution":{"observed_at":"2026-08-06T04:37:23.570897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.562237Z","title":"Taming transformers for high-resolution image synthesis, 2021","venue":null,"work_id":"368027b8-90e9-49d6-b175-871bfda55959","year":2021},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.462788Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:0222d2ab1bcebe4ce132b1c5a657ffa4fd96bb46a748e7b2975c59ef807f37fd","observation_id":"9aa3efeb-b9bf-412a-816e-7f7e53cea51d","resolution":{"observed_at":"2026-08-06T04:37:23.564492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13436","last_updated":"2025-03-17T17:58:30Z","snapshot_observed_at":"2026-08-07T16:57:02.267787Z","submitted_at":"2025-03-17T17:58:30Z","title":"Unified Autoregressive Visual Generation and Understanding with Continuous Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13436","snapshot_observed_at":"2026-08-06T04:37:13.531723Z","title":"Unified autoregressive visual generation and under- standing with continuous tokens","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.531723Z"},"links":{"cited_paper":"/paper/2503.13436","citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:2abe75be2405b8011c29232f75fd51f47dcb9e48006261351933f21b379b7052","observation_id":"4afd479f-c989-4a8c-92b3-2384f48e0880","resolution":{"observed_at":"2026-08-06T04:37:13.531723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.556095Z","title":"Geneval: An object-focused framework for evaluating text-to-image alignment, 2023","venue":null,"work_id":"26fe5309-3b08-4715-98f0-2fb68f499a05","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.577449Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:76ba9ad7e817fe5a44728dfe20d80f01e6fa27c0192b762cab1fb49a1eedad48","observation_id":"034209e4-77e0-4780-966e-562f021d4b49","resolution":{"observed_at":"2026-08-06T04:37:23.558265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:13.691792Z","title":"Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.691792Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:42aa3993701fb1b774c2d61f2bef942a69b2b3f4783c749b44f124666217af46","observation_id":"c89294ae-5a0d-427c-8dc3-15d696a68e9a","resolution":{"observed_at":"2026-08-06T04:37:13.691792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.546062Z","title":"Gemini 2.0 flash","venue":null,"work_id":"ea6c61e0-2c28-4acb-81a5-10e49a640b47","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.782304Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:fe5734f081331acdf53168a6ffdcda22c41cf180fa9554bab483b6c582070fe3","observation_id":"fcb91365-d167-459c-829c-ac59d18423a3","resolution":{"observed_at":"2026-08-06T04:37:23.548019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.538651Z","title":"Denoising diffusion probabilistic models, 2020","venue":null,"work_id":"6160b683-811c-4f23-b1e9-4629276ff74d","year":2020},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.866010Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:9459702852e9a802d33bc7e7f1cf0c321a4c6433113d1e0069273f88ddc60c79","observation_id":"8441cf70-ae9d-461a-8ba7-0e3509d902d3","resolution":{"observed_at":"2026-08-06T04:37:23.540726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.532003Z","title":"Ella: Equip diffusion models with llm for enhanced semantic alignment, 2024","venue":null,"work_id":"4421d5d0-1ea9-4de3-b4e3-c0663f7468fe","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.914935Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:7e67a194ed936966d7366af4f5e983521e72dadc82c319c7dcb4255462e7511b","observation_id":"0d03ecbc-d62a-45cf-9b01-ea4e6e777c5b","resolution":{"observed_at":"2026-08-06T04:37:23.534520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.524835Z","title":"Anyedit: Edit any knowledge encoded in language models,","venue":null,"work_id":"9984b947-cd74-40f7-a763-2d37b6301b31","year":null},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:13.965897Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:26518740d4b4423e5dd3887159cfaa746ec8a67d8b3f33b1bb540fd9c4c49951","observation_id":"cc5be816-e636-47ad-8b7f-f2476e8ec0bc","resolution":{"observed_at":"2026-08-06T04:37:23.527742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.516733Z","title":"Nova: Generative language models for assembly code with hierarchical attention and contrastive learning, 2025","venue":null,"work_id":"0f10ecc0-1639-4e50-ab1c-aed7a963b6e9","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.046573Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:3f490f9b238e63277e8db2a743b8145ace78ed30c66d22f6d44a37bd7a2ae747","observation_id":"e1f1312f-2ded-46c8-b985-b7e39dda56b7","resolution":{"observed_at":"2026-08-06T04:37:23.520298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.396273Z","title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation, 2023","venue":null,"work_id":"cf368cad-0f35-4bdf-870c-860317ac514c","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.122619Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:38007834be0edea6de3381cb5b8482a801f1628043be27aed9df4ed11da85490","observation_id":"be656d5b-b023-4f8b-b46d-72eacd9974c0","resolution":{"observed_at":"2026-08-06T04:37:23.455920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.256171Z","title":"Playground v2.5: Three insights towards enhancing aesthetic quality in text-to-image generation, 2024","venue":null,"work_id":"c57a8a35-d50f-4298-9372-aa8621aae68b","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.230767Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:f1a67d3946f2f79a64f422003924bcba89d84b8182046075ef2d763838861b9c","observation_id":"f05f9102-9ca3-4a44-912e-7aaafa3cbbfb","resolution":{"observed_at":"2026-08-06T04:37:23.332883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:23.114113Z","title":"Superedit: Rectifying and facilitating supervision for instruction-based image editing, 2025","venue":null,"work_id":"1301feac-e6ef-413c-a209-847c5934b4c1","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.294298Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:82168e524f937a228c9e6525946ec678da2bfe51bfedab6f2d9cee72d0cacb08","observation_id":"6412808b-62d0-46e5-8838-54e3d4ad07fb","resolution":{"observed_at":"2026-08-06T04:37:23.197741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:22.974098Z","title":"Autoregressive image generation without vector quantization, 2024","venue":null,"work_id":"d2f97100-0a1c-44c0-a9d2-a7deaa99c731","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.361543Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:cde90f01b13506bdab108f3324b6cda9876b7022454eb604d2e4a27c39bf395b","observation_id":"5a764196-ac51-47b3-94d7-fdd2b0af3f10","resolution":{"observed_at":"2026-08-06T04:37:23.018998Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:22.804153Z","title":"Hunyuan-dit: A powerful multi-resolution diffusion transformer with fine-grained chinese understanding, 2024","venue":null,"work_id":"084460df-0f1a-4b7c-a992-1d9b73e1f688","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.399826Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:e8300ee441c5ecc44d325982901f3b07c88158d328227be4308c3e2db2730a19","observation_id":"8fc833f6-46b8-4d58-93ae-e3a0449bb6ae","resolution":{"observed_at":"2026-08-06T04:37:22.872852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:22.652428Z","title":"Uniworld-v1: High-resolution semantic encoders for unified visual understanding and generation, 2025","venue":null,"work_id":"ca4ed4be-7ae0-440e-b084-5fb164edff5a","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.494416Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:90a446f1c556d5241e90e997d88e1f2477b79322b1b5909d32705785defed6be","observation_id":"c42d5694-412f-4f8c-93e1-00b62c6a382f","resolution":{"observed_at":"2026-08-06T04:37:22.741366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:22.540913Z","title":"Evaluating text-to-visual generation with image-to-text generation, 2024","venue":null,"work_id":"28ced559-383f-4a81-8058-26dc8e095b47","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.558002Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:8d66824ca555bb557120291ae20d0c5e51ef6c1fac14326d579dc2ba77c9a4a4","observation_id":"71114b46-6df5-4a59-b061-2072414335a9","resolution":{"observed_at":"2026-08-06T04:37:22.599063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:22.376908Z","title":"Step1x-edit: A practical framework for general image editing, 2025","venue":null,"work_id":"51b8199a-2611-4dbc-9ffd-5a9480fdb092","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.619391Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:2b946a628aa0dc32d70daa24c713b277f9b8deb6cf25fc60d0495f8e88bb2347","observation_id":"2f92b74a-e1fc-4c4f-970d-40624cc911a4","resolution":{"observed_at":"2026-08-06T04:37:22.453512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:22.247224Z","title":"Glide: Towards photorealistic image generation and editing with text-guided diffusion models, 2022","venue":null,"work_id":"c9994198-1653-41c9-91e6-9f69f3952bf1","year":2022},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.689669Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:0d2e57f95612fc4d64dedf15f5f53c35b401c8b6448cdbd1658b370549ff2728","observation_id":"6cf927c5-14e5-4aff-b0bd-4fc487a670ec","resolution":{"observed_at":"2026-08-06T04:37:22.285042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:22.057707Z","title":"Improving image generation with better captions","venue":null,"work_id":"e8196ad7-0382-44bc-aa47-87c936ccb47d","year":null},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.742393Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:ffb3b3bbd33f5361418a4c33c3dd876a2c4b55fd196f32ffe3cb18f180122692","observation_id":"b7a4d8c3-b6aa-4ea6-9bf3-7873dc08271b","resolution":{"observed_at":"2026-08-06T04:37:22.154806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:21.871301Z","title":"Gpt-4o system card, 2024","venue":null,"work_id":"10f37aec-1eda-4eda-af3c-8e8bd708415f","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.823714Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:e1a73f7485270a590c6a4def99050505be43c092e15cad6625336ac7809c507e","observation_id":"6798cdeb-609b-4171-9ccd-dd6020edeae4","resolution":{"observed_at":"2026-08-06T04:37:21.985813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:21.690157Z","title":null,"venue":null,"work_id":"5d1facd6-869c-4ec9-b185-d348edc3206a","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.887211Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:2f4299d0c60959a1c682c8d5c209c2ec1813e3b97b3d788097d198434499bbb7","observation_id":"ec07a3f1-7fe1-46b1-8ca8-97ae041b8e9e","resolution":{"observed_at":"2026-08-06T04:37:21.778466Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:21.519475Z","title":"Transfer between modalities with metaqueries, 2025","venue":null,"work_id":"73459eb7-1f7e-4eae-8ae3-65a6fe21bc50","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:14.963632Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:d2fc788011c9e2317e30ad0c3f55a3fdf7d5c44deef2bd67587dfc067fba1710","observation_id":"edb74c67-82d8-4521-af61-c5ae113e2cf6","resolution":{"observed_at":"2026-08-06T04:37:21.595188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:21.339507Z","title":"Sdxl: Improving latent diffusion models for high-resolution image synthesis, 2023","venue":null,"work_id":"9f1d0676-3171-4904-a707-370b1074d2a4","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.061823Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:e8bd9fe19237fc3894a0835eb76f107aa4d8ac4b332a92045a93292c306af9e6","observation_id":"6300a093-0beb-4b10-9488-3c69e3cdea02","resolution":{"observed_at":"2026-08-06T04:37:21.403942Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:21.236584Z","title":"Du, Zehuan Yuan, and Xinglong Wu","venue":null,"work_id":"a84544cd-7022-4d47-aad7-fbfe75677f44","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.093736Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:6cf1126ae48975331093020fac912c8f3338f6999c6327c4754b58e451261ce1","observation_id":"6ed35f91-a0a4-4982-a945-5d0d03df0ee3","resolution":{"observed_at":"2026-08-06T04:37:21.335571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:21.103575Z","title":"Qwen2.5 technical report, 2025","venue":null,"work_id":"cf833e2a-a947-4c6c-83ba-42b47f34b03e","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.223351Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:54b08469ea9ebd8bc28877db45f312679b817e147930d79ec9381bd077f9ce01","observation_id":"5ab9de3e-4638-4f54-98d0-067fd693ce5d","resolution":{"observed_at":"2026-08-06T04:37:21.185100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:20.966896Z","title":"Learning transferable visual models from natural language supervision, 2021","venue":null,"work_id":"b33143cd-af16-41e6-a21d-61c56126cf0a","year":2021},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.287812Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:3e47ee4d21a016363175d56e111be791b491155fb7fe6821d2b8e4a611fd8e75","observation_id":"8f0adabe-9c79-474d-9ee8-8819afd98362","resolution":{"observed_at":"2026-08-06T04:37:21.042863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:20.839682Z","title":"High-resolution image synthesis with latent diffusion models, 2022","venue":null,"work_id":"8bda3291-dc42-43ce-b21c-03e0eb2a00ef","year":2022},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.379501Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:ecf0da3752858d07691676adc7345e5b3b24a60342576981a837892e901f66bf","observation_id":"b3f68411-61a3-4584-932c-c4a1c2fa6e27","resolution":{"observed_at":"2026-08-06T04:37:20.913859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:20.697872Z","title":null,"venue":null,"work_id":"cf8c1caa-2f1f-4672-81da-2e04eec1ed34","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.430371Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:f917da4ff8df5757539ad0cbf131cff87215c09794ba0c6fc78dcd4003a68d63","observation_id":"544989da-0740-4ce6-9c7c-a4d75a25d427","resolution":{"observed_at":"2026-08-06T04:37:20.761822Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:20.520635Z","title":"Kingma, Abhishek Kumar, Stefano Ermon, and Ben Poole","venue":null,"work_id":"8af8011f-6471-44d0-baf1-ae3ce0599265","year":2021},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.510627Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:47b5a840d28ebe426615d063728ec934faa1a463fcbbaa3c73fdac3f461193ea","observation_id":"8a7cee7d-b595-4cf8-a7b3-2ca6af60c488","resolution":{"observed_at":"2026-08-06T04:37:20.608021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06525","last_updated":"2024-06-10T17:59:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-10T17:59:52Z","title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06525","snapshot_observed_at":"2026-08-06T04:37:15.549976Z","title":"Autoregressive model beats diffusion: Llama for scalable image generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.549976Z"},"links":{"cited_paper":"/paper/2406.06525","citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:01c5589ac66c0eb259f41c8214689f8ef84f2fb6af1f9e965409ee4875c7c3c9","observation_id":"46bba553-8468-4013-8d4f-58220b6eb0f6","resolution":{"observed_at":"2026-08-06T04:37:15.549976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:20.371709Z","title":"Generative multimodal models are in-context learners, 2024","venue":null,"work_id":"ff958813-0d24-44e8-91c5-a2510ca9bced","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.617341Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:d75a921ef90b7975689542ac18f2fbc3befeb451959852da8c702b8de4ca84af","observation_id":"dd52bc6a-46ac-4f73-b860-52362e50333c","resolution":{"observed_at":"2026-08-06T04:37:20.433212Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:15.678202Z","title":"Siglip 2: Multilingual vision- language encoders with improved semantic understanding, localization, and dense features,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.678202Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:b522c23c03bb0c719df133750d73670f37c608671f577e886fec775069e2bc29","observation_id":"789e1a77-2dac-4524-bab1-1db716e531e3","resolution":{"observed_at":"2026-08-06T04:37:15.678202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:20.229813Z","title":"Neural discrete representation learning, 2018","venue":null,"work_id":"e8bed4bd-4ef9-4bab-b22a-76b7d40a340c","year":2018},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.740703Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:5bae03a27b963815c2538f610db15b12e8340caafa013fc52a6da91f7a5ecd67","observation_id":"ec280dbb-ef4f-438d-829c-c010c6866057","resolution":{"observed_at":"2026-08-06T04:37:20.277744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:20.060005Z","title":"Illume: Illuminating your llms to see, draw, and self-enhance, 2024","venue":null,"work_id":"c723192f-ba75-4692-a790-650cf0c08b61","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.823422Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:fe1013649629381af29b8560c5bf0ebab6a134684833f50faf9d805bcf0fc5e9","observation_id":"5a988791-c7c1-4de8-b368-196e4746e428","resolution":{"observed_at":"2026-08-06T04:37:20.123409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:19.913528Z","title":"Ovis-u1 technical report, 2025","venue":null,"work_id":"5aa29219-cb43-4940-bea8-b24ccb99ee0e","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.880569Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:46124b9aaf540992e0c8d305447a898832bcfa738684a3ee1e5f8cc4d28d5191","observation_id":"96b58ce0-7310-431e-a985-3953a84e4be9","resolution":{"observed_at":"2026-08-06T04:37:19.956681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:19.782556Z","title":"Emu3: Next-token prediction is all you need, 2024","venue":null,"work_id":"7becd92a-1088-43e1-9307-0b668ee1aeb0","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:15.937622Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:c19754dc5fb4d3afd8c6d0e854fcbfe0ee290d78aab6506015f0b4a6769d6d5f","observation_id":"42ca0117-57b6-4151-ba24-06583934515a","resolution":{"observed_at":"2026-08-06T04:37:19.852119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:19.629631Z","title":"Janus: Decoupling visual encoding for unified multimodal understanding and generation, 2024","venue":null,"work_id":"76f7ac3a-ca21-4f5b-babd-07680db73b2f","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.026041Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:b328b6d55597e138a05807c0cee3afd1f3147d4d6dd52d798ca32a733e021b67","observation_id":"154322e2-a8a7-4799-9933-2f203cf05397","resolution":{"observed_at":"2026-08-06T04:37:19.706383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:19.529338Z","title":"Omnigen2: Exploration to advanced multimodal generation, 2025","venue":null,"work_id":"285c9be9-07d5-4acf-81da-9321d54e6897","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.086323Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:6929891d188810186e77cade3b0c33797453752e5578d8bf680eeee649b0d29f","observation_id":"827d3003-718c-493e-ba5b-fdc998e0dbd0","resolution":{"observed_at":"2026-08-06T04:37:19.589792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23661","last_updated":"2025-06-02T13:04:26Z","snapshot_observed_at":"2026-08-07T12:38:06.235637Z","submitted_at":"2025-05-29T17:09:44Z","title":"OpenUni: A Simple Baseline for Unified Multimodal Understanding and Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23661","snapshot_observed_at":"2026-08-06T04:37:16.161813Z","title":"Openuni: A simple baseline for unified multimodal understanding and generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.161813Z"},"links":{"cited_paper":"/paper/2505.23661","citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:0694e73e58733d799c5b589bb23e491c0536376a6436d143c0fd324658c87228","observation_id":"2440898c-8e79-4572-8e74-09e812db5517","resolution":{"observed_at":"2026-08-06T04:37:16.161813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:19.394550Z","title":"Harmonizing visual representations for unified multimodal understanding and generation, 2025","venue":null,"work_id":"690d9707-07ed-49c1-a18a-c3f8ffaf50cc","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.231109Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:b11c0c1a8bef60bafc64b80be01284eb4d1e46cc6037dd5f59fb79837212e352","observation_id":"1c34bd2b-8879-42ce-b559-bf2add1b4836","resolution":{"observed_at":"2026-08-06T04:37:19.448029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:19.247160Z","title":"Human preference score v2: A solid benchmark for evaluating human preferences of text-to-image synthesis, 2023","venue":null,"work_id":"2c1d9a44-d86e-475f-bdd3-35b38b03fc82","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.306194Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:c5918c7abc7007f70201042fd0039641efbb1813e0d958c72f33fd40824cdfaf","observation_id":"0b5e6aa0-90b0-4296-94ec-64b6aab783c0","resolution":{"observed_at":"2026-08-06T04:37:19.310639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:19.096858Z","title":"Vila-u: a unified foundation model integrating visual understanding and generation, 2025","venue":null,"work_id":"932357f4-b802-4fe7-9b02-5804a1e29d11","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.391233Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:25b01fe77435f7c43d5fdad2f1546d49470d7ddd047cbbb3ee827217f8981c07","observation_id":"d6301a93-b16f-4ba3-b8e1-10bd3215eaec","resolution":{"observed_at":"2026-08-06T04:37:19.175901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.991287Z","title":"Omnigen: Unified image generation, 2024","venue":null,"work_id":"0be525bc-7ed6-492a-9b99-2ad1ae7dfdd5","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.494447Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:1cd22ac389a8781778bbc47cd6086ec164b420af8c8284e1b950126cff3a8a30","observation_id":"046282a6-2ca7-4d73-a323-b4099931dc71","resolution":{"observed_at":"2026-08-06T04:37:19.044625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.894743Z","title":"Show-o: One single transformer to unify multimodal understanding and generation, 2024","venue":null,"work_id":"370f24b7-a05f-4bbf-a8f7-597de4ff0e98","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.594819Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:7ddf1f271069332c10aab32221a571fd7b6b9dc3657e35c5285fdca88ea7d789","observation_id":"5ad7af79-4f04-4ea3-b2d5-d6eb34a7abc7","resolution":{"observed_at":"2026-08-06T04:37:18.935800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:16.701927Z","title":"Imagereward: Learning and evaluating human preferences for text-to-image generation,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.701927Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:a6878ed277ab9bdafa30989d48482370fecc7d2623179f4efd4dc01740942b7e","observation_id":"124a73ff-99fa-40c1-adf5-8f06137714b3","resolution":{"observed_at":"2026-08-06T04:37:16.701927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.760367Z","title":"1.58-bit flux, 2024","venue":null,"work_id":"71ce1a4b-1a8e-4e6b-9872-1ace391a056b","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.764209Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:da123a8b1b570641e7db63337c47e121e9af6696377eb28a4abdbc02ffaffefd","observation_id":"ae7e7112-a6a6-4a43-b523-2e24afb06761","resolution":{"observed_at":"2026-08-06T04:37:18.817586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.640427Z","title":"Imgedit: A unified image editing dataset and benchmark, 2025","venue":null,"work_id":"18f59da3-3994-4d0f-b5ed-d69e4b862d0e","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.843001Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:9735e913a28bfda2fdee6e64e5921faa89908e92dcdfd6a7aafb501a82ec1018","observation_id":"67739b13-640a-4c06-90eb-51c7678e4c97","resolution":{"observed_at":"2026-08-06T04:37:18.702970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.517267Z","title":"Sigmoid loss for language image pre-training, 2023","venue":null,"work_id":"57e60dd8-41f0-423d-9d7d-75ae8cb2479e","year":2023},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:16.970273Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:0a145847056341c4f6e8228d538c6a96d6c2887da0e1a3c85d838e741c1c205d","observation_id":"7e166349-45cb-48a7-8e2d-5a050f248710","resolution":{"observed_at":"2026-08-06T04:37:18.566607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.363000Z","title":"Magicbrush: A manually annotated dataset for instruction-guided image editing, 2024","venue":null,"work_id":"6eb11921-e6c1-4037-a308-5e840ef6e0f8","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:17.099744Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:024bba364e709055ecd24e72337ff2218a1bb460b93885c5f505f9cb2e079343","observation_id":"017486b4-66ff-4c6b-b9d7-5b2344b4d228","resolution":{"observed_at":"2026-08-06T04:37:18.437460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.252087Z","title":"In-context edit: Enabling instructional image editing with in-context generation in large scale diffusion transformer, 2025","venue":null,"work_id":"ecd0f4d7-d467-44e9-823c-154ff8b2ed70","year":2025},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:17.236797Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:14e821842bcd462890345081c6891784d9019aee1d0532a320845c8260071f7e","observation_id":"bf17f735-081b-471e-8c8b-e0838cd5f229","resolution":{"observed_at":"2026-08-06T04:37:18.311784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:18.052247Z","title":"Ultraedit: Instruction-based fine-grained image editing at scale, 2024","venue":null,"work_id":"8a4c8454-d116-4d92-a520-5b38eab69a68","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:17.364639Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:9f61751f3eb7090ad2c9db5a0e4ea13caa4ad24a28e9d33cc44a4e3a8cafd792","observation_id":"84d2ef48-a026-4bc6-97b2-18a099f1445e","resolution":{"observed_at":"2026-08-06T04:37:18.158911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:17.853393Z","title":"Transfusion: Predict the next token and diffuse images with one multi-modal model, 2024","venue":null,"work_id":"cbb6db7c-8115-471a-ae15-69ae36373e9b","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:17.424329Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:d30aeb0766566649aa613df7f063982c87063764ce07b03dddd207c08d92bc38","observation_id":"5e7e1418-2ba1-4ac8-af1b-f2e8234548e1","resolution":{"observed_at":"2026-08-06T04:37:17.955095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T04:37:17.636423Z","title":"Lumina-next: Making lumina-t2x stronger and faster with next-dit, 2024","venue":null,"work_id":"9ba64c9b-6880-426b-96cb-dc44128c6e6d","year":2024},"citing_paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T04:37:17.479761Z"},"links":{"citing_paper":"/paper/2508.03320"},"observation_digest":"sha256:9fa49243b3af586af844913c4efc9028ba738b530e383d1d5ca2391f5aeadeb9","observation_id":"ee94624c-488c-4bd6-b19a-17539b931a1c","resolution":{"observed_at":"2026-08-06T04:37:17.750439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.03320","last_updated":"2025-08-05T10:59:01Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T04:37:03.997933Z","submitted_at":"2025-08-05T10:59:01Z","title":"Skywork UniPic: Unified Autoregressive Modeling for Visual Understanding and Generation"},"reference_resolution":{"displayed":62,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":53},"total_outbound_references":62},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 62 of 62 outbound references and 5 inbound Pith citation observations for arXiv:2508.03320."}