{"as_of":"2026-08-18T23:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6b52bfce7796f1d2fc34b16bbf78b8a60fda2bad40a2319e8e6e87849af68e3d","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:40:26.661269Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2504.12018/citation-record","integrity":"/paper/2504.12018/integrity","json":"/paper/2504.12018/citation-record.json","paper":"/paper/2504.12018"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-16T12:40:26.393404Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.393404Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:646030e94d57cf9f1996d324bcbce9beefa9123a9ba7e9dc64a165f92a97b8aa","observation_id":"a039c636-3c55-4eaa-80e4-ffcb035b111d","resolution":{"observed_at":"2026-08-16T12:40:26.393404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03511","last_updated":"2024-06-28T11:23:11Z","snapshot_observed_at":"2026-08-16T14:37:09.655792Z","submitted_at":"2023-12-06T14:13:38Z","title":"Kandinsky 3.0 Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03511","snapshot_observed_at":"2026-08-16T12:40:26.400427Z","title":"Kandinsky 3.0 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.400427Z"},"links":{"cited_paper":"/paper/2312.03511","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:75d4b30edab203279a31873da4384757e5cd5c3a118715c9b5316ff67c739dd7","observation_id":"e60fa8f0-7b8c-483a-8b78-62f47d8d10ec","resolution":{"observed_at":"2026-08-16T12:40:26.400427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-16T12:40:26.407193Z","title":"Qwen-vl: A versatile vision-language model for un- derstanding, localization, text reading, and beyond","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.407193Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:082780c80215849e27ebcf613435599bfe5ef4ad9482f24764c49067e6fb20db","observation_id":"1da81de3-5e3b-410c-8e34-777f46280c87","resolution":{"observed_at":"2026-08-16T12:40:26.407193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-16T12:40:26.413539Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.413539Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:20e4f9ba252984e096ca8141fed2968ad7a4da56c12d086adf7cd7082caa8f14","observation_id":"b6942d79-9386-4c24-b0e9-c94d1b18d996","resolution":{"observed_at":"2026-08-16T12:40:26.413539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.468851Z","title":null,"venue":null,"work_id":"8de72128-ffac-4b84-9cb8-abcc99b17225","year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.419339Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:80dc9773d24522d9161a6bd5f0191302d18ac153920ae7cb00e3e465555f9712","observation_id":"e51052a3-f2ff-4b79-8db3-b23f4fd2cf92","resolution":{"observed_at":"2026-08-16T12:40:27.475336Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-16T12:40:26.424899Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test- time scaling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.424899Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:8dc6ce12fbf072002e334fc02a7af09e84778146bf79dee5029598757475ce0b","observation_id":"14ca790b-6790-4ff4-bdd8-0f5c54d2038e","resolution":{"observed_at":"2026-08-16T12:40:26.424899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.431686Z","title":"Internvl: Scaling up vision foundation mod- els and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.431686Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:f29186119d229f99efd704a51b7508fe687a1b1ce479535d41d0715890a5767d","observation_id":"6d4418fb-e54f-4c40-8b10-cfdbdf92f72d","resolution":{"observed_at":"2026-08-16T12:40:26.431686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-16T12:40:26.437504Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.437504Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:3fe2d356d66237388d2d67e9207bbdba58cbec69bc8f3c7798f3b9585ed3acf1","observation_id":"82dc28c0-d4dc-4fbd-800d-7edc8d30ce76","resolution":{"observed_at":"2026-08-16T12:40:26.437504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.434433Z","title":"Dreamina","venue":null,"work_id":"977f9492-77ee-495e-b29d-3435f4f0f4e7","year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.444201Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:fee3c6f7c2766d630d4da96abffe0b0d7be31ea7c39c17e0ce97ffcbcd1430cf","observation_id":"44858ea2-b495-49c0-b23d-38c854cc6d35","resolution":{"observed_at":"2026-08-16T12:40:27.441607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.450097Z","title":"Scaling recti- fied flow transformers for high-resolution image synthesis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.450097Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:3c6d0b1648d77838c4cfade6828f1077b6788e0d2d8fad695ed00887d4e6d439","observation_id":"c5fdb2b8-f012-482b-af9b-96683ee950a4","resolution":{"observed_at":"2026-08-16T12:40:26.450097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17002","last_updated":"2024-04-09T07:46:43Z","snapshot_observed_at":"2026-08-16T14:39:32.709797Z","submitted_at":"2023-11-28T17:57:44Z","title":"Ranni: Taming Text-to-Image Diffusion for Accurate Instruction Following","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17002","snapshot_observed_at":"2026-08-16T12:40:26.455283Z","title":"Ranni: Taming text-to-image diffu- sion for accurate instruction following","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.455283Z"},"links":{"cited_paper":"/paper/2311.17002","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:ce09b749b4f7ee54fd08b16682a5fed658e10de8d1e4d9fae33ff78010139227","observation_id":"a77dea49-59fe-459e-b1c2-2fc6a05cd78c","resolution":{"observed_at":"2026-08-16T12:40:26.455283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18150","last_updated":"2024-12-25T15:00:36Z","snapshot_observed_at":"2026-08-18T03:26:48.098712Z","submitted_at":"2024-12-24T04:08:25Z","title":"EvalMuse-40K: A Reliable and Fine-Grained Benchmark with Comprehensive Human Annotations for Text-to-Image Generation Model Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.18150","snapshot_observed_at":"2026-08-16T12:40:26.460713Z","title":"Evalmuse-40k: A reliable and fine-grained benchmark with comprehensive human annotations for text- to-image generation model evaluation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.460713Z"},"links":{"cited_paper":"/paper/2412.18150","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:a597284074d94eb6b96c816b1f1d9caa39515b5899c30fd12bfae6c0c92d8056","observation_id":"73ddecb0-e4a0-4d62-b9b0-dec03dc27092","resolution":{"observed_at":"2026-08-16T12:40:26.460713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.466405Z","title":"NTIRE 2025 challenge on text to image generation model quality assess- ment","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.466405Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:968b177a206369b7cbd3de5336515a082fc5d6c2a0030a8807c327c704217b28","observation_id":"5b129cfa-67e8-4506-b066-9bffb096fd3a","resolution":{"observed_at":"2026-08-16T12:40:26.466405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07614","last_updated":"2024-07-11T11:05:53Z","snapshot_observed_at":"2026-08-16T13:35:27.185190Z","submitted_at":"2024-07-10T12:52:49Z","title":"MARS: Mixture of Auto-Regressive Models for Fine-grained Text-to-image Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07614","snapshot_observed_at":"2026-08-16T12:40:26.471660Z","title":"Mars: Mixture of auto-regressive mod- els for fine-grained text-to-image synthesis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.471660Z"},"links":{"cited_paper":"/paper/2407.07614","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:96d0656b3140b9f192f7b48463a485cef839e8e328963fb03584014278cb66ca","observation_id":"0efaf066-a7b3-4984-84e1-8003036adc20","resolution":{"observed_at":"2026-08-16T12:40:26.471660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08718","last_updated":"2022-03-23T19:47:21Z","snapshot_observed_at":"2026-07-06T11:01:02.207193Z","submitted_at":"2021-04-18T05:00:29Z","title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08718","snapshot_observed_at":"2026-08-16T12:40:26.477919Z","title":"Clipscore: A reference-free evaluation met- ric for image captioning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.477919Z"},"links":{"cited_paper":"/paper/2104.08718","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:565ce17c398907d0180eb38826ace4487817ccd3fd121c34dbaab9f4f9f3dda5","observation_id":"0b509637-dc17-4219-ba9b-4f04119afc3d","resolution":{"observed_at":"2026-08-16T12:40:26.477919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.389930Z","title":"Midjourney","venue":null,"work_id":"711be0cc-522d-4f9e-b9a4-ec5b15a0f9e6","year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.483496Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:91ef856ab3a2ea08dd4c2bd7190dafe8bb68e21d9eab3e22cb9ecebbeeafe0e8","observation_id":"e6203e11-aa58-4904-a891-662c76648e7d","resolution":{"observed_at":"2026-08-16T12:40:27.396441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.370625Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":"19dc1426-0704-4d52-b221-5d2514e18f89","year":2022},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.489037Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:5a517b42de762af4073547b2650585a99d5b8ffb665bb79b3242dd8fa152e342","observation_id":"20e112cf-ee31-475f-9882-5693aa838f24","resolution":{"observed_at":"2026-08-16T12:40:27.377160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.352266Z","title":"Tifa: Accurate and interpretable text-to-image faithfulness evaluation with question answering","venue":null,"work_id":"f6344dc4-ba4e-4d9f-accb-d056dc97b4dc","year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.495712Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:c18e21bf5545c46e7ec86f42888f031cbfe80c0bb7bf97b29d898c6d611ab116","observation_id":"7aa9f23b-e078-4099-8f62-4668b2dcc5c2","resolution":{"observed_at":"2026-08-16T12:40:27.358056Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.331928Z","title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation","venue":null,"work_id":"0b4da826-c89d-4acb-858c-0e6e4718041a","year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.500934Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:15235ad3e322c31119bbe3c846c4e5b68db7c86a6096d0b842c223b05b32039c","observation_id":"228484db-f87f-4904-932b-b0163dae7e69","resolution":{"observed_at":"2026-08-16T12:40:27.338434Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.311342Z","title":"Evaluating and improving composi- tional text-to-visual generation","venue":null,"work_id":"e01a30ba-b57e-4f41-8fe9-c8d9f33fe2f4","year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.505987Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:511626c80f1cf3a3044f07caff2ea6d37023b6aef3582af03518d2fc6d4a6f43","observation_id":"41b50943-5093-4153-86d8-35780eaca9f1","resolution":{"observed_at":"2026-08-16T12:40:27.318846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17245","last_updated":"2024-02-27T06:31:52Z","snapshot_observed_at":"2026-08-16T13:11:20.195602Z","submitted_at":"2024-02-27T06:31:52Z","title":"Playground v2.5: Three Insights towards Enhancing Aesthetic Quality in Text-to-Image Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.17245","snapshot_observed_at":"2026-08-16T12:40:26.511573Z","title":"Playground v2","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.511573Z"},"links":{"cited_paper":"/paper/2402.17245","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:bf977aec22a209235d73f337dbe376173fca2ece3e1492412f82ca7c580b118d","observation_id":"94a28a47-d2fd-4763-b475-f139e210616e","resolution":{"observed_at":"2026-08-16T12:40:26.511573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.516854Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.516854Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:a9dac0c24a99d0a95b40825918d682c00753ca40c876dd36114febae67a6d9c2","observation_id":"e3044d9d-2f82-4d73-9e24-07b074d0c87c","resolution":{"observed_at":"2026-08-16T12:40:26.516854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08748","last_updated":"2024-05-14T16:33:25Z","snapshot_observed_at":"2026-07-06T18:14:16.386835Z","submitted_at":"2024-05-14T16:33:25Z","title":"Hunyuan-DiT: A Powerful Multi-Resolution Diffusion Transformer with Fine-Grained Chinese Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08748","snapshot_observed_at":"2026-08-16T12:40:26.523312Z","title":"Hunyuan-dit: A powerful multi-resolution diffusion transformer with fine-grained chi- nese understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.523312Z"},"links":{"cited_paper":"/paper/2405.08748","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:d4473625f1b5766942ff0d004a6989a091e9bdb5318c74cdb6860bb818d27b27","observation_id":"7ac6b9e6-7cf5-4f49-9e6d-cb4c928b463f","resolution":{"observed_at":"2026-08-16T12:40:26.523312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.528469Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.528469Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:19b453d72984fe3bedcbabda5185669c274ea16bba6ce702b773462db24052f6","observation_id":"700c6fc0-ddbf-46e2-ac51-a1c60f1bdd0a","resolution":{"observed_at":"2026-08-16T12:40:26.528469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.534548Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.534548Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:0b055ddb7a303c5cf7abc49cfda4d8081769cee134cbc75d9535993cc3493997","observation_id":"f90556e0-aeb3-4d94-91a9-d8b7d5583717","resolution":{"observed_at":"2026-08-16T12:40:26.534548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.540586Z","title":"Llavanext: Improved reasoning, ocr, and world knowledge, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.540586Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:41bbe61e6584431a8fc0287b486ccbe6f19c9ef577416930d51b8f29ec22b34c","observation_id":"76aa0313-43c2-454b-aa84-1bd9d64fe4b7","resolution":{"observed_at":"2026-08-16T12:40:26.540586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05525","last_updated":"2024-03-11T16:47:41Z","snapshot_observed_at":"2026-08-11T01:38:59.827005Z","submitted_at":"2024-03-08T18:46:00Z","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05525","snapshot_observed_at":"2026-08-16T12:40:26.545896Z","title":"Deepseek-vl: towards real-world vision- language understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.545896Z"},"links":{"cited_paper":"/paper/2403.05525","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:72c3404c2b5bc5ad90c18196c99155b7b85053be3240c63c894e051416347efc","observation_id":"6487a08c-b4e9-4dc1-bfc0-013f5d676c53","resolution":{"observed_at":"2026-08-16T12:40:26.545896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20797","last_updated":"2024-06-17T17:51:50Z","snapshot_observed_at":"2026-08-16T13:47:26.506119Z","submitted_at":"2024-05-31T13:59:18Z","title":"Ovis: Structural Embedding Alignment for Multimodal Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20797","snapshot_observed_at":"2026-08-16T12:40:26.552528Z","title":"Ovis: Structural embed- ding alignment for multimodal large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.552528Z"},"links":{"cited_paper":"/paper/2405.20797","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:b61d85fb08e576f71f4cf2a1ca167fab68e8c44c6e923055c0bfe2776334c834","observation_id":"4cef6bf6-12ac-4b35-b56b-21fb16ccee8a","resolution":{"observed_at":"2026-08-16T12:40:26.552528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.559152Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.559152Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:77bda0708fef910bbc85019f8a13fc356a2777f5818e58fc48ab4d110888eb27","observation_id":"191beef7-3212-43e2-93c6-bb505d01d53f","resolution":{"observed_at":"2026-08-16T12:40:26.559152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.06125","last_updated":"2022-04-13T01:10:33Z","snapshot_observed_at":"2026-08-15T12:50:58.405488Z","submitted_at":"2022-04-13T01:10:33Z","title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.06125","snapshot_observed_at":"2026-08-16T12:40:26.565081Z","title":"Hierarchical text-conditional image gen- eration with clip latents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.565081Z"},"links":{"cited_paper":"/paper/2204.06125","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:9d1e3d5db1961a22343dce414881124a32d3c64f601a985eb9da8b3f0858dece","observation_id":"0cb6d91b-0910-43c7-9e2a-97883407daf7","resolution":{"observed_at":"2026-08-16T12:40:26.565081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.572519Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.572519Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:a2f1c581edcfaac6f8030e75dd77d71256a2c871f5f265ccefaaff47c3d47a29","observation_id":"24505366-057f-4bdc-974c-7527ec0aa346","resolution":{"observed_at":"2026-08-16T12:40:26.572519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.221108Z","title":"Adversarial diffusion distillation","venue":null,"work_id":"1f8792a3-fc91-4e89-a475-ea71667920d9","year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.577850Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:5b7d7779332389e21b21d5c25a381e0e0cb7d557e0242ab2ac0f58341d930dc7","observation_id":"677aa1d1-1f91-4a58-b64d-8021cd5b4c79","resolution":{"observed_at":"2026-08-16T12:40:27.226754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.201701Z","title":"Hugginggpt: Solving ai tasks with chatgpt and its friends in huggingface","venue":null,"work_id":"7f207cb9-869c-4ce9-95ab-76e54e89e3d7","year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.583786Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:d525e456b278c0910fd7908036c77ca64be25cbaeda79bffe7fab41fa3bc384a","observation_id":"cc23c617-bf0b-4ca4-9609-ba611575f1ed","resolution":{"observed_at":"2026-08-16T12:40:27.208423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-16T12:40:26.589564Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.589564Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:972559400e7fd9fa89088f4a5287aae2795bf4ab7fefdfce60ad6b6023519174","observation_id":"83dae9d7-bcc8-4359-a459-81d9aa2ac634","resolution":{"observed_at":"2026-08-16T12:40:26.589564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.595625Z","title":"Kolors: Effective training of diffusion model for photorealistic text-to-image synthesis","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.595625Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:e74ce76c71bc56e363b13bb166fdd8fd7c7bf4f12221ddd30f2c8455c4dda467","observation_id":"d12f04cc-a6a6-4592-ad0a-8fa2761096de","resolution":{"observed_at":"2026-08-16T12:40:26.595625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.169977Z","title":"Chain-of-thought prompting elicits reasoning in large lan- guage models","venue":null,"work_id":"5cd1fbb1-f107-41a5-a589-03b5b29a7c72","year":2022},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.602565Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:663377100dacc839bc3828b7ea425d3abd109fac955206166380de9e05904bb9","observation_id":"d3eac9ab-a837-4762-bb2a-efdca2f4f1c4","resolution":{"observed_at":"2026-08-16T12:40:27.177195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16820","last_updated":"2025-03-17T15:53:14Z","snapshot_observed_at":"2026-08-18T10:40:40.051476Z","submitted_at":"2024-04-25T17:58:43Z","title":"Revisiting Text-to-Image Evaluation with Gecko: On Metrics, Prompts, and Human Ratings","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16820","snapshot_observed_at":"2026-08-16T12:40:26.609996Z","title":"Revisiting text-to-image evaluation with gecko: On met- rics, prompts, and human ratings","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.609996Z"},"links":{"cited_paper":"/paper/2404.16820","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:ff45f20f976c3b0b0daaacba7e8f47e02929887e048d86db7dbb14c21b76a3f0","observation_id":"aa4a7cf8-23cb-4208-9fa9-1ba2a7e0dada","resolution":{"observed_at":"2026-08-16T12:40:26.609996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.04671","last_updated":"2023-03-08T15:50:02Z","snapshot_observed_at":"2026-08-15T10:03:23.497428Z","submitted_at":"2023-03-08T15:50:02Z","title":"Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.04671","snapshot_observed_at":"2026-08-16T12:40:26.617119Z","title":"Visual chatgpt: Talking, drawing and editing with visual foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.617119Z"},"links":{"cited_paper":"/paper/2303.04671","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:b26e9f712280b625f28ed356fb1ed97bc4974d94ebc643f40f3e46e2049c2f99","observation_id":"4557b27a-5f78-45e7-9426-e61b90c5d129","resolution":{"observed_at":"2026-08-16T12:40:26.617119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.151390Z","title":"Q-align: Teaching lmms for visual scoring via discrete text-defined levels","venue":null,"work_id":"f8080dd5-0c25-4bfd-a0df-6f16dd0e8d33","year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.624217Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:3c2798b4a5c79629ad26acb2ddc4434fcd7238c8102e325c7c71c89292a52045","observation_id":"40a63b09-6cd1-4e1e-b70f-df790981cee3","resolution":{"observed_at":"2026-08-16T12:40:27.157242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09341","last_updated":"2023-09-25T08:19:23Z","snapshot_observed_at":"2026-08-14T02:46:25.655503Z","submitted_at":"2023-06-15T17:59:31Z","title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.09341","snapshot_observed_at":"2026-08-16T12:40:26.629918Z","title":"Human preference score v2: A solid benchmark for evaluating human preferences of text-to-image synthesis","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.629918Z"},"links":{"cited_paper":"/paper/2306.09341","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:8a61269f1e667df7ee8aaec39e233901d777426d49a59fee7a3f544d615e059e","observation_id":"eac17490-2252-4540-9922-a54025055502","resolution":{"observed_at":"2026-08-16T12:40:26.629918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.635879Z","title":"Imagere- ward: Learning and evaluating human preferences for text- to-image generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.635879Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:d1bffc6d3627193440f7a4877aad0974a8e0cf330e0254b3591fd20dd3f94083","observation_id":"309f6589-57dd-473e-895c-105382268df6","resolution":{"observed_at":"2026-08-16T12:40:26.635879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.116832Z","title":"What you see is what you read? improving text- image alignment evaluation","venue":null,"work_id":"3efee5b0-8fe9-4825-a9ca-c2316c67f1ce","year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.641902Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:a40d3d7bfa7360ab66804fc71b7242e8c2a6a3edbb29ffb54ad2d0e00c440d0d","observation_id":"6eada856-94fd-4d4a-b983-3d8988269077","resolution":{"observed_at":"2026-08-16T12:40:27.123600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:27.094788Z","title":"mplug- owl3: Towards long image-sequence understanding in multi- modal large language models","venue":null,"work_id":"44aaeaef-a1e8-48cc-8ef3-31fddc56a518","year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.648908Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:6b13b23eb026a56b260b6bd3a84fe804022ea46aa50b93234df905dc3496db3b","observation_id":"6d3eace7-8ac9-421e-8c84-af0da819edef","resolution":{"observed_at":"2026-08-16T12:40:27.102486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T12:40:26.655485Z","title":"Swift:a scal- able lightweight infrastructure for fine-tuning, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.655485Z"},"links":{"citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:7647813b009fee1dad6ba762a42d49eb73f9c669ffefc6db9d27c2904545df51","observation_id":"de44943b-669d-4577-a7f3-33a215fc197d","resolution":{"observed_at":"2026-08-16T12:40:26.655485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-16T12:40:26.661269Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T12:40:26.661269Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2504.12018"},"observation_digest":"sha256:fb0b835d21ef336df4a0f6eff26aebb28485578e8b8610daf7fe5cce9d73c5b1","observation_id":"83488eb1-ff86-4677-a0fe-16a21ea732ff","resolution":{"observed_at":"2026-08-16T12:40:26.661269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2504.12018","last_updated":"2025-04-16T12:21:49Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-18T10:41:09.205775Z","submitted_at":"2025-04-16T12:21:49Z","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":33,"verified_exact":0,"verified_fuzzy":12},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 0 inbound Pith citation observations for arXiv:2504.12018."}