{"as_of":"2026-08-19T07:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:092faf8e93de7802d99be641510389a1de4491d0470d95ffc5c37f663f8d0e81","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T22:47:49.169054Z","state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.06356/citation-record","integrity":"/paper/2505.06356/integrity","json":"/paper/2505.06356/citation-record.json","paper":"/paper/2505.06356"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.575307Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","venue":null,"work_id":"a20ffa23-f780-4a0c-ba44-f400108fb9ff","year":2022},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.049971Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:6fe88e0d26136b85b616ab03e56e1c26809556eae1e9e07f24e880b098614627","observation_id":"663b3fed-252e-4bcd-9f2b-3b31b2a02610","resolution":{"observed_at":"2026-08-15T22:47:49.579507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.561913Z","title":"Y our vision-language model itself is a strong ﬁlter: Toward s high-quality instruction tuning with data selection, 2024","venue":null,"work_id":"34654a61-6dc9-4dcf-a0d8-404e20f8c8fb","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.054656Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:e9e2cfef85c327c3b3ecd705bd3c1f96295799247289ade4b950d198f8ce21ad","observation_id":"cbaf101d-231b-4de9-9e3a-656974b9fcbb","resolution":{"observed_at":"2026-08-15T22:47:49.566146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.549392Z","title":"Comm: A coherent inter- leaved image-text dataset for multimodal understanding an d generation, 2024","venue":null,"work_id":"22cff962-94ff-4616-b01b-22635faf2360","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.058646Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:6238679940ea86a570a0e6372495adda117d836effceb9b715e139e429c31e2c","observation_id":"c8739de5-0843-4a1d-9212-26007803a608","resolution":{"observed_at":"2026-08-15T22:47:49.553590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.06794","last_updated":"2023-06-05T17:55:12Z","snapshot_observed_at":"2026-08-19T06:18:29.748887Z","submitted_at":"2022-09-14T17:24:07Z","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.06794","snapshot_observed_at":"2026-08-15T22:47:49.063146Z","title":"PaLI: A Jointly- Scaled Multilingual Language-Image Model","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.063146Z"},"links":{"cited_paper":"/paper/2209.06794","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:f898e373aba08125655f3b5543fade98e1442e04817939e91580a081542303b9","observation_id":"571a04cb-9f43-4929-ab0f-3c26ca496045","resolution":{"observed_at":"2026-08-15T22:47:49.063146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18565","last_updated":"2023-05-29T18:58:38Z","snapshot_observed_at":"2026-08-07T07:27:51.369423Z","submitted_at":"2023-05-29T18:58:38Z","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18565","snapshot_observed_at":"2026-08-15T22:47:49.067724Z","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.067724Z"},"links":{"cited_paper":"/paper/2305.18565","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:44eeb716da29ab3d5ee743687d631b0b55e029f03c79ac3e5b228967ce04a6cc","observation_id":"beb628dc-a79c-441d-b713-aeae39b9511f","resolution":{"observed_at":"2026-08-15T22:47:49.067724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.536351Z","title":"Command R","venue":null,"work_id":"498741ef-202f-4d41-9aea-a1f1a85f54c7","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.072264Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:5867c8ba953f26e5f0b866869297893406a6d0f7911f8779e26445ae74323a6b","observation_id":"4b1fa9a7-9aa9-4813-9bd3-f003bb63ac61","resolution":{"observed_at":"2026-08-15T22:47:49.540544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.17146","snapshot_observed_at":"2026-08-15T22:47:49.076972Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the- Art Multimodal Models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.076972Z"},"links":{"cited_paper":"/paper/2409.17146","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:c6da01f67583fe7a3cd91030d7fef05c18dc3d0c4898cb076bd9e86a6037f1c8","observation_id":"4b29b4f1-c834-4464-a24f-ca449b161f2c","resolution":{"observed_at":"2026-08-15T22:47:49.076972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.522857Z","title":"Detoxify","venue":null,"work_id":"6054d85d-3306-4a2d-b140-b71f6d30685b","year":2020},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.081229Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:6569d60f2d5bf0e2c7ad6d3b63b22725ce8234973bc6972f43b83860db953485","observation_id":"11a0ef41-deff-45d3-befd-cdea1cce37d2","resolution":{"observed_at":"2026-08-15T22:47:49.527155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05113","last_updated":"2025-06-06T17:08:12Z","snapshot_observed_at":"2026-08-18T12:58:14.839534Z","submitted_at":"2024-06-07T17:44:32Z","title":"LlavaGuard: An Open VLM-based Framework for Safeguarding Vision Datasets and Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05113","snapshot_observed_at":"2026-08-15T22:47:49.084993Z","title":"LLavaGuard: VLM-based Safeguards for Vision Dataset Curation and Safety Assess- ment","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.084993Z"},"links":{"cited_paper":"/paper/2406.05113","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:3d4262c16a90597cab9d9461abb4f39a73ca2e6f1dbdfb51c93a995eae9dd811","observation_id":"b1b27d42-1677-40a9-80d5-d26b7f492402","resolution":{"observed_at":"2026-08-15T22:47:49.084993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.509594Z","title":"Jailbreakzoo: Survey, landscapes, and horizons in jailbreaking large language an d vision-language models, 2024","venue":null,"work_id":"3e10f0bd-fa18-41c9-9f84-72b31b8db44a","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.089163Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:c323981485f2a7ddd3dec67211b218bc7b21e4acb289688db57837004480637b","observation_id":"28400e92-8541-4288-8602-28fc2a78a483","resolution":{"observed_at":"2026-08-15T22:47:49.513777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.496445Z","title":"Vhelm: A holistic evaluation of vision language mod- els, 2024","venue":null,"work_id":"c91817c2-a7d2-4f65-9f3b-67b1284a2644","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.092974Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:283ed1dbea147cb1fd7b51cf200497ae0a2a2fb60f0e68d7a843d0b2c3204db6","observation_id":"72617fd2-c0ba-45ae-8ca4-8d833f511516","resolution":{"observed_at":"2026-08-15T22:47:49.500892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.482138Z","title":"Elite: Enhanced language-image toxicity evaluation for safety, 2025","venue":null,"work_id":"6c0b89e6-cfb1-4277-be17-547209bdcac2","year":2025},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.096768Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:2a952358915143dd7b246c6a6ad9427d0ec1a08412ec3210ae85df3fe0a3cd97","observation_id":"0016f5da-6da0-4bd6-832d-0f41ca6cdf4f","resolution":{"observed_at":"2026-08-15T22:47:49.487453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.467614Z","title":"Improved Baselines with Visual Instruction Tuning, 2023","venue":null,"work_id":"7b456297-9e17-4fba-92bf-debed505a1ef","year":2023},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.100663Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:6f5d2c28af45632796181387858a1b2bc0451c7ce05d3235cf3d10fe86a623a2","observation_id":"7c52c311-f8aa-44e2-982b-3c03ae64f2f2","resolution":{"observed_at":"2026-08-15T22:47:49.472673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.454245Z","title":"Visual Instruction Tuning, 2023","venue":null,"work_id":"56ee0924-965c-451d-9284-54449a6024ff","year":2023},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.104582Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:e20c0c891383d6b9243d2c326fe013a64802b1432c8b0acfd71ed376961be277","observation_id":"0f0d46d8-289f-45b3-80c1-1269dd77d017","resolution":{"observed_at":"2026-08-15T22:47:49.458294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.441320Z","title":"Mm-safetybench: A benchmark for safety eval- uation of multimodal large language models, 2024","venue":null,"work_id":"a2b112ea-410f-47b2-81da-25d35f03728d","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.108332Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:9529254b037e2b9a7cc378908973663711334eea533c2fcb81014a7918d0ebdb","observation_id":"13475d53-5d5d-486c-b399-112d1f2cb2f0","resolution":{"observed_at":"2026-08-15T22:47:49.445746Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.427351Z","title":"Towards interpreting visual infor - mation processing in vision-language models, 2024","venue":null,"work_id":"d7c9f03e-a3c1-45d5-8ca8-70c5aeedcfb4","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.112379Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:b728621fe5f657eded30bb86a32c39f5a3b48bd03319151a0131b8f2ce0ce372","observation_id":"2a99a062-e29d-44f4-b738-b0a4e56e1c93","resolution":{"observed_at":"2026-08-15T22:47:49.431701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02992","last_updated":"2024-04-26T01:24:57Z","snapshot_observed_at":"2026-08-16T14:55:02.836558Z","submitted_at":"2023-10-04T17:28:44Z","title":"Kosmos-G: Generating Images in Context with Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02992","snapshot_observed_at":"2026-08-15T22:47:49.116160Z","title":"Kosmos-G: Generating Images in Context with Multimodal Large Language Models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.116160Z"},"links":{"cited_paper":"/paper/2310.02992","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:507e52d387708ae8a9a52b7a36cc98fc8cbaa47589025bb27113ca46f61edc0d","observation_id":"92697fc4-b12a-4c82-99dd-ea89b95b9e8a","resolution":{"observed_at":"2026-08-15T22:47:49.116160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-08-12T12:24:23.815073Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-15T22:47:49.120423Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.120423Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:eebb5c8b2d6f45148d4e09edeb36e324cdf4f2f7f2de26e075419ccf2d0a794c","observation_id":"1647ad66-906c-4a99-940b-0121ea53c54e","resolution":{"observed_at":"2026-08-15T22:47:49.120423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.413873Z","title":"Learn- ing Transferable Visual Models From Natural Language Su- pervision","venue":null,"work_id":"670e4ea5-feb8-4f48-99b6-371907cbf038","year":2021},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.124792Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:20dad08bff82307687457b09533272e8576ce5f31ad5e20e58d40230678df15a","observation_id":"088b5361-2522-4ad9-b26a-8b90d41c74a5","resolution":{"observed_at":"2026-08-15T22:47:49.418176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.400003Z","title":"Training-free mitigation of language reasoning degradation after multimodal instruction tuning, 2024","venue":null,"work_id":"6e02c39e-88c1-4502-8a6d-bed925dd46a2","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.130340Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:d0eb146f0aaf0812adcedfb83759f7e25e3970174e99debceb2c75992b553c46","observation_id":"6543f544-6704-401b-9c6b-d4d5116611c4","resolution":{"observed_at":"2026-08-15T22:47:49.404586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.386399Z","title":"Laion-5b: An open large-scale dataset for training next generation image-text models, 2022","venue":null,"work_id":"b4dcb862-0637-4154-ab32-4f4170133c23","year":2022},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.134378Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:44daa23483b3205e52eaf25c35c0af2a536c4e5ea7b4707780a47127454aea56","observation_id":"852ddd3e-c245-4c8d-80de-3d3a7fbc7846","resolution":{"observed_at":"2026-08-15T22:47:49.391510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.373351Z","title":"From pixels to prose: A large dataset of dense image cap- tions, 2024","venue":null,"work_id":"aa871f7d-6569-420a-b666-b8bffd47d8b9","year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.138090Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:346a54db896f640460c7310474991972db6b01168d16fc960336f6f09cd97e0d","observation_id":"66fd9b2b-a4bd-4769-a8c1-b8851402f569","resolution":{"observed_at":"2026-08-15T22:47:49.378068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.141758Z","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.141758Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:e3ba989a63a54a4b9fe67c34769898af82715f036a1d120996201dbaecf0adda","observation_id":"0e45e7a4-cd81-4d42-b2c9-df0cac1f1b5f","resolution":{"observed_at":"2026-08-15T22:47:49.141758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-15T22:47:49.145359Z","title":"Qwen2-VL: Enhancing Vision-Language Model’s Perception of the World at Any Resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.145359Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:2996f2e8faf78e17739feae10838c0ae86069b13a4b90f4cd3673fc9c8fb05b6","observation_id":"e120afa9-4647-45ad-afdd-aae5e0c7bdef","resolution":{"observed_at":"2026-08-15T22:47:49.145359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.148914Z","title":"Florence-2: Advancing a uniﬁed representation for a variet y of vision tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.148914Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:21e8fae7fa324bd61c3d9342a810f41b7f47f983f5715a5cee262a9ae57b2635","observation_id":"1aeacf78-2ebc-4b92-840c-ed201d70f13a","resolution":{"observed_at":"2026-08-15T22:47:49.148914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16153","last_updated":"2025-01-26T15:34:59Z","snapshot_observed_at":"2026-08-19T02:58:15.917051Z","submitted_at":"2024-10-21T16:19:41Z","title":"Pangea: A Fully Open Multilingual Multimodal LLM for 39 Languages","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16153","snapshot_observed_at":"2026-08-15T22:47:49.152993Z","title":"Pangea: A Fully Open Multilin- gual Multimodal LLM for 39 Languages","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.152993Z"},"links":{"cited_paper":"/paper/2410.16153","citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:bcbe12f1241c7bcee64828a9c8ad685d7507cef49e6bd8dc8fe42569f29f8f43","observation_id":"3da6cc34-7eb3-48ab-9aca-e841e01b8453","resolution":{"observed_at":"2026-08-15T22:47:49.152993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.344680Z","title":"Sigmoid Loss for Language Image Pre- Training","venue":null,"work_id":"2c1138f1-7cd3-4abf-919e-1e0801acbd89","year":2023},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.157150Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:4982ea272c075444cd4a75dfc9dc7f2567457fe9f050a171a527554902d7a92f","observation_id":"4f8a7d9c-5f57-4835-9f93-a60c693cfe9e","resolution":{"observed_at":"2026-08-15T22:47:49.348897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.331940Z","title":"Spa-vl: A comprehensive safety preference alignment dataset for vi - sion language model, 2025","venue":null,"work_id":"8a3e2cd9-f9f5-487c-8409-b2d69ad5081b","year":2025},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.161173Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:fcb241bf3a23843135bc6abcf645a2b3cda3f198da23273f052e1f944fb1afc9","observation_id":"23d66c40-6f80-4e29-a430-85d07664c789","resolution":{"observed_at":"2026-08-15T22:47:49.336125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.318481Z","title":"Zero-shot defense against toxic images via inherent multimodal alignment in lvlms, 2025","venue":null,"work_id":"ceb2b722-874c-4d03-96bf-2a99df0eeba6","year":2025},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.165134Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:2a05d8985c2f29b73f0ae7482f5cf623aaddce39bf62a352dcfa4e369c9396bf","observation_id":"aeeabb02-1824-46e6-aa80-75e3ffa4eb70","resolution":{"observed_at":"2026-08-15T22:47:49.322869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:47:49.303477Z","title":"Un- derstanding and rectifying safety perception distortion i n vlms, 2025","venue":null,"work_id":"d11e45b1-20c3-48c4-bd2b-a5792a7dc2bc","year":2025},"citing_paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T22:47:49.169054Z"},"links":{"citing_paper":"/paper/2505.06356"},"observation_digest":"sha256:a3865a8d73008f1f6ae1f27f45dfb3e6466254e5c6417ee8c434f59420e79bbb","observation_id":"53243062-5057-4e78-81f9-1e3cb69f06cd","resolution":{"observed_at":"2026-08-15T22:47:49.309450Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.06356","last_updated":"2025-05-09T18:01:50Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-19T03:39:43.560799Z","submitted_at":"2025-05-09T18:01:50Z","title":"Understanding and Mitigating Toxicity in Image-Text Pretraining Datasets: A Case Study on LLaVA"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":0,"verified_fuzzy":20},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 0 inbound Pith citation observations for arXiv:2505.06356."}