{"as_of":"2026-08-16T04:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:abdfaa1a63eaf08f3502227af688d22e82684eb311ca63f5d8a05aae9ce2b919","coverage":[{"denominator":76,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":76,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:03:54.710735Z","state":"measured"},{"denominator":76,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":76,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.02791/citation-record","integrity":"/paper/2608.02791/integrity","json":"/paper/2608.02791/citation-record.json","paper":"/paper/2608.02791"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.758237Z","title":"Better, stronger, faster: Tackling the trilemma in mllm-based segmentation with simultaneous textual mask prediction,","venue":null,"work_id":"e1d6d2d1-189a-4416-bf74-b3537df0150b","year":2026},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.451796Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:307c105b48eac590962ba517fd3a1d98a6af305e47d8a746fdaad5653508bdfa","observation_id":"04b08a1d-1618-4655-a739-2cf0e76629af","resolution":{"observed_at":"2026-08-15T15:03:55.761672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-15T15:03:54.455763Z","title":"Qwen3 technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.455763Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:574d4e96300bff70b098a44169821ad6c6430cfbce83dbe6dd861a7595b10a2a","observation_id":"b8bfe145-ca25-4b8d-8ae0-6e0548181006","resolution":{"observed_at":"2026-08-15T15:03:54.455763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.748927Z","title":"LION: Empowering multimodal large language model with dual-level visual knowledge,","venue":null,"work_id":"26fdf41b-c080-49e8-9b40-e4f2a2cf5993","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.458646Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:88e44ba26287811f5ccd096a2bdc7ed9c64c82bc3108a6e14e2f0649fa62d3aa","observation_id":"3d1abe1e-a1fb-45e0-b662-4ffe1e1796eb","resolution":{"observed_at":"2026-08-15T15:03:55.752270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.739015Z","title":"MLLMs know where to look: Training-free perception of small visual details with multimodal llms,","venue":null,"work_id":"843b3cc8-4160-4552-a365-74bbb7d9e87f","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.461511Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:4495be1323f6ee7f7fd48a89299baaae4c59f8cb4402cca81bab0945e7696476","observation_id":"ca71740f-6953-4b02-9706-29476b94628c","resolution":{"observed_at":"2026-08-15T15:03:55.742657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.464691Z","title":"Detect anything via next point prediction,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.464691Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:8e7d6c7a9c7fd23df083944a8f5caeb51162edd0382855adc638e3658367c68f","observation_id":"fd34f1a2-1f9d-4b9a-92b8-e75336cd8852","resolution":{"observed_at":"2026-08-15T15:03:54.464691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-10T18:37:57.419939Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-15T15:03:54.467561Z","title":"InternVL3: Exploring advanced training and test-time recipes for open-source multimodal models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.467561Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:afec0129877ac4e43e26cdd544d44cec35ab9f96a1d6f303caf85d53e1abc0af","observation_id":"ee3bf2eb-43a2-496a-881a-07ff99bee524","resolution":{"observed_at":"2026-08-15T15:03:54.467561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-15T15:03:54.471468Z","title":"LLaV A- OneVision: Easy visual task transfer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.471468Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:d389de7fb536080b12f0e99c7aabc95d9d2faa8f232e8923fbf107788b795119","observation_id":"43e2b31a-79ee-4d7c-a186-bdf99bf31cac","resolution":{"observed_at":"2026-08-15T15:03:54.471468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.728530Z","title":"PhD: A ChatGPT-prompted visual halluci- nation evaluation dataset,","venue":null,"work_id":"0e87e6a3-4427-421f-a736-7f45ea637a4b","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.475046Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f76d5f4c8596c771855bad80e3e441d9a519dff1c2445d021079b9ab72b43a3a","observation_id":"c482a154-3187-4ee1-abb6-cc7a6053be76","resolution":{"observed_at":"2026-08-15T15:03:55.732220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-15T15:03:54.478261Z","title":"Qwen2.5- VL technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.478261Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:6e23f32769fe236c0ec977c3a9a842b36dc5947934ae3df700d2485a775ef493","observation_id":"fe07297c-e113-4a74-a93a-53d6c32697a0","resolution":{"observed_at":"2026-08-15T15:03:54.478261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.16785","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.991024Z","title":"Segmentation as a plug-and-play capability for frozen multimodal LLMs,","venue":null,"work_id":"d6463179-e8aa-49ac-a856-7527a3e87024","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.481589Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:9838da27fd7d88587adc0b5dff91d5f82a8b829778b84368c47c01b676379af5","observation_id":"74ed9fa0-c20d-4ae6-b070-15bd8cf4e1c8","resolution":{"observed_at":"2026-08-15T15:03:54.999538Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.717491Z","title":"Text4Seg: Reimagining image segmentation as text generation,","venue":null,"work_id":"f8d3de6d-884a-4855-91a9-1455c743fdff","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.485332Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:cc28eb88146ab84687910987e4dd396713d2669630dc92dbd6805440d8d42646","observation_id":"a224531b-01d7-47f8-abe9-b5079e6bed0b","resolution":{"observed_at":"2026-08-15T15:03:55.721351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.706038Z","title":"LISA: Reasoning segmentation via large language model,","venue":null,"work_id":"a264eff0-a699-4d4d-97ae-08156a4304a2","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.488627Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f5c3642353254302ae5b6deda76d0032ea92fde090d22adcb29b471d8a76c9a5","observation_id":"cfd23eaa-e655-4fde-b097-85598dce54c2","resolution":{"observed_at":"2026-08-15T15:03:55.710085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.695107Z","title":"GSV A: Generalized segmentation via multimodal large language models,","venue":null,"work_id":"1e9563bb-a170-4359-aeaa-c34cbb2c1489","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.491801Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:1427d24649e0aa904a1c40a1504f9789e5c0cfd2b5e3891f6b5dff40cccf42df","observation_id":"e441a837-c859-4918-bd43-56a6b0bd4abb","resolution":{"observed_at":"2026-08-15T15:03:55.699166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.682959Z","title":"PixelLM: Pixel reasoning with large multimodal model,","venue":null,"work_id":"78fef7ce-5f1a-4d50-9e5a-90a76fbf54dc","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.495099Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:aa20481c3e93f9eb5ae508bfa9dae52cae399d4673b9354d447580d747724e80","observation_id":"eb594ba4-8168-43c7-8986-3c49e5eeae95","resolution":{"observed_at":"2026-08-15T15:03:55.687046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.670619Z","title":"MMR: A large-scale benchmark dataset for multi-target and multi- granularity reasoning segmentation,","venue":null,"work_id":"84b42882-f6b5-417a-8138-9858a6edaea4","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.498237Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:98b4c1075514ddc10a1abf274072988328151609dc14607b2c639849ad2c518b","observation_id":"7c0d53ec-299e-45d1-b124-5e19aee050d1","resolution":{"observed_at":"2026-08-15T15:03:55.675152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.654286Z","title":"Reasoning to attend: Try to understand how<SEG>token works,","venue":null,"work_id":"92af199c-9082-470f-b190-805a7bd59962","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.501558Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:c43ea849759fe2d72cd3e0a8f2ceed82d1a48c5ce070bd95a911fb10a90714b9","observation_id":"fef7e871-c616-421b-9dec-eb69d416b313","resolution":{"observed_at":"2026-08-15T15:03:55.658171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.643603Z","title":"VisionLLM v2: An end-to-end generalist multimodal large language SUBMISSION 17 model for hundreds of vision-language tasks,","venue":null,"work_id":"7abedd77-1cc7-404a-bc2c-3eed306b9546","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.504748Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:3ff31d5303221d5199e7a81fc90edbd5dbb0be124ba9f8318a5989c96aa5c2a7","observation_id":"03e114b1-1560-4466-9e71-3d6dd8b1a0d0","resolution":{"observed_at":"2026-08-15T15:03:55.647250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06520","last_updated":"2026-05-31T09:29:40Z","snapshot_observed_at":"2026-08-07T17:19:13.101842Z","submitted_at":"2025-03-09T08:48:51Z","title":"Seg-Zero: Reasoning-Chain Guided Segmentation via Cognitive Reinforcement","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06520","snapshot_observed_at":"2026-08-15T15:03:54.508438Z","title":"Seg-Zero: Reasoning-chain guided seg- mentation via cognitive reinforcement,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.508438Z"},"links":{"cited_paper":"/paper/2503.06520","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:8ee83e452f96e45cbd8a59f9b829307e28938cde2bb284502e9d2199b26ae833","observation_id":"a2b4c67b-43b8-4186-bbfb-521c4f802bb2","resolution":{"observed_at":"2026-08-15T15:03:54.508438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.632145Z","title":"SegAgent: Exploring pixel understanding capabilities in mllms by imitating human annotator trajectories,","venue":null,"work_id":"69e6686a-7003-4646-8b55-ef3a1504000e","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.512508Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:b777fa12fbb287fed639ad44aac66a19dc54f3fe3a3dfab906051c25a3e5dcf4","observation_id":"1cd02046-2b11-4129-b374-725484920792","resolution":{"observed_at":"2026-08-15T15:03:55.636086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.620259Z","title":"Text4Seg++: Advancing image segmentation via generative language modeling,","venue":null,"work_id":"a0c10607-41ad-4719-bdb0-e7b4af4a6e77","year":2026},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.515797Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:2ccfdb620b0b50c1784d07067684d5a191b957002891ed2198aa1698c6297c13","observation_id":"406235f7-a893-424f-82dc-c6601765be84","resolution":{"observed_at":"2026-08-15T15:03:55.624006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.604886Z","title":"GLaMM: Pixel grounding large multimodal model,","venue":null,"work_id":"9633ba3b-6d90-4640-9733-27a65ff5e12a","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.519060Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f3a95510255a50be001705f4e61e30b87b5828edf057ef393c526748991ac326","observation_id":"0593f58f-a32d-4410-bbc3-0c92be3785c8","resolution":{"observed_at":"2026-08-15T15:03:55.610352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.592220Z","title":"See say and segment: Teaching LMMs to overcome false premises,","venue":null,"work_id":"4e5def89-ddb4-4eaf-89fe-d32e4b351791","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.522174Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f1719dad39607f4138368ccbbaa208ce4d76df95286ea15167009cb68c0d6f6d","observation_id":"2f93bd10-bd8f-48bc-8cda-0272fd706c1d","resolution":{"observed_at":"2026-08-15T15:03:55.597190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.580307Z","title":"VisionLLM: Large language model is also an open-ended decoder for vision-centric tasks,","venue":null,"work_id":"46ed41b9-1d73-4017-b0b2-b1340dfc97bd","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.525407Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:2f9c66a703034631c0ff9cfacbde8714d5310e9c75606b11a17c732ec26a202d","observation_id":"6c2cd546-1986-45f4-9431-bc70bd7b7acd","resolution":{"observed_at":"2026-08-15T15:03:55.584499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.568626Z","title":"ReferItGame: Referring to objects in photographs of natural scenes,","venue":null,"work_id":"93c56780-4f60-4c38-b917-6c108e3e0298","year":2014},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.529505Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:7fddbd5cae27e11581fd79b22e16a3a28be9c4572ee70772a80b4228db61a639","observation_id":"29f44c12-a969-43fd-9612-dd62a074efcb","resolution":{"observed_at":"2026-08-15T15:03:55.572467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.558495Z","title":"Generation and comprehension of un- ambiguous object descriptions,","venue":null,"work_id":"7eed6b1a-22f7-4d56-92fd-151c810c1e2f","year":2016},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.532686Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:72ec9b6226849caf0d08d69ef4b0776fc611497d95a896acde5fed7bc0a7eb75","observation_id":"4b02c12e-dc90-4336-ac4f-24207e27f9fb","resolution":{"observed_at":"2026-08-15T15:03:55.562135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-15T15:03:54.535460Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.535460Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:315d35960fac9d99ad12f6f45f2847358dc81b1281da07c9aac2ce9851e60356","observation_id":"719a921b-2021-4c44-b8e0-e1ba6d91a1df","resolution":{"observed_at":"2026-08-15T15:03:54.535460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-08-13T17:41:53.092611Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-15T15:03:54.538716Z","title":"Scaling laws for neural language models,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.538716Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:10ffc47062311203932ffe8fbbc44fb8fa6ca09cc77db704b42c297e2c09bcf4","observation_id":"d939e361-afbb-4349-97ee-e866c4f81170","resolution":{"observed_at":"2026-08-15T15:03:54.538716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.549313Z","title":"Hello GPT-4o,","venue":null,"work_id":"4cc076a5-782d-4c1f-b601-cbcf5330674e","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.541848Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:57463fc9f5307f129a5cd1a6611aefe951d378ea7dec0c7e06bfd3ad15c0da49","observation_id":"1426642a-64c0-4a48-85f4-5549ba467a59","resolution":{"observed_at":"2026-08-15T15:03:55.552541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.539703Z","title":"Google Gemini 2.5 Pro,","venue":null,"work_id":"d991ae1d-6d85-473b-99e2-b4a72676b553","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.544489Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:2fb033bf4f5907da93f023c0dd8b7942a4e59ceb582818694a9ebcc9ca3ec46c","observation_id":"e219408a-a3d9-46a8-9b7e-8476dfae5a2c","resolution":{"observed_at":"2026-08-15T15:03:55.543034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.529199Z","title":"Empowering small VLMs to think with dynamic memorization and exploration,","venue":null,"work_id":"85cb7c72-cebd-4401-aac5-1bb07d4f7e61","year":2026},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.547166Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:700b07399674789aca28a99e687ce30917fe41e31a7faaa61f8195ae6cd9d14b","observation_id":"912d1f00-b430-4f06-84fd-bc4c4677137f","resolution":{"observed_at":"2026-08-15T15:03:55.532537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.549912Z","title":"Flamingo: a visual language model for few-shot learning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.549912Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f106918108019ab4be1ca0c4ff80fc6388cad3a37ff02f2bb04766e8c0a43e83","observation_id":"ba301dd4-4fb2-48c0-9f4b-9cd6aa462077","resolution":{"observed_at":"2026-08-15T15:03:54.549912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.511652Z","title":"InstructBLIP: Towards general- purpose vision-language models with instruction tuning,","venue":null,"work_id":"f407d24d-c0d2-488d-89cd-c89a3f71c8d7","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.552627Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:af1ef2163b63236eda3e3098ef50bb44d6a364941e6cd389511a2fa14b6e14e7","observation_id":"4ef2b056-5ca2-4e0f-b851-d7cfdeda56ed","resolution":{"observed_at":"2026-08-15T15:03:55.515579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.499624Z","title":"Visual instruc- tion tuning,","venue":null,"work_id":"a0b74af1-79ad-4eee-a644-1530f2cddfef","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.555846Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:d3ff2e9ae50dfc1059f748f755d129e603ce1949ce492af4158738a9d2023a8f","observation_id":"b1bfb3f6-1214-4997-bad2-880feadda6ea","resolution":{"observed_at":"2026-08-15T15:03:55.503849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.559036Z","title":"Improved baselines with visual instruction tuning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.559036Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f690a2ade774f82510693760eab7edc275c68a20713f11fb8ed6ddf4af9c2523","observation_id":"4bcfd626-2366-4f5a-9e93-9a24e3ad82a9","resolution":{"observed_at":"2026-08-15T15:03:54.559036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-15T15:03:54.562380Z","title":"Qwen-VL: A versatile vision- language model for understanding, localization, text reading, and beyond,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.562380Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:5ca6ee5f7d53947dfa6c8dc5aa18ed1dda0958bed959b5f6400fcee201975323","observation_id":"601a7695-856f-42b3-9a2d-7cf4149bce0b","resolution":{"observed_at":"2026-08-15T15:03:54.562380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-08-11T14:42:37.584016Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-15T15:03:54.565807Z","title":"Qwen3- VL technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.565807Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:bbc005ee931a362c9900b6230372a70cfa2e20e826f210a9f4ea9d3cd399b17b","observation_id":"b60787d1-80fa-49d2-a7ce-2e81bbe1d8cf","resolution":{"observed_at":"2026-08-15T15:03:54.565807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.24251","last_updated":"2025-10-05T04:01:18Z","snapshot_observed_at":"2026-08-10T21:20:13.670725Z","submitted_at":"2025-09-29T03:52:01Z","title":"Latent Visual Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.24251","snapshot_observed_at":"2026-08-15T15:03:54.569025Z","title":"Latent visual reason- ing,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.569025Z"},"links":{"cited_paper":"/paper/2509.24251","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:8b0a8417d3880a3a8a7aa0823f61372ecf8fb770895906991eaaa45a9ceae750","observation_id":"d7be95b9-76e9-4630-add3-b89b088f9cb8","resolution":{"observed_at":"2026-08-15T15:03:54.569025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.572969Z","title":"Segment anything,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.572969Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:bf074ca21c2d6f73fda94af98ca4924b63b2fd66d57aae7c30cdf908f0638967","observation_id":"cc610fa0-ba34-4b73-82ad-5953674b34df","resolution":{"observed_at":"2026-08-15T15:03:54.572969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.473311Z","title":"SegLLM: Multi- round reasoning segmentation with large language mod- els,","venue":null,"work_id":"9edaece7-e266-4f09-91eb-a32f5225b2cb","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.576145Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:ee42af489a7e5b12efbe6f919786811b1084e0775dd875217ec8a76ac0728033","observation_id":"89d0900f-e339-48d5-9708-b9d2454358b2","resolution":{"observed_at":"2026-08-15T15:03:55.477234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.461940Z","title":"MLLM can see? dynamic correction decoding for hallucination mitigation,","venue":null,"work_id":"bcce5e80-29b0-4b4d-b5d4-d81e46a67c13","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.579585Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:c472f48b01b108759385b3dcc629fc7a339b06a9eeb9d71b0f77651ff46b72c0","observation_id":"880f5ac2-2e34-495f-ab32-5e666196e4dd","resolution":{"observed_at":"2026-08-15T15:03:55.465603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.451291Z","title":"Grounding multimodal large lan- guage models to the world,","venue":null,"work_id":"33d3a59b-8363-4052-90eb-980f66b4910e","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.582934Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:6f0c8df8657e8595fe20da4e2b62c0d4eeb539b29c5303fff8e3127f66283dd8","observation_id":"0dbd1d34-3a1f-4c57-a305-322c6a077190","resolution":{"observed_at":"2026-08-15T15:03:55.454938Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-08-13T14:42:26.982679Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-15T15:03:54.586131Z","title":"Shikra: Unleashing multimodal LLM’s refer- ential dialogue magic,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.586131Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:a832714cac35858e920841537b376f4b1e1091eb7b5bda2b0e77e121c3c2e241","observation_id":"7a34b36a-3858-4006-9236-d7ef689eefbc","resolution":{"observed_at":"2026-08-15T15:03:54.586131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.439783Z","title":"Perception tokens en- hance visual reasoning in multimodal language models,","venue":null,"work_id":"0a892bc1-6798-4de2-9c35-2528ff214b66","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.590186Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:98b93a9b41fbb3f6c7739bfbf11845c4a66a2ef8aca88710b14c8b07df891932","observation_id":"aa5a3832-948a-4297-8a62-0e08b2f85793","resolution":{"observed_at":"2026-08-15T15:03:55.444001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.427866Z","title":"ViperGPT: Visual inference via python execution for reasoning,","venue":null,"work_id":"d8538bbf-1278-40d1-90f8-b8e87d8f7886","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.593395Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:4b4da9861bc3f2aee9c4121315da354b8cdb7a2cb17fa3431871ff30911ad1c9","observation_id":"eb3ae115-811f-4f68-b83c-eef6c98af6df","resolution":{"observed_at":"2026-08-15T15:03:55.432028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.415736Z","title":"Chameleon: Plug- and-play compositional reasoning with large language models,","venue":null,"work_id":"7110df58-1bc9-458b-8853-7c0439e86c8c","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.596946Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:a05f8a6e2488521c3abfe8639925840a8b912e3f5528582492831388899a53a1","observation_id":"0a5f2e30-0404-48a2-a1c9-a8cbc821d645","resolution":{"observed_at":"2026-08-15T15:03:55.419995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11381","last_updated":"2023-03-20T18:31:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-20T18:31:47Z","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11381","snapshot_observed_at":"2026-08-15T15:03:54.601092Z","title":"MM-REACT: Prompting ChatGPT for multimodal reasoning and ac- tion,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.601092Z"},"links":{"cited_paper":"/paper/2303.11381","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:7eac3982f3b075dbdba18a385e77e63b77548fa6314a132dd286d4e67aee3917","observation_id":"bb0adc8a-f9b2-425d-908b-274ec596363f","resolution":{"observed_at":"2026-08-15T15:03:54.601092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.404062Z","title":"Visual sketchpad: Sketching as a visual chain of thought for multimodal language models,","venue":null,"work_id":"b205aa59-740a-4893-9b89-07034a661f28","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.604239Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:0fb8ce29136b8ef7f49d44297b0025070ed0e352fd3c50742059d794796ba949","observation_id":"b7e4c956-5862-495f-832d-88bbaf4587e6","resolution":{"observed_at":"2026-08-15T15:03:55.408294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.17218","last_updated":"2025-06-20T17:59:31Z","snapshot_observed_at":"2026-08-13T03:28:19.590253Z","submitted_at":"2025-06-20T17:59:31Z","title":"Machine Mental Imagery: Empower Multimodal Reasoning with Latent Visual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.17218","snapshot_observed_at":"2026-08-15T15:03:54.608308Z","title":"Machine mental imagery: Empower multimodal reasoning with latent visual tokens,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.608308Z"},"links":{"cited_paper":"/paper/2506.17218","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:770f8ef81002f02941765893b316f91d4c465e0da9541f65bb9d0d3aef7f4a5a","observation_id":"039fe8d4-6bfa-4653-aa56-f6504ca3128a","resolution":{"observed_at":"2026-08-15T15:03:54.608308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.611963Z","title":"Multimodal chain of continu- ous thought for latent-space reasoning in vision-language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.611963Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:88e57c72f8042e924632f31751271614a338f7a1db840583084fd64993cb0dcc","observation_id":"d51d1135-2c12-457c-8f96-f903be002fff","resolution":{"observed_at":"2026-08-15T15:03:54.611963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.392050Z","title":"GRES: Generalized referring expression segmentation,","venue":null,"work_id":"a8f056bd-9d49-4099-a207-ee26bd65aed6","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.615118Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:aae7187dda89c75e56d370498ec4456c70a0d6ec5970767e5b4176007987e016","observation_id":"11a287d2-81e2-49de-92d2-4c52baffc397","resolution":{"observed_at":"2026-08-15T15:03:55.396241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.380486Z","title":"Rotated multi-scale interaction network for referring remote sensing image segmentation,","venue":null,"work_id":"f04cbb5e-6c6e-47b3-85de-8ff5fb881276","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.618641Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:a7bd85c0542e124c51da774ee72bde14260fc4da61e40abdd4c44caf35b6895e","observation_id":"47842052-bb3f-4e0b-83c1-8a29e5673b6b","resolution":{"observed_at":"2026-08-15T15:03:55.384191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.09644","last_updated":"2025-04-13T16:36:47Z","snapshot_observed_at":"2026-08-07T16:06:07.544996Z","submitted_at":"2025-04-13T16:36:47Z","title":"SegEarth-R1: Geospatial Pixel Reasoning via Large Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.09644","snapshot_observed_at":"2026-08-15T15:03:54.622033Z","title":"SegEarth-R1: Geospatial pixel reasoning via large language model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.622033Z"},"links":{"cited_paper":"/paper/2504.09644","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f3b6e13bc997d0321043a9123a318b1b8a7c009acf203791b6b37a98a3c98990","observation_id":"1620cbd2-528e-4e73-91ae-c95901437297","resolution":{"observed_at":"2026-08-15T15:03:54.622033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.369315Z","title":"COCO-Stuff: Thing and stuff classes in context,","venue":null,"work_id":"7eebab0f-9be6-49e1-82f1-f77ff813fa3c","year":2018},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.625478Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:14189ef696fe821f743d4e1f2a8b9038d3c20949aa1947314fffb667fde4bdaa","observation_id":"0b84a510-00d2-4954-b6bd-49f51a874a90","resolution":{"observed_at":"2026-08-15T15:03:55.373193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.628758Z","title":"Microsoft COCO: Common objects in context,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.628758Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:fcd1c6cd66093827bc9459e59258150ec438c68de555d018f7f2720e0cb4e67c","observation_id":"d2717e80-0ea9-4c2e-9947-75f9d4fdbce6","resolution":{"observed_at":"2026-08-15T15:03:54.628758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.349226Z","title":"Universal instance perception as object discovery and retrieval,","venue":null,"work_id":"96cdf1d8-9fee-4230-9b24-748d41b752ab","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.632035Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:d3f11e8d1ea42ad20cc414b44f197483efa3059713f9e2bc661a18ba02bb446e","observation_id":"eb93b69d-18f0-4682-9bec-984b6b5b5123","resolution":{"observed_at":"2026-08-15T15:03:55.352548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.338202Z","title":"PolyFormer: Referring im- age segmentation as sequential polygon generation,","venue":null,"work_id":"b785017e-258d-4a39-9f53-bb9dff9546b7","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.636007Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:fc9998d84bb45fc3c742e0db021a653f9c56d266a2ae1e5a61cee3af34149fad","observation_id":"3ec1a4bd-74fd-4935-8aa8-974b566a410c","resolution":{"observed_at":"2026-08-15T15:03:55.342361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.326559Z","title":"Language-aware vision transformer for referring segmentation,","venue":null,"work_id":"1e42d3d8-5c45-4e5a-b890-43e6a5308afb","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.640328Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:9357034a120b0de684f149df62cc7f29fe9341f4b8a174edc8b74252abe85e59","observation_id":"725c51a9-44cd-4e61-9b12-4e53da9d4a99","resolution":{"observed_at":"2026-08-15T15:03:55.330716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.316628Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip,","venue":null,"work_id":"cc271d86-71bd-4b6e-a791-9f7020d13c1e","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.644661Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:05419c17fc45a3c1dbac0c7d7163a7548b121b72240dc94adee22c9e5e27e354","observation_id":"26b018bf-29de-478c-8b7d-71d677686b35","resolution":{"observed_at":"2026-08-15T15:03:55.320074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.306216Z","title":"NExT- Chat: An LMM for chat, detection and segmentation,","venue":null,"work_id":"17389977-74d3-4b3f-819b-114a23c00df8","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.648336Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:2df160bfe3f4efd9b640284a14fa18f7502e0749c9d92264962f35f62a5ae7d6","observation_id":"803c6677-c8c7-45ba-a00e-0f9ba6204390","resolution":{"observed_at":"2026-08-15T15:03:55.309625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11904","last_updated":"2025-05-10T15:43:29Z","snapshot_observed_at":"2026-08-14T08:28:09.837202Z","submitted_at":"2024-11-16T05:12:11Z","title":"GeoGround: A Unified Large Vision-Language Model for Remote Sensing Visual Grounding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11904","snapshot_observed_at":"2026-08-15T15:03:54.651396Z","title":"GeoGround: A unified large vision-language model for remote sensing visual grounding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.651396Z"},"links":{"cited_paper":"/paper/2411.11904","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:58b88cd81953b19a522bdc1e035c9153dfbc1a974d0b6f88bfdd19c817c1af86","observation_id":"46fea211-ec73-42f4-af8c-c4d41d115b7f","resolution":{"observed_at":"2026-08-15T15:03:54.651396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.296075Z","title":"Semantic understanding of scenes through the ADE20K dataset,","venue":null,"work_id":"b046bfae-24f5-419d-abf7-0eaf331b21e6","year":2019},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.655043Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:45d072412c6a6ce9ae1b98070948d7dea408f27227d7181eb999501cb649bd44","observation_id":"7e48c61d-4e4a-4342-99f8-ab0fd99b6040","resolution":{"observed_at":"2026-08-15T15:03:55.299727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.285513Z","title":"The role of context for object detection and semantic segmentation in the wild,","venue":null,"work_id":"ab419ded-e2c0-4f21-9cbb-37c7bd361562","year":2014},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.658153Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:38d27ea03d64f08f0f1e468e57334ca119b000cff4e8f84c224fdaa6c2dba875","observation_id":"181e0006-ebcc-47e4-bf79-0cb248aab9ad","resolution":{"observed_at":"2026-08-15T15:03:55.289134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.274474Z","title":"The PASCAL visual object classes (VOC) challenge,","venue":null,"work_id":"0640d10c-eb17-47d8-b244-18beaa484c6a","year":2010},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.661299Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:aaedb7bb9760bb95279343a52d2e74bad8930e9036735cfe1cc32fb7bd9b326b","observation_id":"eddf5a35-7074-4569-a919-967d4eb3d5a1","resolution":{"observed_at":"2026-08-15T15:03:55.278159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.264026Z","title":"ClearCLIP: Decomposing clip representa- tions for dense vision-language inference,","venue":null,"work_id":"20c40c31-2ca6-4738-8699-c88900ef80bd","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.664544Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:2442d4c0c9dd0cdb6e5ccb7cca79f99e13d0d1a4deda59d3c4342aa584f06100","observation_id":"e1620c24-8c25-44af-a28b-225d4a69d47e","resolution":{"observed_at":"2026-08-15T15:03:55.267908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.253347Z","title":"ProxyCLIP: Proxy attention improves clip for open-vocabulary segmentation,","venue":null,"work_id":"0f275c0e-3790-4242-b7ea-cce18af05a7c","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.667767Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:c411a65344e8d7caf18b33f1e2f09c7214d574d9aac9e2c3c326a42234bf9838","observation_id":"dc9135cc-a1c3-4133-8670-8aa8dab1b5c1","resolution":{"observed_at":"2026-08-15T15:03:55.257251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.242182Z","title":"Open-vocabulary universal image segmentation with maskclip,","venue":null,"work_id":"528a3d64-bf96-4530-a23b-93ef17682184","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.670894Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:62769de7513a7fd7aaa3cc9053a3ddb70201b0dab340c900b98b94e068c0cb42","observation_id":"ae84c898-2067-4b56-abdb-828bdc0b91e7","resolution":{"observed_at":"2026-08-15T15:03:55.246133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.232184Z","title":"GroupViT: Semantic segmenta- tion emerges from text supervision,","venue":null,"work_id":"bb7c2457-8ca5-453c-9027-b2a00dcc2633","year":2022},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.674129Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:f1c0481826af3ef63715f26a3b4b8a7c49f9fdc63db24cccfba230bd2635f999","observation_id":"284bd96e-62dd-4aec-8b06-0bf807a575c3","resolution":{"observed_at":"2026-08-15T15:03:55.235335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.222154Z","title":"SAN: Side adapter network for open-vocabulary semantic seg- mentation,","venue":null,"work_id":"7aed77e2-a35a-476d-8b44-56edc8bf76ed","year":2023},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.677975Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:245e8729971552160282102a8175c57d0e0222be19f981a2747409b9e4c67533","observation_id":"d0702a6f-98d4-4145-aafa-76ed3cdfd46c","resolution":{"observed_at":"2026-08-15T15:03:55.225448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08506","last_updated":"2024-04-12T14:40:45Z","snapshot_observed_at":"2026-08-13T00:31:54.264749Z","submitted_at":"2024-04-12T14:40:45Z","title":"LaSagnA: Language-based Segmentation Assistant for Complex Queries","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08506","snapshot_observed_at":"2026-08-15T15:03:54.681762Z","title":"LaSagnA: Language-based segmentation assistant for complex queries,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.681762Z"},"links":{"cited_paper":"/paper/2404.08506","citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:4264c5fbf8f5881ddddb970ff46f30c90627c5cb6c64df69e6d25762881ca8fb","observation_id":"6bfc685c-e110-4a95-aa14-55fcf932c696","resolution":{"observed_at":"2026-08-15T15:03:54.681762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.210732Z","title":"POPEN: Preference-based optimization and ensemble for LVLM-based reasoning segmentation,","venue":null,"work_id":"abfbc91d-ddee-4385-a43d-6a467575c486","year":2025},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.685764Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:b70fcaa67049597512dfd86f7b0765cc1e79ff6860cb66e9e78f333c88a690a5","observation_id":"b02ad1b6-b270-415a-afc2-63967b3a525d","resolution":{"observed_at":"2026-08-15T15:03:55.214842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.197729Z","title":"MMMU: A massive multi-discipline multimodal understanding and reasoning benchmark for expert agi,","venue":null,"work_id":"7ef8a5aa-fe6a-45bb-b8df-55f802c06e0d","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.689762Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:3b4747163b39dfd6a3c80d0ff1973e08008356c60202668094d9d69901a9d9a8","observation_id":"f4f4dae6-1f65-4854-9c44-ec1f3a0942bc","resolution":{"observed_at":"2026-08-15T15:03:55.202978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.183959Z","title":"MMBench: Is your multi-modal model an all-around player?","venue":null,"work_id":"0ebec7b9-5abf-4237-91c2-63624846a056","year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.693636Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:9374119bb8f6461286f668c96bea06febc2d1aa42f3e5589acce0ce745892f47","observation_id":"70d01109-c09b-4ca4-aea6-09f72332c59d","resolution":{"observed_at":"2026-08-15T15:03:55.187936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:54.697655Z","title":"Are we on the right way for evaluating large vision-language models?","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.697655Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:68665b93abf0356556be778a00dd161357c697e58ee8a8ff8e4c7e799b8d0a27","observation_id":"93f5a652-b3c4-4b62-a4d2-0f96a4779fce","resolution":{"observed_at":"2026-08-15T15:03:54.697655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.164202Z","title":"Learn to explain: Multimodal reasoning via thought chains for science question answering,","venue":null,"work_id":"6c9c7042-775b-4662-acb4-d98f46396b3d","year":2022},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.701863Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:6f04f6a257db568dbfe9b8593d2e490dc9b6325354c4bf658d8c3b5e6430cbd6","observation_id":"9def69ac-b21b-49dc-9a2f-35e2016ac5a6","resolution":{"observed_at":"2026-08-15T15:03:55.168942Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.151901Z","title":"Towards VQA models that can read,","venue":null,"work_id":"4262cfcb-bbea-456b-8507-5828b4612c5b","year":2019},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.706323Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:639af88d790fa01510734f6127b8a9c2fa5349476eeb01c4a23cfd0aa91b44e4","observation_id":"2815211c-9384-4853-9633-3416355e1333","resolution":{"observed_at":"2026-08-15T15:03:55.155787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:03:55.140911Z","title":"VizWiz grand challenge: Answering visual questions from blind people,","venue":null,"work_id":"bba1e28a-a9f6-4e40-9905-daa1ffb95066","year":2018},"citing_paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-15T15:03:54.710735Z"},"links":{"citing_paper":"/paper/2608.02791"},"observation_digest":"sha256:a422a393389766fc739423ab8c88fd2cf7ab26f16dfe520743e6b53652b3b2ca","observation_id":"f8b30706-2dfe-42d5-9f36-854c70f92195","resolution":{"observed_at":"2026-08-15T15:03:55.144779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.02791","last_updated":"2026-08-03T18:42:11Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T00:20:22.811058Z","submitted_at":"2026-08-03T18:42:11Z","title":"Better, Stronger, Faster, and Broader: Structured All-Mask Prediction for MLLM-Based Segmentation"},"reference_resolution":{"displayed":76,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":1,"verified_fuzzy":52},"total_outbound_references":76},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 76 of 76 outbound references and 0 inbound Pith citation observations for arXiv:2608.02791."}