{"as_of":"2026-08-09T20:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5a95dd7dade227c8f64479ada9e276fc9e1556ca288658b93c5f7c99719a1543","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T04:34:23.056458Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.05670/citation-record","integrity":"/paper/2608.05670/integrity","json":"/paper/2608.05670/citation-record.json","paper":"/paper/2608.05670"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.11919","last_updated":"2024-11-28T13:35:56Z","snapshot_observed_at":"2026-07-06T19:52:11.774784Z","submitted_at":"2024-11-18T04:06:04Z","title":"VL-Uncertainty: Detecting Hallucination in Large Vision-Language Model via Uncertainty Estimation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11919","snapshot_observed_at":"2026-08-08T04:34:22.718875Z","title":"Vl-uncertainty: Detecting hallucination in large vision-language model via uncertainty estimation.arXiv preprint arXiv:2411.11919, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.718875Z"},"links":{"cited_paper":"/paper/2411.11919","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:f72c2aa6cc7e4ef6061aeacda77c189cb2932e6e57a93e5709392d8a263c3b4f","observation_id":"d8450b18-d026-4d72-800f-7b14558c59b5","resolution":{"observed_at":"2026-08-08T04:34:22.718875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.724591Z","title":"Questioning the stability of visual question answering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.724591Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:33b5ab88ce2a3931876d3ab41811c6a8acff2708b49e55e1512535cd08d5d7cb","observation_id":"8e9e669a-16d2-4301-b7e3-1bee98167ba1","resolution":{"observed_at":"2026-08-08T04:34:22.724591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.979768Z","title":"Efficient test-time scaling for small vision-language models","venue":null,"work_id":"b518ba6c-eb05-448c-a727-d9348e5128d3","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.729343Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:b710ac50d046f8b9a2c3212be5dfd36ef5f261725903a556c78e3a3ed7558c92","observation_id":"291444bc-2dd8-4952-8eb4-8b6fe1c8f41e","resolution":{"observed_at":"2026-08-08T04:34:23.984472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.734303Z","title":"Multi-llm debate: Framework, principals, and interventions.Advances in Neural Information Processing Systems, 37:28938–28964, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.734303Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:0ca70a4950e3971e37302592b4d825c614ee33a53735dc98e9fd73f0d73a2490","observation_id":"e18ebd78-67fa-49bb-a4d7-30900cb577cc","resolution":{"observed_at":"2026-08-08T04:34:22.734303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-08T04:34:22.739239Z","title":"Self-consistency improves chain of thought reasoning in language models.arXiv preprint arXiv:2203.11171, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.739239Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:7cf3a0dc27a38fd22bc1e62f9667d7d55a3b371b7e846d7f96ea97cc6e42038e","observation_id":"94f291c1-af88-4a44-9efd-50b0df790b75","resolution":{"observed_at":"2026-08-08T04:34:22.739239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.744339Z","title":"Consistency and uncertainty: Identifying unreliable responses from black- box vision-language models for selective visual question answering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.744339Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:8b55832e8f9ac344228191f4691e9cbc2a417d4bb910d36b50db9eab56d1fd8a","observation_id":"cf8a19ea-9e1c-41e4-9244-11906c5e7eff","resolution":{"observed_at":"2026-08-08T04:34:22.744339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.945996Z","title":"Decompose and compare consistency: Measuring vlms’ answer reliability via task-decomposition consistency comparison","venue":null,"work_id":"a0859a76-2671-4782-b33b-ce87044e8bf3","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.749605Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:df774e19d1c41e021484a0609d6e7a936ea6f2065095cff3139f3a7affdcd8c0","observation_id":"e0098460-5a52-42ed-a8d1-22f42d7d02eb","resolution":{"observed_at":"2026-08-08T04:34:23.950491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.09664","last_updated":"2023-04-15T12:55:45Z","snapshot_observed_at":"2026-07-06T14:53:27.667483Z","submitted_at":"2023-02-19T20:10:07Z","title":"Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.09664","snapshot_observed_at":"2026-08-08T04:34:22.754207Z","title":"Semantic uncertainty: Linguistic invariances for uncertainty estimation in natural language generation.arXiv preprint arXiv:2302.09664, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.754207Z"},"links":{"cited_paper":"/paper/2302.09664","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:495cd61b207815eb10aad40d256ff98c9fe2e2f661e26e31f55fe7e8571e9864","observation_id":"0ee27387-dfc1-44c7-b617-2707784ac49e","resolution":{"observed_at":"2026-08-08T04:34:22.754207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.780102Z","title":"Detecting hallucinations in large language models using semantic entropy.Nature, 630(8017):625–630, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.780102Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:8f325d52f8c41374361c6558353c69035d3053bf6611ecdc84a34bd96045d49c","observation_id":"4de494b6-d26b-413d-b9b0-2a97935f6137","resolution":{"observed_at":"2026-08-08T04:34:22.780102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15376","last_updated":"2026-04-15T20:47:08Z","snapshot_observed_at":"2026-07-06T23:02:58.361418Z","submitted_at":"2026-04-15T20:47:08Z","title":"Zoom Consistency: A Free Confidence Signal in Multi-Step Visual Grounding Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.15376","snapshot_observed_at":"2026-08-08T04:34:22.806745Z","title":"Zoom consistency: A free confidence signal in multi-step visual grounding pipelines.arXiv preprint arXiv:2604.15376, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.806745Z"},"links":{"cited_paper":"/paper/2604.15376","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:bfa075a0556696047653709f40210af5e7b5314c45548d05ec250bd6d8548b64","observation_id":"2dfa8160-52a3-4f9d-9fc8-bc8f1c3d96e5","resolution":{"observed_at":"2026-08-08T04:34:22.806745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.922795Z","title":"Vauq: Vision-aware uncertainty quantification for lvlm self-evaluation","venue":null,"work_id":"c509ffb6-c006-4240-b2ee-9d27a16891f7","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.851896Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:b6878a9c5694541b009809f26492346531153c98337961a62b9cbe9a3b9f3f39","observation_id":"f0a2656f-c19f-4150-8fb3-5a990c6c7b66","resolution":{"observed_at":"2026-08-08T04:34:23.927081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.907511Z","title":"Vl-calibration: Decoupled confidence calibration for large vision-language models reasoning","venue":null,"work_id":"a029f9d0-731e-4f1d-a9f2-09147a260f15","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.872825Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:614a52c657565f6eb447a9f6260591f2b627893f29ef4dd9f266b51e30484db0","observation_id":"f8d698ac-680a-42c0-9e18-136d2fc439fd","resolution":{"observed_at":"2026-08-08T04:34:23.912793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.892630Z","title":null,"venue":null,"work_id":"8476976b-bae9-437f-b527-1eebdf7812db","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.895077Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:f8d3218ef9b5b3a47dd7e867f64253bb3c2e598945098197a9d72a596d20b91a","observation_id":"1be52bab-4a6c-498e-9077-525e51081542","resolution":{"observed_at":"2026-08-08T04:34:23.897455Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1710.07300","last_updated":"2018-02-22T22:50:42Z","snapshot_observed_at":"2026-08-04T06:17:25.388464Z","submitted_at":"2017-10-19T18:01:38Z","title":"FigureQA: An Annotated Figure Dataset for Visual Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1710.07300","snapshot_observed_at":"2026-08-08T04:34:22.922646Z","title":"Figureqa: An annotated figure dataset for visual reasoning.arXiv preprint arXiv:1710.07300, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.922646Z"},"links":{"cited_paper":"/paper/1710.07300","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:926eb5039b9442908b8a1f26be7499bbaf3f8887f81accac85a01a9a56171d5f","observation_id":"be3b93ca-1489-4db5-b200-8ea065cfb549","resolution":{"observed_at":"2026-08-08T04:34:22.922646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.877136Z","title":"Plotqa: Reasoning over scientific plots","venue":null,"work_id":"c2ef4e71-d2f2-48c0-b98d-0ef7270a8b6e","year":2020},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.944732Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:0a4ae85f76a4a43a115da270db12a41b0b05b996be95c7a19264c1928fb40843","observation_id":"91a9a571-af79-410c-8cca-feaca54835d6","resolution":{"observed_at":"2026-08-08T04:34:23.882252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.959247Z","title":"Chartqa: A benchmark for question answering about charts with visual and logical reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.959247Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:d5c52b874585dd4360f00a8de3f811feff4e90b27608217c704c8f14333a4f6c","observation_id":"76933f1f-993c-48bf-9550-082e13848bbf","resolution":{"observed_at":"2026-08-08T04:34:22.959247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.850669Z","title":"Chartverse: Scaling chart reasoning via reliable programmatic synthesis from scratch","venue":null,"work_id":"ea4d00d2-5b69-418b-ac55-0680aac28ab5","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.964070Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:15c6d55c9012a8a2811147a8bf70637b5f8a3a77289241d87eff16ff7bd1d4a9","observation_id":"ef0ea0cd-0692-46ea-b2df-3cc67458edfe","resolution":{"observed_at":"2026-08-08T04:34:23.856452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.833237Z","title":"Chart-rl: Generalized chart comprehension via reinforcement learning with verifiable rewards","venue":null,"work_id":"a5b93b3b-a103-4abe-84e8-01fd32234f34","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.968703Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:e0f490346269c6d671a843598e5a25d3e53b9e3e9e5600c7a1bd4fa7652f31ac","observation_id":"02ba6739-5a20-4ecf-a225-60c48dcbd307","resolution":{"observed_at":"2026-08-08T04:34:23.839203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.973225Z","title":"Chart-rvr: Reinforce- ment learning with verifiable rewards for explainable chart reasoning.arXiv preprint arXiv:2510.10973, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.973225Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:2b4aa5da38cea1645650f0b192691b876cde8a5022f6e451392b8199cee4e153","observation_id":"49a66d6a-5a60-4c1b-a7ba-48dd6f9878e9","resolution":{"observed_at":"2026-08-08T04:34:22.973225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19492","last_updated":"2025-05-31T02:35:38Z","snapshot_observed_at":"2026-08-07T12:04:49.777765Z","submitted_at":"2025-05-31T02:35:38Z","title":"ChartGen: Scaling Chart Understanding Via Code-Guided Synthetic Chart Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.19492","snapshot_observed_at":"2026-08-08T04:34:22.977760Z","title":"Chartgen: Scaling chart understanding via code-guided synthetic chart generation.arXiv preprint arXiv:2507.19492, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.977760Z"},"links":{"cited_paper":"/paper/2507.19492","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:9e504210d4044a9e110f60740b4772de5d80cc16c51c82120044b2c514a8cba6","observation_id":"232e86df-497a-4702-9de9-47195476250b","resolution":{"observed_at":"2026-08-08T04:34:22.977760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.817445Z","title":"Unraveling the truth: Do vlms really understand charts? a deep dive into consistency and robustness","venue":null,"work_id":"927c6b21-8e5d-4cec-ad65-8781211eec34","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.982359Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:60870000269db0713ebc654d6e6e8d7231bf69c5a97f303595f6d0bd3ac0e025","observation_id":"6f94e84d-a710-4113-aa3e-03287894ce84","resolution":{"observed_at":"2026-08-08T04:34:23.822707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.987169Z","title":"Losing the plot: How vlm responses degrade on imperfect charts.arXiv preprint arXiv:2509.18425, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.987169Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:2f69081db093332b660459316f37e762531f8209dcf2a6b2df062cb45d17bb14","observation_id":"3d84fd80-551f-4539-ba7b-85f5847d8e25","resolution":{"observed_at":"2026-08-08T04:34:22.987169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12506","last_updated":"2026-05-21T07:28:16Z","snapshot_observed_at":"2026-07-06T22:45:44.135338Z","submitted_at":"2026-02-13T01:12:00Z","title":"On Robustness and Chain-of-Thought Consistency of RL-Finetuned VLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.12506","snapshot_observed_at":"2026-08-08T04:34:22.991328Z","title":"On robustness and chain-of-thought consistency of rl-finetuned vlms.arXiv preprint arXiv:2602.12506, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.991328Z"},"links":{"cited_paper":"/paper/2602.12506","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:4f3113afb1f0484b858f7776832b15b3cba0e2d2c79a7e339fb5b97ecd230adf","observation_id":"7c2d8530-723c-4b78-8c3c-fc111247c42d","resolution":{"observed_at":"2026-08-08T04:34:22.991328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.801678Z","title":"Perception-r1: Advancing multimodal reasoning capabilities of mllms via visual perception reward","venue":null,"work_id":"efe3beaf-0d7b-43f8-8761-ba658f56f2a9","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.996226Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:59f81fe6e1e2b77ccbd13e59262022c165d55b28c17b8bad3601d1608e7023cd","observation_id":"434fa5ec-450c-44ad-b825-1bf2cc705593","resolution":{"observed_at":"2026-08-08T04:34:23.806830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.000518Z","title":"V-fat: Benchmarking visual fidelity against text-bias.arXiv preprint arXiv:2601.04897, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.000518Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:257cd61f32b863ce57fc5358f2c7c163d72b5e492845392ed31c5482cddcca2c","observation_id":"ba2de16b-0e19-466b-ab7d-dfd83ea8c0aa","resolution":{"observed_at":"2026-08-08T04:34:23.000518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.004811Z","title":"Cdh-bench: A commonsense- driven hallucination benchmark for evaluating visual fidelity in vision-language models.arXiv preprint arXiv:2603.27982, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.004811Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:5a3e986f469b046ab67e37a0fcfddd63504319ae189ec85c8c396452ef083e9e","observation_id":"8078357c-3d8d-4bc3-b03a-774b03dd47e8","resolution":{"observed_at":"2026-08-08T04:34:23.004811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.10400","last_updated":"2026-06-09T04:18:38Z","snapshot_observed_at":"2026-08-06T05:17:07.543850Z","submitted_at":"2026-06-09T04:18:38Z","title":"Do Vision-Language Models See or Guess? Measuring and Reducing Textual-Prior Reliance with a Phrasing-Controlled Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.10400","snapshot_observed_at":"2026-08-08T04:34:23.009317Z","title":"Do vision-language models see or guess? measuring and reducing textual-prior reliance with a phrasing-controlled benchmark.arXiv preprint arXiv:2606.10400, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.009317Z"},"links":{"cited_paper":"/paper/2606.10400","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:c71306c8abc3cc2862a02f524f10daa35f60f2f024f2b29d9f2179f64a6249e5","observation_id":"9f4b3506-1e79-4111-a926-4c3af9b59b1d","resolution":{"observed_at":"2026-08-08T04:34:23.009317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.014499Z","title":"On the foundations of noise-free selective classification.Journal of Machine Learning Research, 11(5), 2010","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.014499Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:a094f3e75263ea5874e8364e9249aaf6107a4d4fc4188a5fa66736969e1ab813","observation_id":"01de0153-e369-4b08-9685-0c519f3fb323","resolution":{"observed_at":"2026-08-08T04:34:23.014499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.018756Z","title":"On calibration of modern neural networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.018756Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:cc000d8e8524112b9f67c461b60436c9eb13edddff1a7ea5ac5bdac1248959a0","observation_id":"b7caf36b-99b5-48eb-9345-7f336fbfabea","resolution":{"observed_at":"2026-08-08T04:34:23.018756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.767011Z","title":"selective prediction","venue":null,"work_id":"c614e192-1821-4db7-b881-76e30c9713f4","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.023142Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:1179ba38a94eb070c920c525854e56e99435776057603d7c4533e92b94ccc611","observation_id":"c00800a2-f918-4bf0-8388-7e0954a91eea","resolution":{"observed_at":"2026-08-08T04:34:23.772362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-08T04:34:23.027631Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters.arXiv preprint arXiv:2408.03314, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.027631Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:fd74ff346e74aa002125c2e1956ad354e7e7d84507ddb5a2a25eb162204ac887","observation_id":"5a86d081-bba8-4623-b336-48a798995f1e","resolution":{"observed_at":"2026-08-08T04:34:23.027631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-08T04:34:23.032333Z","title":"Qwen2.5-vl technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.032333Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:35a818b304835cab089e9a34336b81203990caeaa96305d60bc35aae113c00a5","observation_id":"dbd5e928-f152-4ede-b039-94f0d677288c","resolution":{"observed_at":"2026-08-08T04:34:23.032333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.037110Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites.Science China Information Sciences, 67(12):220101, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.037110Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:912025fec10649fe24c78dcca1d8f680e81dc2f1420f86452ab3ddd088031f62","observation_id":"287517cb-49c6-4dde-a2fe-106c27557308","resolution":{"observed_at":"2026-08-08T04:34:23.037110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.742343Z","title":"Equivalence guaranteed","venue":null,"work_id":"773cb8a7-0a1c-4e7d-aa13-df5f1c4d7254","year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.041257Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:a3d10e6f295c7d58417024e94602ea15ac5f4718449d1cf62e59b1a8a089d15e","observation_id":"07d488ce-5029-4dbc-8eae-32a09c129c48","resolution":{"observed_at":"2026-08-08T04:34:23.746762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.727149Z","title":"When the answer space has two elements and 𝐾 is odd the first inequality is an equality, and the middle quantity is strictly decreasing along odd𝐾","venue":null,"work_id":"ad3678ac-5104-43b3-a33e-a46b2975e28e","year":null},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.046022Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:de960f89e301dcd34bd6cd1bf9583546195c508ae4caf3e39945daf0cfad145b","observation_id":"82ad8d73-8ef7-4751-ab81-f49ffb356e5b","resolution":{"observed_at":"2026-08-08T04:34:23.732491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.712522Z","title":null,"venue":null,"work_id":"835e901e-d012-4a43-b040-07671125441d","year":null},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.051294Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:258729c06b466125524f4620a3bbc0389e3d7af01d47308b00942df1d2821d53","observation_id":"865fc1df-4c31-44a8-833e-280e49b81d0b","resolution":{"observed_at":"2026-08-08T04:34:23.717004Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.674895Z","title":"too many arguments","venue":null,"work_id":"975c2900-bf49-419d-8ed0-16ad5017b30c","year":null},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.056458Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:5ccff09d6d20c745eefc64aee6ffb2fe0a6aefb69e23c242f2d48b5b9efc179e","observation_id":"a9b7f48b-c950-40dd-a8a7-6ccfe160cb49","resolution":{"observed_at":"2026-08-08T04:34:23.702478Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T20:11:32.536924Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":12},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 0 inbound Pith citation observations for arXiv:2608.05670."}