{"as_of":"2026-08-14T11:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3d5241211263f6dc09e8fbc70ffb4fe638bdfc165355eb93caa4d8c2b1edfaf4","coverage":[{"denominator":32,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T15:32:59.803624Z","state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.10908/citation-record","integrity":"/paper/2412.10908/integrity","json":"/paper/2412.10908/citation-record.json","paper":"/paper/2412.10908"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.667437Z","title":"Creating such a system is one of the main goals of computer vision and can enable autonomous robots, cars, and numerous other applications[6]","venue":null,"work_id":"736e91e0-d80b-4c40-b290-2caa0363d0ce","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.617247Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:577da0bcf13a59220b5f76eab229d99f25be9afe0a2653011cf1b0847124b093","observation_id":"3034be33-4458-4393-be03-3d806614462b","resolution":{"observed_at":"2026-08-11T15:33:00.672489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.650316Z","title":"The use of CGI allows for massive amounts of object materials and environments as well as easy replacement of object materials","venue":null,"work_id":"d8881316-bc47-432d-9a8a-f82370856c45","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.623477Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:4ad9767f11c8838dfd4b1c6541c11369be06c9c616ca9abf5b204e4b4f914f83","observation_id":"0a98dca0-2506-4489-b256-8723388d0a50","resolution":{"observed_at":"2026-08-11T15:33:00.655761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.631764Z","title":null,"venue":null,"work_id":"9ef32c55-d9e5-4264-bf41-202b4ab34bba","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.629201Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:861d7bd9a37723fc4154ba434ce038b4732744fc4dbd79e071836a8c9173c02d","observation_id":"dc14ceb1-d4df-4d9f-8fd7-075eb122aa50","resolution":{"observed_at":"2026-08-11T15:33:00.637457Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.611730Z","title":null,"venue":null,"work_id":"0071539f-c23d-463d-bdd5-bac578f8a510","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.635538Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:f0118cf2285595590b26840130061f0af5402714acb0e556b98acd7ea9a54580","observation_id":"a36ba2db-b468-4525-b700-c15386f9024f","resolution":{"observed_at":"2026-08-11T15:33:00.617632Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.593131Z","title":null,"venue":null,"work_id":"9a14aab6-7cb8-405f-960f-c2fb06431eef","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.640895Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:484e2846c9f4d7a2dd9d62da013c9dfe2676e3719abf8501dff1f9e30b153f92","observation_id":"501796c8-e85a-4d6b-9aea-6ed22d964a8b","resolution":{"observed_at":"2026-08-11T15:33:00.598836Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.573267Z","title":"Note that tests 2,4,6 force the model to rely only on the 3D shape for matching, while tests 1,3,5,7 allow the model to use the color/texture or the 2D projection for recognition","venue":null,"work_id":"6f6e6d90-8280-41dd-ba51-3066b334c054","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.646636Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:853a0c776e378e5ed270dd736a71fc56b76b53ac0f6d795daf4e3dff5bf27ddb","observation_id":"e686b0bb-aee8-43e1-9337-462d38ce1b52","resolution":{"observed_at":"2026-08-11T15:33:00.579705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.556411Z","title":"Which of the panels contains an object with an identical 3D shape to the object in panel A. Your answer must come as a single letter","venue":null,"work_id":"10a02eec-e90a-467e-b404-810266eef263","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.653051Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:19a293d001bdfa71183287c97396654b03e379d79aaea60f015d0e3460e3938a","observation_id":"abc6e0d9-52d2-4f76-9914-571fe80d2593","resolution":{"observed_at":"2026-08-11T15:33:00.562114Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.538597Z","title":"2) Has a different orientation compared to the object in panel A","venue":null,"work_id":"ab134c99-4539-4ef7-8d49-cdbf615a089d","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.658106Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:01ae7e7d9a54ece6363817f0115012e7e655c663d8a5af7b918977dce23b496c","observation_id":"6368588a-5d7d-42a8-9a86-55ea9f8c417d","resolution":{"observed_at":"2026-08-11T15:33:00.544037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.518548Z","title":"Gemini and GPT 4o clearly excel in this task but all models show some level of understanding (Table 1)","venue":null,"work_id":"5082ef05-27af-4681-b7c2-fa78b3dab87b","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.663124Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:e40c9cfdf06af288ad72d67de4efc1fae7141b848209392079b6cecd5b893346","observation_id":"680de0eb-c39b-4424-bf54-467d886c0059","resolution":{"observed_at":"2026-08-11T15:33:00.525021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.497338Z","title":"These results are consistent with previous works which show that despite their impressive performance Vision Language Models (VLM) often miss basic aspects of reality[8-12]","venue":null,"work_id":"351d7bda-b043-453c-9292-1f51a2408cd7","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.668844Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:d32768363e71c28b604e6c8a34a2582cbe73e45c47a9ec5eccdbda65f2c8e7e6","observation_id":"8a5c8630-6636-4aac-a126-99e5bc22f2eb","resolution":{"observed_at":"2026-08-11T15:33:00.503699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.471310Z","title":"Code used to evaluate the models on the benchmark available at this URL","venue":null,"work_id":"e9f23a40-f4d4-45bf-b002-f3dfe21e4251","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.674052Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:e8fb07df24587377d22875ad218b9fd44558d474e70f29b69820b10a7153a035","observation_id":"2e6bfac0-3978-4ddf-a952-d83db8758918","resolution":{"observed_at":"2026-08-11T15:33:00.477553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.452890Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"730f1984-bfb4-469d-a6eb-d726df907dcc","year":2022},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.679122Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:5ddfb0953048bc9624e365a6c83fb7795c9fe164bdea18edccaa3d019410424b","observation_id":"7c8821e9-3f3f-4f04-8fd5-2cb6bb35516d","resolution":{"observed_at":"2026-08-11T15:33:00.458914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T15:32:59.684975Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.684975Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:20712b5dc53c12748afc34759c9b011a05fda2aca1e7ff76742dc44f265af7e9","observation_id":"50ea74b2-1b65-4e03-8042-7766b9cbe2cd","resolution":{"observed_at":"2026-08-11T15:32:59.684975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.434671Z","title":"Real-world robot applications of foundation models: A review","venue":null,"work_id":"adae7026-cad5-49d8-815a-1976c1694792","year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.692018Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:9277e276b8a73520069e1dedaa75b38d4bd3da3cc17eddbd6f5f4ce0e4f5c072","observation_id":"a9899b30-3309-445f-8474-9cff7f7ab3f7","resolution":{"observed_at":"2026-08-11T15:33:00.441207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11300","last_updated":"2023-09-07T21:18:15Z","snapshot_observed_at":"2026-08-13T10:29:52.575350Z","submitted_at":"2023-08-22T09:29:05Z","title":"Approaching human 3D shape perception with neurally mappable models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11300","snapshot_observed_at":"2026-08-11T15:32:59.700235Z","title":"Approaching human 3D shape perception with neurally mappable models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.700235Z"},"links":{"cited_paper":"/paper/2308.11300","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:cafcb8b6e0844ea2024c827cc62b2a43c95bc9a7b0881d6d4f3600fcffff11bc","observation_id":"0ab174d4-2dda-4e41-8d9f-03f3439000f4","resolution":{"observed_at":"2026-08-11T15:32:59.700235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.415659Z","title":"Lvlm-ehub: A comprehensive evaluation benchmark for large vision-language models","venue":null,"work_id":"c43043fa-a4d1-437a-bea8-4501c38ff16a","year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.706021Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:e477e843188671a8614695b259ced29b479d259cd584c2f647422386862dc522","observation_id":"c95c0070-cf76-4ea2-a227-819545c5a26f","resolution":{"observed_at":"2026-08-11T15:33:00.421219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00947","last_updated":"2025-07-13T22:29:38Z","snapshot_observed_at":"2026-08-12T04:48:19.441994Z","submitted_at":"2024-12-01T19:46:22Z","title":"VisOnlyQA: Large Vision Language Models Still Struggle with Visual Perception of Geometric Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00947","snapshot_observed_at":"2026-08-11T15:32:59.711784Z","title":"VisOnlyQA: Large Vision Language Models Still Struggle with Visual Perception of Geometric Information","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.711784Z"},"links":{"cited_paper":"/paper/2412.00947","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:f7a752e2dd2ecf98a0eb1f9ee2e405ba4f0e19f9366fbc538ded9308445fb6a8","observation_id":"b51dd177-3386-4663-a34b-04f75132c30f","resolution":{"observed_at":"2026-08-11T15:32:59.711784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09193","last_updated":"2025-03-05T19:01:00Z","snapshot_observed_at":"2026-08-13T00:54:31.014752Z","submitted_at":"2024-03-14T09:07:14Z","title":"Can We Talk Models Into Seeing the World Differently?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.09193","snapshot_observed_at":"2026-08-11T15:32:59.717893Z","title":"Are Vision Language Models Texture or Shape Biased and Can We Steer Them?","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.717893Z"},"links":{"cited_paper":"/paper/2403.09193","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:2c136ffee268f57f124fce75b31695982c4627d407d945f31172646c1b4a9ecd","observation_id":"19fba42e-ae53-433a-832d-7bb50df268e8","resolution":{"observed_at":"2026-08-11T15:32:59.717893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:32:59.723920Z","title":"Exploring the frontier of vision-language models: A survey of current methodologies and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.723920Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:c897debd197179b2e979284ccf77c167945dee930ec1b04ef8b1b448e51ac379","observation_id":"9b2d850c-06bd-4181-be6d-f1398fa68724","resolution":{"observed_at":"2026-08-11T15:32:59.723920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12390","last_updated":"2024-07-03T08:44:45Z","snapshot_observed_at":"2026-08-14T00:50:08.075021Z","submitted_at":"2024-04-18T17:59:54Z","title":"BLINK: Multimodal Large Language Models Can See but Not Perceive","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12390","snapshot_observed_at":"2026-08-11T15:32:59.729389Z","title":"Blink: Multimodal large language models can see but not perceive","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.729389Z"},"links":{"cited_paper":"/paper/2404.12390","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:2ea61690b4c3639408fcf877d35f3c9dbd0ff6af0326819b8d2600f93b4d05ef","observation_id":"b98382ba-30bc-42b8-915d-d86fbd198ffc","resolution":{"observed_at":"2026-08-11T15:32:59.729389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.14760","last_updated":"2024-07-03T05:21:38Z","snapshot_observed_at":"2026-08-13T00:48:24.698830Z","submitted_at":"2024-03-21T18:02:20Z","title":"Can 3D Vision-Language Models Truly Understand Natural Language?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.14760","snapshot_observed_at":"2026-08-11T15:32:59.735368Z","title":"Can 3D Vision-Language Models Truly Understand Natural Language?","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.735368Z"},"links":{"cited_paper":"/paper/2403.14760","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:34790685e20b9c443743b95cb2033ed070aef0fa9c010c181bf6917f3bea58ef","observation_id":"89992980-f643-4bcd-9190-ff61916e22ca","resolution":{"observed_at":"2026-08-11T15:32:59.735368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:32:59.741436Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.741436Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:ea41660a339ca6a73dc5de280ba9491e9c7a6f4dc697585ed59af06bfd62981a","observation_id":"7767142a-5c19-4f6a-a0aa-7a0d04da4b3c","resolution":{"observed_at":"2026-08-11T15:32:59.741436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15732","last_updated":"2024-03-12T01:07:14Z","snapshot_observed_at":"2026-08-13T05:17:09.204464Z","submitted_at":"2023-11-27T11:29:10Z","title":"GPT4Vis: What Can GPT-4 Do for Zero-shot Visual Recognition?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15732","snapshot_observed_at":"2026-08-11T15:32:59.747061Z","title":"GPT4Vis: what can GPT-4 do for zero-shot visual recognition?","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.747061Z"},"links":{"cited_paper":"/paper/2311.15732","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:7663b89b5c0d8161a7218a2d6a41cbb57201942bd65caeaf0783f71962df1ef1","observation_id":"7ed68ce4-b4cd-4062-aa72-90accc550ae2","resolution":{"observed_at":"2026-08-11T15:32:59.747061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.376553Z","title":"A general protocol to probe large vision models for 3d physical understanding","venue":null,"work_id":"30c9f350-4b54-4456-b90f-a9d68590c1a5","year":2023},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.752641Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:037bff4b8a88e85d656f4888782600e4050f8448d588e26917af7cdecb74a1ab","observation_id":"ec3a7477-e733-413f-8262-f7e521146d38","resolution":{"observed_at":"2026-08-11T15:33:00.384059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00253","last_updated":"2024-05-06T01:10:01Z","snapshot_observed_at":"2026-08-13T12:48:26.884823Z","submitted_at":"2024-02-01T00:33:21Z","title":"A Survey on Hallucination in Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00253","snapshot_observed_at":"2026-08-11T15:32:59.757593Z","title":"A survey on hallucination in large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.757593Z"},"links":{"cited_paper":"/paper/2402.00253","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:2c8d54bc99dd1aa51b706ce6ed0bf58308ce6f43d70294b84df3d275f2944927","observation_id":"f96fe009-55fb-4756-a18d-1f54a1fb6376","resolution":{"observed_at":"2026-08-11T15:32:59.757593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:32:59.764351Z","title":"When LLMs step into the 3D World: A Survey and Meta-Analysis of 3D Tasks via Multi-modal Large Language Models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.764351Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:850bd55b7953f259a70f7d3f2531f6d5754eedcffe8b09455624928cd9fa936d","observation_id":"4765c8e0-0de6-46d9-9039-00709859032f","resolution":{"observed_at":"2026-08-11T15:32:59.764351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03685","last_updated":"2024-05-06T17:57:27Z","snapshot_observed_at":"2026-08-13T00:13:45.105761Z","submitted_at":"2024-05-06T17:57:27Z","title":"Language-Image Models with 3D Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03685","snapshot_observed_at":"2026-08-11T15:32:59.771499Z","title":"Language-Image Models with 3D Understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.771499Z"},"links":{"cited_paper":"/paper/2405.03685","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:8b94fcd5a1d9565e9827758fad3380e0f1be8c1bd509c7f37e85f5775614ff7a","observation_id":"add2d72e-3b04-461f-bfcc-1d9e03910ff2","resolution":{"observed_at":"2026-08-11T15:32:59.771499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.357189Z","title":"Shapellm: Universal 3d object understanding for embodied interaction","venue":null,"work_id":"1fbbff72-5e29-4eaf-b7d9-d05b92b3a74e","year":2025},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.778276Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:28f51f1ecf00a5421ced7f2431a313789da548386742390591fef0cf6b13ae29","observation_id":"56fee059-30fe-4cdb-86e0-03dacf78109a","resolution":{"observed_at":"2026-08-11T15:33:00.362487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.338837Z","title":"Objaverse: A universe of annotated 3d objects","venue":null,"work_id":"1cdaf962-2bc6-451c-a7f3-3907acb9c296","year":2023},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.786119Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:5314afd4fcfeffd2a7c5e91a6fa550024ca8898943e8b24b58969395555d2d00","observation_id":"e5721b71-513f-4a10-bdab-8aa799b477dc","resolution":{"observed_at":"2026-08-11T15:33:00.345067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17146","last_updated":"2025-06-11T22:33:31Z","snapshot_observed_at":"2026-08-12T23:35:43.291209Z","submitted_at":"2024-06-24T21:36:01Z","title":"Vastextures: Vast repository of textures and PBR materials extracted from real-world images using unsupervised methods","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17146","snapshot_observed_at":"2026-08-11T15:32:59.792476Z","title":"Vastextures: Vast repository of textures and PBR materials extracted from real-world images using unsupervised methods","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.792476Z"},"links":{"cited_paper":"/paper/2406.17146","citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:1f32076f29c6be788aa76ee69877c6b142e0ab7fc4161f8cf56d13e7f1347569","observation_id":"2af7571c-6ed6-4cee-82fc-5a8b44b5bc94","resolution":{"observed_at":"2026-08-11T15:32:59.792476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.319626Z","title":"Infusing Synthetic Data with Real-World Patterns for Zero-Shot Material State Segmentation","venue":null,"work_id":"f87ffa1b-8287-482f-9af0-5611f507820d","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.798061Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:30fe9091e4b21fa8d4deb5f10f13d18dad803fb6d2ada478ffa9b19cc6250834","observation_id":"bf7155e3-a1fc-48f5-ab46-ed90f98c84ec","resolution":{"observed_at":"2026-08-11T15:33:00.325661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:00.299395Z","title":"Which panel contains an object that has identical 3D shape to the object in panel A, but different in orientation and texture. Explain","venue":null,"work_id":"200f6931-6375-4c99-ad68-8a28f9d9500b","year":null},"citing_paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T15:32:59.803624Z"},"links":{"citing_paper":"/paper/2412.10908"},"observation_digest":"sha256:fc12d62a65ba49bb7b80d843ab95ca894377792e5750a222a02fc5bf6d6e1115","observation_id":"cc5021e5-b612-4a76-af9f-6b71d0dfdd6a","resolution":{"observed_at":"2026-08-11T15:33:00.307386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.10908","last_updated":"2025-07-22T16:46:37Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T02:46:43.802308Z","submitted_at":"2024-12-14T17:35:27Z","title":"Do large language vision models understand 3D shapes?"},"reference_resolution":{"displayed":32,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":16},"total_outbound_references":32},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 32 of 32 outbound references and 0 inbound Pith citation observations for arXiv:2412.10908."}