{"as_of":"2026-08-08T11:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:68c5d71586a82c051aaec192ea416bf8e8de4895e302d21ba0a9f9a23254dab5","coverage":[{"denominator":19,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T12:21:22.544303Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T16:32:28.422930Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T16:32:28.671394Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"cited_work":{"arxiv_id":"2507.21871","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.21871","snapshot_observed_at":"2026-08-05T16:32:28.671394Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","venue":"q-bio.NC","work_id":"0a5a4d65-64ae-436b-ae1b-2d5d2e15300e","year":2025},"citing_paper":{"arxiv_id":"2508.18226","last_updated":"2025-08-25T17:23:27Z","snapshot_observed_at":"2026-08-06T11:02:51.926446Z","submitted_at":"2025-08-25T17:23:27Z","title":"Disentangling the Factors of Convergence between Brains and Computer Vision Models","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T16:32:28.422930Z"},"links":{"cited_paper":"/paper/2507.21871","citing_paper":"/paper/2508.18226"},"observation_digest":"sha256:f9838761c36c23dfd3e49d8d5f4d85e6f6f575408ffec83a48625dbb97f7eb67","observation_id":"e30cad62-9bbc-4c50-a7ce-b698b07ceb1e","resolution":{"observed_at":"2026-08-05T16:32:28.674050Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.21871/citation-record","integrity":"/paper/2507.21871/integrity","json":"/paper/2507.21871/citation-record.json","paper":"/paper/2507.21871"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.774454Z","title":"Emerging evidence suggests that human brain representations in both vision and language are well predicted by semantic feature spaces obtained from large language models (LLMs)","venue":null,"work_id":"9159cc79-77f7-433c-82b9-2c7687120cea","year":2024},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.476705Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:64d48693ebd2d8c8ca88bad54c50a58b758bbf4a47daae8427549a3a60377eee","observation_id":"9bdf7fb3-8a6f-4f18-8ba3-ce6f87e63edf","resolution":{"observed_at":"2026-08-06T12:21:22.778065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.723501Z","title":"(A) Cross-validated non-negative least squares regression was used to model the brain RDMs at every searchlight location using the behavioural RDMs derived from our MA tasks","venue":null,"work_id":"c2d5a347-8003-46d6-a6c2-7e2e8d224618","year":2024},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.496414Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:2ade81a05440d3069abf0fe36a9ba6f7f2457f4dd45d7408c2d01c7e62756064","observation_id":"808fd061-f75c-4eae-8e7f-de22674eed3a","resolution":{"observed_at":"2026-08-06T12:21:22.727176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.713456Z","title":null,"venue":null,"work_id":"4adf6e94-589a-49f2-b8de-ce03ffd1add6","year":1970},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.500243Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:3bf5a8411710f62d366826e00bd14f364ffb48e16ec247a76df03432b3b6e196","observation_id":"5bbd874c-07bd-406d-97ab-218d9a5d3fd2","resolution":{"observed_at":"2026-08-06T12:21:22.716796Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.744596Z","title":"linguistic modality","venue":null,"work_id":"3f26c411-67f4-4494-a8ed-d7be6d7cc679","year":2012},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.488582Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:1eafd0f721d81dc36d4eba770d767e6448fc6c531bac322dc835e695b09f2f10","observation_id":"09891198-0646-410c-acc0-115c75458ddb","resolution":{"observed_at":"2026-08-06T12:21:22.748159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.734274Z","title":"(A) Participants completed the MA task either on 100 natural scene images (visual modality left) or 100 sentence captions 8 describing the images (linguistic modality right)","venue":null,"work_id":"1ed05f7d-b7ef-48f9-8515-37bd2317ec68","year":2022},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.492359Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:7b44570e9e298b352be674458a13d82295a1f7e1fabd40c7f7f5f61d64b378e5","observation_id":"2a6fb438-672d-4b6a-b27c-01a0d165e43e","resolution":{"observed_at":"2026-08-06T12:21:22.738082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.01293","last_updated":"2021-02-02T04:07:38Z","snapshot_observed_at":"2026-08-01T22:46:19.170916Z","submitted_at":"2021-02-02T04:07:38Z","title":"Scaling Laws for Transfer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.01293","snapshot_observed_at":"2026-08-06T12:21:22.532478Z","title":"A., Feder, A., Emanuel, D., Cohen, A., Jansen, A., Gazula, H., Choe, G., Rao, A., Kim, S","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.532478Z"},"links":{"cited_paper":"/paper/2102.01293","citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:7c9639b848be43808cb43b274f25d5fc4757f3893f8f763eea105c8b157bf7e2","observation_id":"6cb16c37-3857-4953-adea-9511c086b9a4","resolution":{"observed_at":"2026-08-06T12:21:22.532478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.04135","last_updated":"2023-10-30T01:56:14Z","snapshot_observed_at":"2026-07-06T14:02:40.989915Z","submitted_at":"2022-10-09T01:49:58Z","title":"VoLTA: Vision-Language Transformer with Weakly-Supervised Local-Feature Alignment","version":3},"cited_work":{"arxiv_id":"2210.04135","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.04135","snapshot_observed_at":"2026-08-06T12:21:22.574470Z","title":"VoLTA: Vision-Language Transformer with Weakly-Supervised Local-Feature Alignment","venue":"cs.CV","work_id":"03203508-9a91-4fe4-b9be-ca7d0049a34e","year":2022},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.544303Z"},"links":{"cited_paper":"/paper/2210.04135","citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:cf580803d825903912f65354b741f8a1e8b74bdb9f19dae2fca55e69887b77d2","observation_id":"698df9c9-edc8-4cb8-b09b-0509435d986b","resolution":{"observed_at":"2026-08-06T12:21:22.581138Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1101/2022.03.28.485868.abstract","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.631535Z","title":"A., Schmitz, T","venue":null,"work_id":"21370512-c7c8-49e7-aca2-ede847726cfa","year":2014},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":134,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.525364Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:4a3a92aba53dda942b8d685a9fa075f464e6fa062850d794d11d2057dddf7427","observation_id":"678bc90a-5ea0-4743-acde-4b632dd97752","resolution":{"observed_at":"2026-08-06T12:21:22.636069Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.620674Z","title":"A., Kiani, R., Bodurka, J., Esteky, H., Tanaka, K., & Bandettini, P","venue":null,"work_id":"cd7c4f31-8beb-4514-b1c2-7d4cf24bd9f6","year":2008},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":245,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.537199Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:388dccff8db6372793bb0713e5bd059ef6284825434099e9fb78bc71dcd2b23d","observation_id":"24dcb9c7-e65c-45d3-be8e-6ace6e4c0c1b","resolution":{"observed_at":"2026-08-06T12:21:22.624190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.755295Z","title":"visual modality","venue":null,"work_id":"18c4016c-fc13-4e2e-9865-94d360c97763","year":2022},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2012,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.485065Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:5959ffc6f1ec78643eb5af97b58ac821fa16a508613b92eb09e6952d23ada731","observation_id":"0f02e926-5b21-49b1-aaf4-c65258533fd1","resolution":{"observed_at":"2026-08-06T12:21:22.758598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.702999Z","title":"We show that a similar relational structure emerges for both linguistic and visual inputs","venue":null,"work_id":"46ed3e69-8bc4-49a6-8c05-3ff89216096c","year":2024},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.503950Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:9762900a21da2be9e5324b3ca67b6eabff586677de9e427ac70d612539f6af14","observation_id":"c8bfdb37-929c-4e63-94b7-f280d91f9d21","resolution":{"observed_at":"2026-08-06T12:21:22.706659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.655083Z","title":"The significance of correlations was tested using one-sided t-test across participants and corrected for multiple comparisons at FDR p < 0.05","venue":null,"work_id":"43a2df10-6f75-4a7f-9de3-3bd58aefd974","year":2022},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.518202Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:6cace7cafeb697232d137ca8376a988d733143fd9c4858bf42ada8a653e284d9","observation_id":"5545cf45-c302-4c49-ac71-85dcceb97a61","resolution":{"observed_at":"2026-08-06T12:21:22.659171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.679021Z","title":null,"venue":null,"work_id":"020484a2-bee0-4427-bd52-710127f62c17","year":2011},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.510706Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:e8c7e19af7fe824fb0af0920c20d3260badbb1ed9faf115e4dd094f8cfba9cd4","observation_id":"b50d41b6-b50e-4c2d-bff9-78862206375c","resolution":{"observed_at":"2026-08-06T12:21:22.682747Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.691579Z","title":"It may be that the visual system translates sensory inputs into modality-agnostic representations that reflect stable, relational patterns observed in the real-world environment","venue":null,"work_id":"8edb606e-1fbd-481f-94d4-909de7a96630","year":2024},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.507342Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:d1205ccd4ba49aeae788f305b4901e8662cfc5551790c472b1e3562a45a630d7","observation_id":"044c86dc-3609-493a-a98d-061e1f87a0a4","resolution":{"observed_at":"2026-08-06T12:21:22.695493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.667146Z","title":"The sentence captions were collected from five human annotators as part of the Microsoft Common Objects in Context database (Lin et al., 2014)","venue":null,"work_id":"92b62b5f-a3df-4a3c-8330-8665ae81bdf1","year":2012},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.514391Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:7903e94dd575ae16591380f81a9ca2af33687f3dcda5c0a5aec16859a581d780","observation_id":"c1b187da-df2f-4334-a4a4-b53ba3204fdc","resolution":{"observed_at":"2026-08-06T12:21:22.671282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.764740Z","title":null,"venue":null,"work_id":"860b7574-7d03-455e-9805-d81946a6176b","year":2024},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.481066Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:23ffa5530938a70d081d667e8e55b63bbef21758a0b88dfffaec333db5f12a8a","observation_id":"8746e166-da76-43e8-851d-94d81dc74e1e","resolution":{"observed_at":"2026-08-06T12:21:22.767941Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T12:21:22.643902Z","title":null,"venue":null,"work_id":"c87a8cd8-c611-4808-80d8-122413c91e4b","year":2022},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":4081,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.522047Z"},"links":{"citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:3fb2e13b80c5c9b3a2553a731f4af09fc48b74770ffade2757c335190175a9c4","observation_id":"be82ce1a-0acd-48f6-9705-f40062c5cb0f","resolution":{"observed_at":"2026-08-06T12:21:22.647477Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.02265","last_updated":"2019-08-06T17:33:52Z","snapshot_observed_at":"2026-07-06T08:12:46.398794Z","submitted_at":"2019-08-06T17:33:52Z","title":"ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.02265","snapshot_observed_at":"2026-08-06T12:21:22.540405Z","title":"T., Pan, B., Jin, S","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":6241,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.540405Z"},"links":{"cited_paper":"/paper/1908.02265","citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:aa1f5d0441faf6dd38cb41a4633b2ea55e565194c7094da35d40a2a07554fea4","observation_id":"233de5ba-3820-4e13-b61c-6c4c11479083","resolution":{"observed_at":"2026-08-06T12:21:22.540405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11737","last_updated":"2024-07-06T05:26:33Z","snapshot_observed_at":"2026-07-06T13:55:41.552052Z","submitted_at":"2022-09-23T17:34:33Z","title":"Visual representations in the human brain are aligned with large language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11737","snapshot_observed_at":"2026-08-06T12:21:22.528735Z","title":"C., Janarthanan, S., Culham, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities","version":1},"reference_index":9383,"source":"pdf_text","source_observed_at":"2026-08-06T12:21:22.528735Z"},"links":{"cited_paper":"/paper/2209.11737","citing_paper":"/paper/2507.21871"},"observation_digest":"sha256:817fb847b192eecef073ac28f305cca8c69e66e5742ad1241980bcce0b8906dd","observation_id":"2112a0ed-47c6-4cc6-acf6-2f903dea97d9","resolution":{"observed_at":"2026-08-06T12:21:22.528735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.21871","last_updated":"2025-07-29T14:42:31Z","latest_version":1,"primary_category":"q-bio.NC","snapshot_observed_at":"2026-08-06T12:21:21.420482Z","submitted_at":"2025-07-29T14:42:31Z","title":"Representations in vision and language converge in a shared, multidimensional space of perceived similarities"},"reference_resolution":{"displayed":19,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":2,"verified_fuzzy":10},"total_outbound_references":19},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 19 of 19 outbound references and 1 inbound Pith citation observation for arXiv:2507.21871."}