{"as_of":"2026-08-13T18:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:09f939947b4c52aa3c778974eb4d9687cba90415eb030aa864efc64719c2928a","coverage":[{"denominator":53,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T20:21:26.879903Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.08689/citation-record","integrity":"/paper/2509.08689/integrity","json":"/paper/2509.08689/citation-record.json","paper":"/paper/2509.08689"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.563152Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.563152Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:171d27d4d76e69c32aad7fbe84d8ed371c4752f40f67ec1f1e1e5aa1688bb657","observation_id":"74659d58-e8b2-4b41-8620-eb28ef27e8db","resolution":{"observed_at":"2026-08-04T20:21:26.563152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.567362Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.567362Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:ab7ae06d4636d83242465c885ecb218d0bc0868572be76c1a4daafa57bf75d55","observation_id":"8f5db441-1eff-41f2-ab85-d2261aad1738","resolution":{"observed_at":"2026-08-04T20:21:26.567362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.570976Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.570976Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:858c64814724414f3161f1daa1c83c8b824011bbef100f99fc7f2409b9868897","observation_id":"e7a62aea-352b-45eb-ad85-342975d3218e","resolution":{"observed_at":"2026-08-04T20:21:26.570976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.574812Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.574812Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:6a231a3704dad50774f07ebd8e899109d581bcea4a259fd3f856f069a40f724e","observation_id":"de69edd3-43cf-4908-b29d-b9e94012fe50","resolution":{"observed_at":"2026-08-04T20:21:26.574812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.578426Z","title":"Put-That-There","venue":null,"work_id":null,"year":1980},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.578426Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:38f5f1ca9febad0b2d43ab66fa8d684efc2f640efbc8d2b0a1b2a84d6d4161d6","observation_id":"c19d2bd4-6351-4f5b-a2ee-77e7b43df89c","resolution":{"observed_at":"2026-08-04T20:21:26.578426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3555","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:24:00.785233Z","title":null,"venue":null,"work_id":"deee13aa-f483-448d-9121-b6be1e59fd7a","year":2022},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.582082Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:1f7a97ca790ccefdd2309130a61a2d458ee99660677978b566e44d85a99591f5","observation_id":"810a6f13-d55a-477b-bcfc-7be1cd80336c","resolution":{"observed_at":"2026-08-04T20:24:00.828897Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.590199Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.590199Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:756ae4500baf20717fd027664994eb1fb2a25d3aac4fef321fffa77a7d3e553e","observation_id":"3ae7aad1-6a97-4556-aaca-9438cf4905dd","resolution":{"observed_at":"2026-08-04T20:21:26.590199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.593581Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.593581Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:b2cc864bc9668e0c8e7ff4f817a24694994a4071b785ccf35cb402359bd16e7c","observation_id":"96a68bf1-4ed8-46b4-a6c0-f70737a51613","resolution":{"observed_at":"2026-08-04T20:21:26.593581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.596913Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.596913Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:052428aa0c897b5aec77b2931849e3a387831c7d36c123e0a95f2ce566f8bfd7","observation_id":"a4a03b7d-f1e0-4493-b693-bb361ea911e3","resolution":{"observed_at":"2026-08-04T20:21:26.596913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.600889Z","title":null,"venue":null,"work_id":null,"year":1998},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.600889Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:6be7fe1fbc2f6cdcb4bd4a07cb381d95006d36ef3d169e1948651457bb099903","observation_id":"f7f0d052-db5d-4b2b-84b9-6fd5eefe1da4","resolution":{"observed_at":"2026-08-04T20:21:26.600889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.12981","last_updated":"2023-07-24T17:59:02Z","snapshot_observed_at":"2026-08-13T10:48:53.214889Z","submitted_at":"2023-07-24T17:59:02Z","title":"3D-LLM: Injecting the 3D World into Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.12981","snapshot_observed_at":"2026-08-04T20:21:26.604504Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.604504Z"},"links":{"cited_paper":"/paper/2307.12981","citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:90c4d4a27cd5f65e94edb48c8e1ae77e212d9ad504da8d71f68b4d2aa87bf3a6","observation_id":"87c4c263-2497-465c-bdca-c309f4b1596e","resolution":{"observed_at":"2026-08-04T20:21:26.604504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.608456Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.608456Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:a118b69f321a53cfd4739d931704f898d3ff7f1e8cb19f19748944b4d1c1c4c2","observation_id":"0cac9dd5-709e-468f-aab4-33fba9aa60b0","resolution":{"observed_at":"2026-08-04T20:21:26.608456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.615084Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.615084Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:9a1097d12efcf4a7ee39252ebeba03402e80cd26c713302ea7f30a4ece9bc57f","observation_id":"1b40e1ec-d8f7-476e-81de-212eb2dcbcaf","resolution":{"observed_at":"2026-08-04T20:21:26.615084Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.618401Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.618401Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:bd625e9c86db2ec3e30ef113fe464d592f2dfece0dbe33e47c573da5b8f6cc0a","observation_id":"35677031-b9af-4d18-8546-b7e186ecfad6","resolution":{"observed_at":"2026-08-04T20:21:26.618401Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.621654Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.621654Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:3670c2432a3c392216df4e08184e6c19cb671d92927f99c2403304cf013e0d7c","observation_id":"e9ad11dc-9992-4e4e-a6ba-71480624158d","resolution":{"observed_at":"2026-08-04T20:21:26.621654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.625127Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.625127Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:abd7bc10af8c30ebcbcb083b76abc13c3fe4e77e61e44e64b969e5bdffe6add0","observation_id":"76b75d40-ef1c-449b-946f-bc73aa9c9207","resolution":{"observed_at":"2026-08-04T20:21:26.625127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.632628Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.632628Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:90ed0f9ee3bb3ec38196930fd9f4544e9baaf4f883d6ab9a21e2970d244256c5","observation_id":"862b9596-6bdf-49aa-95ad-bf69254b9173","resolution":{"observed_at":"2026-08-04T20:21:26.632628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08667","last_updated":"2021-10-20T23:42:35Z","snapshot_observed_at":"2026-08-12T20:11:49.694712Z","submitted_at":"2021-04-18T00:14:29Z","title":"SIMMC 2.0: A Task-oriented Dialog Dataset for Immersive Multimodal Conversations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08667","snapshot_observed_at":"2026-08-04T20:21:26.628534Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.628534Z"},"links":{"cited_paper":"/paper/2104.08667","citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:d3ffdeee0c25ea7b4cfe5c77c9fedc5f5ff32c8c3fc672e166d4c8bcbe2637a8","observation_id":"4fdc81d4-c1fd-423c-b292-d25293f39212","resolution":{"observed_at":"2026-08-04T20:21:26.628534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.640597Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.640597Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:06c52f866d50b66b2b6bc6cfaea65272a1d9c4f110f2b9851ed284eae542ba89","observation_id":"a422f4f3-63bf-42f9-bb81-4eabc0cb1e36","resolution":{"observed_at":"2026-08-04T20:21:26.640597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.636874Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.636874Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:7e0f0a22eda6d5b9706261a144aaa4ebfc7b19492359c7f1fca79204ee0133cb","observation_id":"ede375e8-6ce6-4c3d-b30a-551a4722a173","resolution":{"observed_at":"2026-08-04T20:21:26.636874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10535","last_updated":"2023-06-22T01:37:02Z","snapshot_observed_at":"2026-08-13T13:17:52.031804Z","submitted_at":"2022-12-20T18:46:16Z","title":"A Survey of Deep Learning for Mathematical Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10535","snapshot_observed_at":"2026-08-04T20:21:26.646950Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.646950Z"},"links":{"cited_paper":"/paper/2212.10535","citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:5faf799b678e795a9cfbf3b44923f836352615c38116fad5e112e0afc78c835e","observation_id":"4f2d73f5-1375-4dde-92bc-f3bc39fa2951","resolution":{"observed_at":"2026-08-04T20:21:26.646950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.643752Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.643752Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:76624c06aa106ac0a8e5242d756a71b3c914afe1250396f56bcdf826782990de","observation_id":"5cba37ea-65f0-41fb-8327-88746b27b0b4","resolution":{"observed_at":"2026-08-04T20:21:26.643752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.654273Z","title":"Manning, Mihai Surdeanu, John Bauer, Jenny Finkel, Steven J","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.654273Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:ead2a9965b2023597630413150230e0c9a81b1a9948d2b2b9ac5094087d8cb9c","observation_id":"7936566c-0fea-4dfb-80fb-934001afcefc","resolution":{"observed_at":"2026-08-04T20:21:26.654273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.650376Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.650376Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:69c1626ef70565cba2f2afa15e84c6f720299c58fbb05965903988f352eb9e86","observation_id":"92173f3f-4039-49a2-9950-bdfb89ef5825","resolution":{"observed_at":"2026-08-04T20:21:26.650376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.790777Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.790777Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:be1f0704753a89c0bf7cf2036aec6180c4ccbc031815440f743e181a19e4770e","observation_id":"b1a2c799-8eb9-4505-9955-21b91fa8abac","resolution":{"observed_at":"2026-08-04T20:21:26.790777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.658138Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.658138Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:65f1c73f21362b0f675e441a50e938385345a6915d7444991c1c0259fe6a0d4b","observation_id":"c72c653e-7eef-4dc7-b01f-d4303ead1c31","resolution":{"observed_at":"2026-08-04T20:21:26.658138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.797902Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.797902Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:90a678f7d34aab14dc50bfb6be48b59c29eda1129274edbe11b4875897b2a16c","observation_id":"c6d3d9da-894a-427c-b671-c6b9ee2028cb","resolution":{"observed_at":"2026-08-04T20:21:26.797902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.794373Z","title":"Scott MacKenzie","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.794373Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:f79cb2abaefb96b5f3adc5e609b3ba55c73329c7a861d1d8a585ffdb36d247e4","observation_id":"893ff80e-8fda-45ad-a499-d6bdb6ec640a","resolution":{"observed_at":"2026-08-04T20:21:26.794373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.805169Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.805169Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:809fc1f1d6d1fe93bc6c090ac60de5ba8fccd1b5766fdbeb9070287213f2e348","observation_id":"5b154828-4cae-4fd3-9649-6cd1efd0b445","resolution":{"observed_at":"2026-08-04T20:21:26.805169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.01116","last_updated":"2023-06-01T20:03:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-01T20:03:56Z","title":"The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data, and Web Data Only","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.01116","snapshot_observed_at":"2026-08-04T20:21:26.801286Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.801286Z"},"links":{"cited_paper":"/paper/2306.01116","citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:8330cafbbfd7a5c2dc88a7ef7a2f51f88ab331827dd22c752c5068b760ae1eac","observation_id":"bbf80846-5ec0-4f29-bc6b-da0da4ab9efd","resolution":{"observed_at":"2026-08-04T20:21:26.801286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.815285Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.815285Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:65a226aeb1ad14885602b349067b07ebea463e0f55d18ac09ef7543771435945","observation_id":"e1a15269-5141-4f63-a9b5-bbe786654a02","resolution":{"observed_at":"2026-08-04T20:21:26.815285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.808564Z","title":"InProceedings of the 2021 International Conference on Multimodal Interaction, 341–351","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.808564Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:4b1430a8285e08a4f7987ebfbddfd6df49be4e5519b04ec1b030320683587801","observation_id":"e7086bd8-713e-4050-bf0d-c0bf106407e4","resolution":{"observed_at":"2026-08-04T20:21:26.808564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.811941Z","title":null,"venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.811941Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:ef2198ec9f58fa97666bd7c28645a88af1667ba99d0cbe8bf6bae2a6e1ffbdbf","observation_id":"77cc7e98-98a5-49bd-84de-b4d74f9e0737","resolution":{"observed_at":"2026-08-04T20:21:26.811941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.825177Z","title":"Salvucci and Joseph H","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.825177Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:0945f650e9860638512f245d4ab53526ce3fce43a6e4a1282de209c30560208b","observation_id":"6402f80a-5dea-4767-bc0d-b9f2ae1ec46d","resolution":{"observed_at":"2026-08-04T20:21:26.825177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.818671Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.818671Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:9429a96dd7cbf4e4cdbbbdecbeb638933c5c58fd5802cae4de625f9b787a893b","observation_id":"0c68d3db-00e4-4b41-8922-b4946a4c86cd","resolution":{"observed_at":"2026-08-04T20:21:26.818671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.821958Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.821958Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:2955e1e397ab4242ea40115495fb897ba791a3524c49645558a11876fd372a8a","observation_id":"02926ba3-0bd0-4434-865b-0ea7b4e8661e","resolution":{"observed_at":"2026-08-04T20:21:26.821958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-04T20:21:26.835652Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.835652Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:dd77185091913fad79feb7604c4af854d4f34c22f420a168a4f1697f177ca8fb","observation_id":"fb9ef76e-74f6-4ef6-8fd8-28f2cf095368","resolution":{"observed_at":"2026-08-04T20:21:26.835652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.828415Z","title":null,"venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.828415Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:b2d9215ee07106829d8095834d292ed5a1ee0bc023b7de93ef13633fc3050a7c","observation_id":"593087f7-a9ff-4318-8194-f2a466d62bb2","resolution":{"observed_at":"2026-08-04T20:21:26.828415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.832443Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.832443Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:207ead578c1e23562f895a31fb374467e54886898496d994c301b3b28d131891","observation_id":"ea8bcd52-0a75-4eac-8735-f0b6bb8ec029","resolution":{"observed_at":"2026-08-04T20:21:26.832443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.846421Z","title":"Stewart, and Sidney K","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.846421Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:e9d4ecba000b054263f3ad47446bcd8ac8fdc3cc0a4778063aa70401728ebae0","observation_id":"59bea344-2506-46f1-ad63-ef39b7e6476d","resolution":{"observed_at":"2026-08-04T20:21:26.846421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/32135","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:24:00.658490Z","title":null,"venue":null,"work_id":"884f5b94-59c8-46aa-b20f-8a4ccdf8fdee","year":2018},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.839381Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:758847c7c351025bbc841e2abd8013672cc80bbc408881abe64a2216f0e47781","observation_id":"e1447c76-6675-4a0b-ab90-ce467db8d35c","resolution":{"observed_at":"2026-08-04T20:24:00.728724Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.842962Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.842962Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:86411cc8b79b5fd2efa1e62c632f5155a01fb8d9881ace64905d114394ff3281","observation_id":"853f3768-4567-4c0b-bef0-2486bc3a654b","resolution":{"observed_at":"2026-08-04T20:21:26.842962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.863359Z","title":null,"venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.863359Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:51c8baba573ba12b3193241a8f36757981d68f95416f31a24a674a2f0182380b","observation_id":"bc1f4cbe-01af-4c11-bc0f-94036c6e9496","resolution":{"observed_at":"2026-08-04T20:21:26.863359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/j.patcog.2022.108","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:24:00.522568Z","title":null,"venue":null,"work_id":"026725ce-a41b-40d3-bf9e-d15ae4d85770","year":2022},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.866316Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:84e550c5ac3b3a4ed6c3cadcd92db5a67c17bec06e5c37f65ac3f549d32563d8","observation_id":"a719407d-12df-4a46-90ad-aca6fde37b2e","resolution":{"observed_at":"2026-08-04T20:24:00.581788Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.853409Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.853409Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:e8304f6efd88c84c3cf3210556d4fe3bf703a91eb7acb8988fecc11c036a0396","observation_id":"09d5415c-1510-4856-886a-3ac66241eabe","resolution":{"observed_at":"2026-08-04T20:21:26.853409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.07940","last_updated":"2019-09-18T04:08:27Z","snapshot_observed_at":"2026-08-10T18:09:08.747313Z","submitted_at":"2019-09-17T17:24:37Z","title":"Do NLP Models Know Numbers? Probing Numeracy in Embeddings","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.07940","snapshot_observed_at":"2026-08-04T20:21:26.856544Z","title":null,"venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.856544Z"},"links":{"cited_paper":"/paper/1909.07940","citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:558f38e8b633f0f0e5f171f1211c24a8defd0e1fcf1bee511a68c60e79a2078b","observation_id":"46ccc0c2-0ad7-4668-a2fc-9a67f7642c8e","resolution":{"observed_at":"2026-08-04T20:21:26.856544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.860119Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.860119Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:b5ebb5a5231cad9ae7ceb127278f4f7d4d71fe6c9147a5507e2dedf56fd63934","observation_id":"870fc309-f146-4eb4-b7e9-89e48979e497","resolution":{"observed_at":"2026-08-04T20:21:26.860119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.869847Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.869847Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:1d8d366639a8e9032387dc0add32d8bbc786f9311fcab9d508b9cfe8c75d0297","observation_id":"15996501-9d7f-4704-b786-549dccf7f1a5","resolution":{"observed_at":"2026-08-04T20:21:26.869847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.00421","last_updated":"2019-09-01T15:47:03Z","snapshot_observed_at":"2026-08-09T06:58:05.024347Z","submitted_at":"2019-09-01T15:47:03Z","title":"What You See is What You Get: Visual Pronoun Coreference Resolution in Dialogues","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.00421","snapshot_observed_at":"2026-08-04T20:21:26.873622Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.873622Z"},"links":{"cited_paper":"/paper/1909.00421","citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:f654c8b252ed84f59eda4321921e17eaba01ce8f6980e370c408a80d57714377","observation_id":"8b814328-4599-4e89-8c7d-384c209c4005","resolution":{"observed_at":"2026-08-04T20:21:26.873622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.876954Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.876954Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:2a253bded5a88afcb53bb1a63e881e9d5dfb14a6f0264628ca5e575832763c63","observation_id":"0b0f24d5-245c-47fc-aaaf-4d9c2d657fe6","resolution":{"observed_at":"2026-08-04T20:21:26.876954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.879903Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.879903Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:3ffa6f86028d63940cb2c394cb995bb3a710f0bf69c8787adf10bf84c96ca418","observation_id":"e71e024c-20f6-4fb0-afba-b737c80e807d","resolution":{"observed_at":"2026-08-04T20:21:26.879903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.849746Z","title":"doi:10.11 45/3290605.3300572","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.849746Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:0bafdb6677d200c18d057d3601daa4449b0c4e5f3595569d7c615a66be9fb6df","observation_id":"b5958067-d92e-4dbc-a34f-838de6771e9b","resolution":{"observed_at":"2026-08-04T20:21:26.849746Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:21:26.611834Z","title":"InConference on Human Factors in Com- puting Systems - Proceedings","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:26.611834Z"},"links":{"citing_paper":"/paper/2509.08689"},"observation_digest":"sha256:224b7ee87b4b1b3a0c00841808b3eece143ea970ac9fcfb3fb4af4024b071a74","observation_id":"07d96f6f-8a41-4005-a3aa-7b38e483ceb4","resolution":{"observed_at":"2026-08-04T20:21:26.611834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.08689","last_updated":"2025-09-10T15:27:17Z","latest_version":1,"primary_category":"cs.HC","snapshot_observed_at":"2026-08-12T20:12:09.921904Z","submitted_at":"2025-09-10T15:27:17Z","title":"Augmenting speech transcripts of VR recordings with gaze, pointing, and visual context for multimodal coreference resolution"},"reference_resolution":{"displayed":53,"state_counts":{"malformed_identifier":5,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":47,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":53},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 53 of 53 outbound references and 0 inbound Pith citation observations for arXiv:2509.08689."}