{"as_of":"2026-08-18T20:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ff45853043939fe1be32861cd3bc8994514eb424f64e2db26486290b7abb62f3","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T17:34:42.782501Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.16143/citation-record","integrity":"/paper/2508.16143/integrity","json":"/paper/2508.16143/citation-record.json","paper":"/paper/2508.16143"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:46.637560Z","title":"Survey on Frontiers of Language and Robotics,","venue":null,"work_id":"3cff4c30-d035-4b75-ad82-7bba5ca7c8ab","year":2019},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.374396Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:4f029429fd37c43f873e660e792513d7deb05bf94bf1f9e1cd0c782182c457ac","observation_id":"24d0eb62-8dee-4a3a-bbe8-d64f196edec4","resolution":{"observed_at":"2026-08-05T17:34:46.696164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:46.548361Z","title":"Visual Language Integration: A Survey and Open Challenges,","venue":null,"work_id":"d2620836-7bf5-4c4c-8c3e-1ce8623ddbf5","year":2023},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.459008Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:2ee4e95150643b2845e85febb0bdc7dff32e3f6ee548cdf5eab7465d33f737a6","observation_id":"f9f2efe7-c557-4134-9e1f-fd6be87804f8","resolution":{"observed_at":"2026-08-05T17:34:46.576011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:46.521661Z","title":"Exophora Resolution of Linguistic Instructions with a Demonstrative based on Real-World Multimodal Information,","venue":null,"work_id":"13b15300-5679-4f57-b4d2-62d70021e568","year":2023},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.572677Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:13ab2c7372a2afd88235713f4671930acaacd1e611054f005e82d4597dce3276","observation_id":"9deaeb83-cf73-473e-9fb7-8a82992aea12","resolution":{"observed_at":"2026-08-05T17:34:46.540207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:46.467546Z","title":"Gesture-Informed Robot Assistance via Foundation Models,","venue":null,"work_id":"b59a97bd-e33c-4785-a691-beee08729036","year":2023},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.634548Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:26622c6d66f370039bd4f7119adffef41617d5067658c02e498876996b4b1696","observation_id":"a45587cc-f53b-471e-98d6-c44914d6e572","resolution":{"observed_at":"2026-08-05T17:34:46.490033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:46.371390Z","title":"Pointing Frame Estimation with Audio-Visual Time Series Data for Daily Life Service Robots,","venue":null,"work_id":"e77954e7-bac3-4856-b4c0-73a72480a06f","year":2024},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.726796Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:126bb633d47ba1377d05f8967d630f03a0016a7ccd72355ba9c6d6478ab3010e","observation_id":"788e142b-0e03-4e69-a610-54a5e0c17c82","resolution":{"observed_at":"2026-08-05T17:34:46.433945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:46.156593Z","title":"Learning Transferable Visual Models from Natural Language Supervision,","venue":null,"work_id":"1fbe5285-9388-4612-8713-7772fab3689c","year":2021},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.804380Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:585d23dc5e7442f8196c64d1f7675c7353454a9fe0ee55e2176b2832edd76292","observation_id":"efdea9b3-39cd-46f8-805d-652b62f37b8d","resolution":{"observed_at":"2026-08-05T17:34:46.300547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:45.851727Z","title":"DINOv2: Learning Robust Visual Features without Supervision,","venue":null,"work_id":"9903f2ab-864c-4d1d-8a94-8dce8026f9ee","year":2024},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.899537Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:0093be78962a290c3394395666bb38519baf16a5b76c567ffe37fdc35ece0e95","observation_id":"0b239b9e-1721-4e4d-b7fa-19d696bdc69d","resolution":{"observed_at":"2026-08-05T17:34:45.975852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:45.569140Z","title":"What You See is What You Get: Visual Pronoun Coreference Resolution in Dialogues,","venue":null,"work_id":"d30dd600-3ca1-4949-9c39-6c65e4904072","year":2019},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:40.985462Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:07aed3734347f87943ce7f81efd1296463e716574f8c4bca4054d134807d9c43","observation_id":"7264a837-1c45-40c6-992c-25ac7b3909e2","resolution":{"observed_at":"2026-08-05T17:34:45.699271Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:45.278670Z","title":"Exophoric Pronoun Resolution in Dialogues with Topic Reg- ularization,","venue":null,"work_id":"624c0ddf-2fea-46db-976e-1956e48d3c2c","year":2021},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.050478Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:c08a1939476584cdeccd94406761f68dfa60c4f19660f5971d8cc0f122c7cfbb","observation_id":"00761231-c0b8-48b9-8740-e75a9ee1b777","resolution":{"observed_at":"2026-08-05T17:34:45.416152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.983389Z","title":"Dual Attention Networks for Visual Reference Resolution in Visual Dialog,","venue":null,"work_id":"db3a2d08-df37-444e-b155-dc6b9cce4e90","year":2019},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.177824Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:9da2bd928065a0a9d20bc7f67ab1954b7d69d03e269aebb07f20c62cb20a06fb","observation_id":"875936f8-84ea-4ca0-a951-bf229da22577","resolution":{"observed_at":"2026-08-05T17:34:45.089094Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.877725Z","title":"VD-PCR: Improving Visual Dialog with Pronoun Coreference Resolution,","venue":null,"work_id":"aa84c2fa-a839-4934-a3eb-d4269721bf16","year":2022},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.282219Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:b945bcb0e25e66a18cdf93a7ad25c3f63392fa4298f923dfda9a20bb79b29c5b","observation_id":"199f6d0b-e5d1-4b34-99e8-00be5269dad7","resolution":{"observed_at":"2026-08-05T17:34:44.922265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.786861Z","title":"A Gaze-grounded Visual Question Answering Dataset for Clarifying Ambiguous Japanese Questions,","venue":null,"work_id":"5f9a4195-af86-4d3e-9a15-8cb3800ce871","year":2024},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.349015Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:5775e78cae91d55b5ad72e489e980c087e861dc4f15a7b1021638bb079e6b9cf","observation_id":"ed422ba5-2f7e-4b8c-87e3-3d34699bbf6b","resolution":{"observed_at":"2026-08-05T17:34:44.823538Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.716443Z","title":"ECRAP: Exophora Resolution and Classifying User Commands for Robot Action Planning by Large Language Models,","venue":null,"work_id":"64885cf3-6abd-4d48-ba2a-47e397595829","year":2024},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.450582Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:a0dcb958b1b7cdc1cf5a9695e8367d25ebd56766c2b007def70d78a146943c84","observation_id":"89eeb5df-5b08-401e-9eb3-2ba5da59f45a","resolution":{"observed_at":"2026-08-05T17:34:44.763356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T17:34:41.531064Z","title":"GPT-4o System Card,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.531064Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:435f62c988b2d869c74fb237bfd32b626e9174ad5299cc3c1264df064f64472f","observation_id":"b4cb4ec8-16d5-4f78-9492-9e172316f1c8","resolution":{"observed_at":"2026-08-05T17:34:41.531064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.556616Z","title":"J-CRe3: A Japanese Conversation Dataset for Real- world Reference Resolution,","venue":null,"work_id":"9a401397-b780-416f-b40a-395340d01ef2","year":2024},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.589619Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:5f20c9a1bf4de818ee950da7ab7fb685a3a34c44b764d5f234c77334cd6d2420","observation_id":"16a7f76a-0f85-44fa-b64e-e20929b3846b","resolution":{"observed_at":"2026-08-05T17:34:44.651196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.360406Z","title":"This&That: Language-Gesture Controlled Video Generation for Robot Planning,","venue":null,"work_id":"de6540ba-fd4a-44ad-90fd-222b257b9d1c","year":2025},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.679499Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:348461dd1d99df981380593dfa9bb3194ce2be2af8275aeddbc8ba9f0c66b17a","observation_id":"2b3c8da0-e4ce-49ba-b27f-1d7fe84358a1","resolution":{"observed_at":"2026-08-05T17:34:44.456531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.217541Z","title":"Robots That Ask For Help: Uncertainty Alignment for Large Language Model Planners,","venue":null,"work_id":"1e01d1bc-176d-47e0-9d15-32e06373ce7e","year":2023},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.770957Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:66a7478a4c3893a5732e6015b37fb2a03983dde7d4e463782af2310b926a256e","observation_id":"c8ed7888-31c8-43db-950c-329fbf1f218c","resolution":{"observed_at":"2026-08-05T17:34:44.286263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:44.017381Z","title":"CLARA: Classifying and Disambiguating User Com- mands for Reliable Interactive Robotic Agents,","venue":null,"work_id":"e23247ac-55eb-44c1-93e0-f7f2bd179d6c","year":2023},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.874225Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:1f76416b475939c07bb41f3eaef1c95a9353f578d8081c28087dcb21e894e880","observation_id":"42483345-ffbd-4308-a929-e5be14f085f6","resolution":{"observed_at":"2026-08-05T17:34:44.116903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:43.888164Z","title":"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks,","venue":null,"work_id":"ae12d2b7-3c82-4dd7-9775-8334e30875b3","year":2019},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:41.965896Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:41b82be166608e18ade2cd9a605811457106d290b72d0b9d1ea5381ee1a4af5a","observation_id":"09312c8e-6062-4af7-bcec-154f33325dee","resolution":{"observed_at":"2026-08-05T17:34:43.948644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:43.745064Z","title":"Open-V ocabulary Queryable Scene Representations for Real World Planning,","venue":null,"work_id":"14ee8199-11de-4c0b-a804-5d2d49fcdb0c","year":2023},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:42.085790Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:7468450755d254970b7cab4a5f354efc6d3511107c45f6a367911608652c7f12","observation_id":"269920f9-e7da-455c-8fc0-4b2d501ec889","resolution":{"observed_at":"2026-08-05T17:34:43.820032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1906.08172","last_updated":"2019-06-14T05:49:22Z","snapshot_observed_at":"2026-08-18T14:57:44.459315Z","submitted_at":"2019-06-14T05:49:22Z","title":"MediaPipe: A Framework for Building Perception Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.08172","snapshot_observed_at":"2026-08-05T17:34:42.172004Z","title":"MediaPipe: A Framework for Building Perception Pipelines,","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:42.172004Z"},"links":{"cited_paper":"/paper/1906.08172","citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:ef6866ef0dd982a43d1b373d6f3d0485cc8ab1e1161d594ae043393ad5e4248d","observation_id":"781d0aee-e71e-411a-82fc-1753ea7d0f6e","resolution":{"observed_at":"2026-08-05T17:34:42.172004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:43.582351Z","title":"Software Development Environment for Collabo- rative Research Workflow in Robotic System Integration,","venue":null,"work_id":"cb1d8c52-91bb-4c97-ba6f-249914d574f8","year":2022},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:42.306092Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:0589814c5f9c5b7efe5f1f3c9865263f74ff8d832d76f9f5b18a78122909247a","observation_id":"d5be5f03-abc5-4848-823a-44f48ca25b73","resolution":{"observed_at":"2026-08-05T17:34:43.642545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:43.409578Z","title":"Development of Human Support Robot as the research platform of a domestic mobile manipulator,","venue":null,"work_id":"27a0e828-445c-4752-b005-75d96ae887e3","year":2019},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:42.439075Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:daf59a71136971d7a35f1e29387347684f568eb78a43fe1479d41222168d8c34","observation_id":"9d25c5ee-1a82-4271-975c-ba39d50da2e4","resolution":{"observed_at":"2026-08-05T17:34:43.483592Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:43.238689Z","title":"Detecting Twenty-Thousand Classes using Image- Level Supervision,","venue":null,"work_id":"5ac1b394-bc52-4eed-937e-574e623d4e1b","year":2022},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:42.571612Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:6b1287da43d95e4b7b952fba6c65273afd0d091745b95cb2782553b9b8735f5b","observation_id":"c2f5b1ec-28cf-4b86-9c3d-2b118e926a48","resolution":{"observed_at":"2026-08-05T17:34:43.313745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:43.109121Z","title":"Objects365: A Large-Scale, High-Quality Dataset for Object Detection,","venue":null,"work_id":"862e2b42-8c52-43c9-8d80-82f1475989f2","year":2019},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:42.699967Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:75b09e9bde47dd464f3e1375c2890e832c72470884ad02a6e547b3059c7be8ab","observation_id":"4fb35c5d-de75-4cf2-8d4e-09a00ca1cc4f","resolution":{"observed_at":"2026-08-05T17:34:43.182943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T17:34:42.896853Z","title":"VGPN: V oice-Guided Pointing Robot Navigation for Humans,","venue":null,"work_id":"41a15281-b5b5-4145-b41f-b7e8891e18be","year":2018},"citing_paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T17:34:42.782501Z"},"links":{"citing_paper":"/paper/2508.16143"},"observation_digest":"sha256:cd900cceb83aa37d8f91d575bd9adb0baeb95296a81ea72bda305f170a7d295f","observation_id":"aa7534dc-c874-4f88-b350-0f024578b84f","resolution":{"observed_at":"2026-08-05T17:34:42.985630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.16143","last_updated":"2025-08-22T07:09:06Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-05T17:34:39.638685Z","submitted_at":"2025-08-22T07:09:06Z","title":"Take That for Me: Multimodal Exophora Resolution with Interactive Questioning for Ambiguous Out-of-View Instructions"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":2,"verified_exact":0,"verified_fuzzy":24},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 0 inbound Pith citation observations for arXiv:2508.16143."}