{"as_of":"2026-08-09T15:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:635110e5c555e70b6a4b5a9173c5bffc4fd4a505308697c9fa46bebec2689497","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-09T18:37:32.865795Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.00963/citation-record","integrity":"/paper/2605.00963/integrity","json":"/paper/2605.00963/citation-record.json","paper":"/paper/2605.00963"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.305535Z","title":"An advanced medical robotic system augment- ing healthcare capabilities-robotic nursing assistant","venue":null,"work_id":"1739c173-5870-4db6-8778-125af2d22ed3","year":2011},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:70ee4237adb7c13b79a535daf88ada10e3c87549f6d2af41cc90a656cec137df","observation_id":"183148c1-d87f-4995-936b-ad28412c9836","resolution":{"observed_at":"2026-05-25T22:57:14.203037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.301151Z","title":"A human-robot interac- tion applicution based on augmented reality (ar) for industrial robot grasping process","venue":null,"work_id":"16e25abc-6eb2-4b2e-9893-2a0899c272f3","year":2022},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:ba5f081e336ccd6a4469f9f7b3b4e8181181a313b4b90093439c3f59aa323389","observation_id":"4c39b7fb-c675-4a5e-b346-a92b0075a63a","resolution":{"observed_at":"2026-05-25T22:57:14.192863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.272575Z","title":"An educational robot system of visual question answering for preschoolers","venue":null,"work_id":"335a48d5-f7c8-447d-a426-6c999b844904","year":2017},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:2df84701f40a4858c48222ca1b98183878c82cca48f802c0414338c42c0971d7","observation_id":"5f767e74-d661-4836-97ee-93e2fa331a0d","resolution":{"observed_at":"2026-05-25T22:57:14.196266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.274728Z","title":"Home robot service by ceiling ultrasonic locator and microphone ar- ray","venue":null,"work_id":"4bddd1dc-523e-4e5c-8583-92dddb4a64df","year":2006},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:bba5bae4b2152d74df95fc5108383893ec7913906ecd0d47822c404c9862b02a","observation_id":"1b97e164-789b-43b4-bb42-d7ad824ecdbb","resolution":{"observed_at":"2026-05-25T22:57:14.199644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.303577Z","title":"The human intention: a taxonomy attempt and its applications to robotics","venue":null,"work_id":"28cfe93f-0dac-4c1f-9ad8-5b450f433a2a","year":2025},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:c9615531ff072b22bb9b8de229825e9af463acadf0cf0f73fab2857a69b653a6","observation_id":"4cb39b36-ef97-4a59-97be-d4ef7086790e","resolution":{"observed_at":"2026-05-25T22:57:14.209649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.296996Z","title":"Anticipatory robot control for efficient human-robot collaboration","venue":null,"work_id":"64052c32-f248-4c33-b184-957288c6075f","year":2016},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:b4e8cfc2fb07122a86aa854eb42500e2b426978e850783fae50f2cabc13becac","observation_id":"dad20ba3-845b-467a-8972-462403d80010","resolution":{"observed_at":"2026-05-25T22:57:14.212910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.299182Z","title":"Pointing gestures for human-robot interaction with the humanoid robot digit","venue":null,"work_id":"c3b47315-b539-4bf6-b2b8-681cc15ef01c","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:cc58d4f8957553e70e0d0107d31ec4d41c583be18cb199f04ada2e593425151b","observation_id":"2497d62a-9f5e-4f98-8efc-6ffa89c4794e","resolution":{"observed_at":"2026-05-25T22:57:14.175250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.294126Z","title":"Autonomous laparoscopic robotic suturing with a novel actuated suturing tool and 3d endoscope","venue":null,"work_id":"04bf2375-6398-4dbd-9d5a-58319fa36758","year":2019},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:188c60aab1764799f6490ad2af1fc3c578b59cd97825a8ce5839c1b511ce790d","observation_id":"b4cf76bc-5a15-4b13-bf98-bd7e7f99f3ed","resolution":{"observed_at":"2026-05-25T22:57:14.178563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Perception– intention–action cycle in human–robot collaborative tasks: the col- laborative lightweight object transportation use-case","venue":null,"work_id":"59bc4d20-e83f-4f02-a8b0-11f932d2a4fe","year":1927},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:b082e6981672e6b59e520fc4030d3bbc3b0b9de21e6452bd90fe7a7738bbc1d3","observation_id":"57376c0c-cd56-4d10-b263-a654bf42da18","resolution":{"observed_at":"2026-05-25T22:57:14.182051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.291966Z","title":"Exploring transformers and visual transformers for force prediction in human-robot collaborative transportation tasks","venue":null,"work_id":"fc61b4e1-3e97-4391-b007-b776a042e916","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:e562a7843a5176bacc835e3db4483627a2aa8da6a05280739fa3cf90a66c6c95","observation_id":"72930273-a607-48c6-8b7a-fa1bf6619a00","resolution":{"observed_at":"2026-05-25T22:57:14.185616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.293211Z","title":"Force and velocity predic- tion in human-robot collaborative transportation tasks through video retentive networks","venue":null,"work_id":"3ba3acbe-c477-48cb-a4d1-475691460963","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:afcb1a4ef6475986f948ff0689a5d9591061b8e9b2c8a70e627118248627e72a","observation_id":"0bae2e8b-63e6-4330-afbc-a4e404d56cb2","resolution":{"observed_at":"2026-05-25T22:57:14.167802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.295249Z","title":"Language and sketching: An llm-driven interactive multimodal multitask robot navigation framework","venue":null,"work_id":"2adc80bc-c0a8-4f61-8200-d879f6d10b87","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:ac5396b7ecf2a8d1f979904cda5aa93d05ff52a3d0fa924069002dd5a5996268","observation_id":"f2372dc3-f71f-4b7e-afbe-ae0f8c7f8912","resolution":{"observed_at":"2026-05-25T22:57:14.160638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.285725Z","title":"Interactive navigation in environments with traversable obstacles using large language and vision-language models","venue":null,"work_id":"87f2376f-2477-4615-ad9f-754429401264","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:d062b2ed95f4b59d6e17f352a3d1c088f84340116bdeb1743ea123244e097a5e","observation_id":"22cf6f80-e16a-4957-ba16-d774eb3240a8","resolution":{"observed_at":"2026-05-25T22:57:14.164292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.289433Z","title":"Physically grounded vision-language models for robotic manipulation","venue":null,"work_id":"74c99908-c5dd-4973-b431-0a06336184f3","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:872d96b88e496bc0ef9424c079fdbc92fc91ebe20c192dd931432c969e0732fe","observation_id":"8ed54e66-2d45-4925-a56b-e6218c2ba471","resolution":{"observed_at":"2026-05-25T22:57:14.171128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.290850Z","title":"When the inference meets the explicitness or why multimodality can make us forget about the perfect predictor","venue":null,"work_id":"4bd5bf2a-0ab5-4012-b2d9-8d15665669a9","year":2025},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:d1ee3eb28426dd0d63e58ce32427645be9d11ea36fd68062f384eb758d7f3595","observation_id":"030fc9cd-05b7-426a-b924-e44f6f4c3c36","resolution":{"observed_at":"2026-05-25T22:57:14.189369Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.270262Z","title":"Anticipation and proactivity. unraveling both concepts in human-robot interaction through a han- dover example","venue":null,"work_id":"52c74583-aa5e-4a0b-99ec-1e1c9baa0961","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:d82ba015df7dcb1a6839dd2c08bc84038f69b43e3722c3674c41838aef28fce8","observation_id":"f8fd0236-cd83-4c0c-ae9e-ac6ae973bd37","resolution":{"observed_at":"2026-05-25T22:57:14.206291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:e75b100e5616713c7af9cfe6ce26bd544b3f3e7548522b53cb0960297c247e71","observation_id":"f2799689-edbc-4285-8b07-4686b201b390","resolution":{"observed_at":"2026-05-11T16:11:08.303610Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:f3c8ba65fb2c9be450592e3d099604af8e9b4400ef27c5af81c700d2cb550296","observation_id":"3b09c363-9f43-462d-82f8-163f0228952d","resolution":{"observed_at":"2026-05-11T16:11:08.332019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00693","last_updated":"2024-11-27T12:30:23Z","snapshot_observed_at":"2026-07-06T18:08:21.770341Z","submitted_at":"2024-03-26T15:36:40Z","title":"Leveraging Large Language Models in Human-Robot Interaction: A Critical Analysis of Potential and Pitfalls","version":2},"cited_work":{"arxiv_id":"2405.00693","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00693","snapshot_observed_at":"2026-07-02T11:56:54.936111Z","title":"Leveraging large language models in human-robot in- teraction: a critical analysis of potential and pitfalls","venue":null,"work_id":"85cd3b4c-7bfe-4b44-bac1-3ecfb1529e64","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2405.00693","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:7e57eb591ec9b74dcd3bdffbdabf596b809b36b79f168455cae201d62cc0a2dc","observation_id":"de45a080-e331-48cc-8793-67ce3e8e4c70","resolution":{"observed_at":"2026-05-11T16:11:08.312260Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T16:47:24.923516Z","title":"Florence-2: Advancing a unified representation for a variety of vision tasks","venue":null,"work_id":"eebbb503-7be4-493b-b550-ea3d4b11961c","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:1bcc4979f5545463e352bb90e30cd0e53a4d08b1771e486f7f83ec1abeacb492","observation_id":"2d6f11dc-a3a9-4e3e-9ad5-b08864f22db5","resolution":{"observed_at":"2026-05-25T22:57:14.216378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.309896Z","title":"Robust speech recognition via large-scale weak super- vision","venue":null,"work_id":"e370c627-d78e-4144-bb57-708bb23e78f2","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:973c1b37b81c959b10dbdf7f7a5e447229925323686d7eba447e3d4d9f0e6a23","observation_id":"648ebf6d-0598-46ac-96bf-602e3686f29a","resolution":{"observed_at":"2026-05-25T22:57:14.219864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.01778","last_updated":"2021-07-08T20:16:28Z","snapshot_observed_at":"2026-08-08T11:33:21.735489Z","submitted_at":"2021-04-05T05:26:29Z","title":"AST: Audio Spectrogram Transformer","version":3},"cited_work":{"arxiv_id":"2104.01778","doi":"10.48550/arxiv.2104.01778","metadata_source":"pith","pith_arxiv_id":"2104.01778","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ast: Audio spectrogram transformer","venue":"cs.SD","work_id":"b697ee73-6e22-4cba-84d1-4a1dab594872","year":2021},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2104.01778","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:01287f6747982d36630de014562cfb4370cf95f600d70653aa64d0d89b653d16","observation_id":"1e947acc-fc63-438d-920b-2950bf62b7eb","resolution":{"observed_at":"2026-05-11T16:11:08.290066Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-07-13T18:20:56.483183+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T18:20:56.483183+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.268382Z","title":"Fuzzy logic systems for engineering: a tutorial","venue":null,"work_id":"c243c4bd-a053-43ee-a3ab-30f7ffa1a1b2","year":2002},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:2fc98e10b00d6bbb3bc69e8cb7ab3285520ee70b1401d42aa9a678ec0e393d57","observation_id":"b3a39b44-fbaa-42af-a3df-96d3d1081215","resolution":{"observed_at":"2026-05-25T22:57:14.149913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.262036Z","title":"Interval type-2 fuzzy logic systems: theory and design","venue":null,"work_id":"35502576-d58d-42d7-aa8e-e9924a247e97","year":2000},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:5634ffd9084b07ff1eb8a7cd7082028cac4c0c0a29fce9cd86d0ad118502458c","observation_id":"3c8abc05-8cd9-455e-82ed-4a69708aef0f","resolution":{"observed_at":"2026-05-25T22:57:14.153373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.264268Z","title":"Fuzzy logic introduction","venue":null,"work_id":"382d01b8-f09c-4a72-90c6-f873524615ea","year":2001},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:713d00124372e1f795113a1b82a7f0b965d89988fdac063eaf879d8c0d695e21","observation_id":"8491ac5d-1db2-4bc8-8f56-ca3ac55856c1","resolution":{"observed_at":"2026-05-25T22:57:14.156900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.266079Z","title":"A type-2 fuzzy logic controller for autonomous mobile robots","venue":null,"work_id":"77449584-5ff3-40d2-83eb-c9ddfdcf924e","year":2004},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:e03d1b2956d57fe3332af83ab64c1d189be04fbe18d139d1c70597d475af370e","observation_id":"b35d1674-42f5-4558-8681-a1df0b965274","resolution":{"observed_at":"2026-05-25T22:57:14.146744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.20219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T11:56:54.946441Z","title":"An approach to combining video and speech with large language models in human-robot interaction","venue":null,"work_id":"dc9ee25a-94d7-4e6c-8d22-a122903c157f","year":2026},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:26a71ddae5f6984c6da97e4411a48360ced0a7f7d26cfbc1fb294cc4f13ea624","observation_id":"d5daa156-1f5f-444d-8b37-f23969b1d5f3","resolution":{"observed_at":"2026-05-11T16:11:08.325436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":5,"verified_fuzzy":22},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2605.00963."}