{"as_of":"2026-08-10T12:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b7ba21a96db9aa6768cab535eb5b4e01ca8d87095d08483ec679e9c113703de8","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-09T18:37:32.865795Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.00963/citation-record","integrity":"/paper/2605.00963/integrity","json":"/paper/2605.00963/citation-record.json","paper":"/paper/2605.00963"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.305535Z","title":"An advanced medical robotic system augment- ing healthcare capabilities-robotic nursing assistant","venue":null,"work_id":"1739c173-5870-4db6-8778-125af2d22ed3","year":2011},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:a432d0a248357b39dd17575ddb0b3d6a4440d01b3433e4a017d331c5774641d2","observation_id":"183148c1-d87f-4995-936b-ad28412c9836","resolution":{"observed_at":"2026-05-25T22:57:14.203037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.301151Z","title":"A human-robot interac- tion applicution based on augmented reality (ar) for industrial robot grasping process","venue":null,"work_id":"16e25abc-6eb2-4b2e-9893-2a0899c272f3","year":2022},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:7287e5b54a0e6c52d325bb68fb15dc0b880134c59b85ba08b3694f8b73608c36","observation_id":"4c39b7fb-c675-4a5e-b346-a92b0075a63a","resolution":{"observed_at":"2026-05-25T22:57:14.192863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.272575Z","title":"An educational robot system of visual question answering for preschoolers","venue":null,"work_id":"335a48d5-f7c8-447d-a426-6c999b844904","year":2017},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:568a78a8d3abfb828264f6edb7583ee77712db9e59d108d6e3d277578af77587","observation_id":"5f767e74-d661-4836-97ee-93e2fa331a0d","resolution":{"observed_at":"2026-05-25T22:57:14.196266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.274728Z","title":"Home robot service by ceiling ultrasonic locator and microphone ar- ray","venue":null,"work_id":"4bddd1dc-523e-4e5c-8583-92dddb4a64df","year":2006},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:596a80165888f85823b75497e5287b1c46b5e9731cc064a880fa5a857d2df328","observation_id":"1b97e164-789b-43b4-bb42-d7ad824ecdbb","resolution":{"observed_at":"2026-05-25T22:57:14.199644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.303577Z","title":"The human intention: a taxonomy attempt and its applications to robotics","venue":null,"work_id":"28cfe93f-0dac-4c1f-9ad8-5b450f433a2a","year":2025},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:9f6794814bf7246e241d549da40ea27b0fdc4724316d1838efc4bc2b940cac7a","observation_id":"4cb39b36-ef97-4a59-97be-d4ef7086790e","resolution":{"observed_at":"2026-05-25T22:57:14.209649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.296996Z","title":"Anticipatory robot control for efficient human-robot collaboration","venue":null,"work_id":"64052c32-f248-4c33-b184-957288c6075f","year":2016},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:bf43cec06691f119c6e3ffdd91049cda951c792f944a3b14a55e5d26a851f150","observation_id":"dad20ba3-845b-467a-8972-462403d80010","resolution":{"observed_at":"2026-05-25T22:57:14.212910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.299182Z","title":"Pointing gestures for human-robot interaction with the humanoid robot digit","venue":null,"work_id":"c3b47315-b539-4bf6-b2b8-681cc15ef01c","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:c68bf476f4bc106df55a86b32585ec1fcda5d77684f3eaa04ff85a7827bbb124","observation_id":"2497d62a-9f5e-4f98-8efc-6ffa89c4794e","resolution":{"observed_at":"2026-05-25T22:57:14.175250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.294126Z","title":"Autonomous laparoscopic robotic suturing with a novel actuated suturing tool and 3d endoscope","venue":null,"work_id":"04bf2375-6398-4dbd-9d5a-58319fa36758","year":2019},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:de188c74a5b66ecd4ee68fe65658c161de072ee9b8126015ef3d18295b716cca","observation_id":"b4cf76bc-5a15-4b13-bf98-bd7e7f99f3ed","resolution":{"observed_at":"2026-05-25T22:57:14.178563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Perception– intention–action cycle in human–robot collaborative tasks: the col- laborative lightweight object transportation use-case","venue":null,"work_id":"59bc4d20-e83f-4f02-a8b0-11f932d2a4fe","year":1927},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:89a496e8ebe1a77a4795adb9c96d663fdebd22185f7a8f86125f8a50f9f21280","observation_id":"57376c0c-cd56-4d10-b263-a654bf42da18","resolution":{"observed_at":"2026-05-25T22:57:14.182051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.291966Z","title":"Exploring transformers and visual transformers for force prediction in human-robot collaborative transportation tasks","venue":null,"work_id":"fc61b4e1-3e97-4391-b007-b776a042e916","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:c2e4ba8bba6d48a628cfcada4543470df8045dcd82e71a3fc74d88d69342aa0f","observation_id":"72930273-a607-48c6-8b7a-fa1bf6619a00","resolution":{"observed_at":"2026-05-25T22:57:14.185616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.293211Z","title":"Force and velocity predic- tion in human-robot collaborative transportation tasks through video retentive networks","venue":null,"work_id":"3ba3acbe-c477-48cb-a4d1-475691460963","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:815f9b97e297a337f004222768282c7a572d1c9f9725e98cf5c7a22671a9f050","observation_id":"0bae2e8b-63e6-4330-afbc-a4e404d56cb2","resolution":{"observed_at":"2026-05-25T22:57:14.167802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.295249Z","title":"Language and sketching: An llm-driven interactive multimodal multitask robot navigation framework","venue":null,"work_id":"2adc80bc-c0a8-4f61-8200-d879f6d10b87","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:d698e168ce3c85a91089c4a2e7de70e3aa878fb579c6eb68f44b3a4d181d4e9b","observation_id":"f2372dc3-f71f-4b7e-afbe-ae0f8c7f8912","resolution":{"observed_at":"2026-05-25T22:57:14.160638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.285725Z","title":"Interactive navigation in environments with traversable obstacles using large language and vision-language models","venue":null,"work_id":"87f2376f-2477-4615-ad9f-754429401264","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:5be5ca422215af019ce7d683a493958b9a50f6ee887fc219d93375ebbb5dd88d","observation_id":"22cf6f80-e16a-4957-ba16-d774eb3240a8","resolution":{"observed_at":"2026-05-25T22:57:14.164292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.289433Z","title":"Physically grounded vision-language models for robotic manipulation","venue":null,"work_id":"74c99908-c5dd-4973-b431-0a06336184f3","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:afea8e03a60dded7cb0c10d025725115252d7b35257e08a5d57327471edb0111","observation_id":"8ed54e66-2d45-4925-a56b-e6218c2ba471","resolution":{"observed_at":"2026-05-25T22:57:14.171128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.290850Z","title":"When the inference meets the explicitness or why multimodality can make us forget about the perfect predictor","venue":null,"work_id":"4bd5bf2a-0ab5-4012-b2d9-8d15665669a9","year":2025},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:a68d89372eddd3c26d10d94133dca20569d088ba2150d712cc7f802b0b4ffcc9","observation_id":"030fc9cd-05b7-426a-b924-e44f6f4c3c36","resolution":{"observed_at":"2026-05-25T22:57:14.189369Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.270262Z","title":"Anticipation and proactivity. unraveling both concepts in human-robot interaction through a han- dover example","venue":null,"work_id":"52c74583-aa5e-4a0b-99ec-1e1c9baa0961","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:3333eab4b31071a691f883a1edab9d153be24bde32d2d2608e41f02ba34b445f","observation_id":"f8fd0236-cd83-4c0c-ae9e-ac6ae973bd37","resolution":{"observed_at":"2026-05-25T22:57:14.206291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:f6c6bdc57bc4d8e1ad845216530be5c87ebbff89278aa511518f694c198a7824","observation_id":"f2799689-edbc-4285-8b07-4686b201b390","resolution":{"observed_at":"2026-05-11T16:11:08.303610Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:6adfe8512d0a19581d93cc9c76f6101f4241a286874bd757b10ba6e5df840a2d","observation_id":"3b09c363-9f43-462d-82f8-163f0228952d","resolution":{"observed_at":"2026-05-11T16:11:08.332019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00693","last_updated":"2024-11-27T12:30:23Z","snapshot_observed_at":"2026-08-10T06:27:31.193352Z","submitted_at":"2024-03-26T15:36:40Z","title":"Leveraging Large Language Models in Human-Robot Interaction: A Critical Analysis of Potential and Pitfalls","version":2},"cited_work":{"arxiv_id":"2405.00693","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00693","snapshot_observed_at":"2026-07-02T11:56:54.936111Z","title":"Leveraging large language models in human-robot in- teraction: a critical analysis of potential and pitfalls","venue":null,"work_id":"85cd3b4c-7bfe-4b44-bac1-3ecfb1529e64","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2405.00693","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:2e372ae387e609feaa02a3ca5074f6c2941e5fc1c58405894878064e31213d01","observation_id":"de45a080-e331-48cc-8793-67ce3e8e4c70","resolution":{"observed_at":"2026-05-11T16:11:08.312260Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T16:47:24.923516Z","title":"Florence-2: Advancing a unified representation for a variety of vision tasks","venue":null,"work_id":"eebbb503-7be4-493b-b550-ea3d4b11961c","year":2024},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:6a2250109a55e3d6b7bfeacd8641014b1f5283b35443aaf506031f60399d1075","observation_id":"2d6f11dc-a3a9-4e3e-9ad5-b08864f22db5","resolution":{"observed_at":"2026-05-25T22:57:14.216378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.309896Z","title":"Robust speech recognition via large-scale weak super- vision","venue":null,"work_id":"e370c627-d78e-4144-bb57-708bb23e78f2","year":2023},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:2618d836a818bfb604ff7f1143c87d29e16b2621c6ccc284b45f8f6900d78772","observation_id":"648ebf6d-0598-46ac-96bf-602e3686f29a","resolution":{"observed_at":"2026-05-25T22:57:14.219864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.01778","last_updated":"2021-07-08T20:16:28Z","snapshot_observed_at":"2026-08-08T11:33:21.735489Z","submitted_at":"2021-04-05T05:26:29Z","title":"AST: Audio Spectrogram Transformer","version":3},"cited_work":{"arxiv_id":"2104.01778","doi":"10.48550/arxiv.2104.01778","metadata_source":"pith","pith_arxiv_id":"2104.01778","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ast: Audio spectrogram transformer","venue":"cs.SD","work_id":"b697ee73-6e22-4cba-84d1-4a1dab594872","year":2021},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"cited_paper":"/paper/2104.01778","citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:686907ef1e94c0442a87a9e5209dbc12f70a6c5938e2121c0f078afb18a188f8","observation_id":"1e947acc-fc63-438d-920b-2950bf62b7eb","resolution":{"observed_at":"2026-05-11T16:11:08.290066Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-07-13T18:20:56.483183+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T18:20:56.483183+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.268382Z","title":"Fuzzy logic systems for engineering: a tutorial","venue":null,"work_id":"c243c4bd-a053-43ee-a3ab-30f7ffa1a1b2","year":2002},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:f008bed44528f24230634dead7dd26adf6d711281574bcca8ef5d459cf289ccc","observation_id":"b3a39b44-fbaa-42af-a3df-96d3d1081215","resolution":{"observed_at":"2026-05-25T22:57:14.149913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.262036Z","title":"Interval type-2 fuzzy logic systems: theory and design","venue":null,"work_id":"35502576-d58d-42d7-aa8e-e9924a247e97","year":2000},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:4e6f7b6c140f8f77a4a1494ff3ccafc9a6b384204783fc16f140960f24075125","observation_id":"3c8abc05-8cd9-455e-82ed-4a69708aef0f","resolution":{"observed_at":"2026-05-25T22:57:14.153373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.264268Z","title":"Fuzzy logic introduction","venue":null,"work_id":"382d01b8-f09c-4a72-90c6-f873524615ea","year":2001},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:ba19461195630dc71744cb7caa32a62cc6fadaa1fcbb0794b6d70e5d38fcd6eb","observation_id":"8491ac5d-1db2-4bc8-8f56-ca3ac55856c1","resolution":{"observed_at":"2026-05-25T22:57:14.156900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:41:50.266079Z","title":"A type-2 fuzzy logic controller for autonomous mobile robots","venue":null,"work_id":"77449584-5ff3-40d2-83eb-c9ddfdcf924e","year":2004},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:032a9b4722abb870256868ad0abc07037cd8532012ea6e493936857291b5b457","observation_id":"b35d1674-42f5-4558-8681-a1df0b965274","resolution":{"observed_at":"2026-05-25T22:57:14.146744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.20219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T11:56:54.946441Z","title":"An approach to combining video and speech with large language models in human-robot interaction","venue":null,"work_id":"dc9ee25a-94d7-4e6c-8d22-a122903c157f","year":2026},"citing_paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-09T18:37:32.865795Z"},"links":{"citing_paper":"/paper/2605.00963"},"observation_digest":"sha256:68937361fcaaedd4c055d76fcf3add309a73d39539860f6c142fc16de274b180","observation_id":"d5daa156-1f5f-444d-8b37-f23969b1d5f3","resolution":{"observed_at":"2026-05-11T16:11:08.325436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.00963","last_updated":"2026-05-01T15:04:53Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-04T09:50:45.017468Z","submitted_at":"2026-05-01T15:04:53Z","title":"Ablation Study of Multimodal Perception, Language Grounding, and Control for Human-Robot Interaction in an Object Detection and Grasping Task"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":5,"verified_fuzzy":22},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2605.00963."}