{"as_of":"2026-08-17T18:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:02b2eb881f998de5064e2e18d284724e3ff292c381d048f071125d6aaaa74800","coverage":[{"denominator":29,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:53:19.917077Z","state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.20373/citation-record","integrity":"/paper/2506.20373/integrity","json":"/paper/2506.20373/citation-record.json","paper":"/paper/2506.20373"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:19.728747Z","title":"Large language models for human–robot interaction: A review,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.728747Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:8ef65478ef3e411f4aba3f57e51e75e1b8c6521adb2bd69d09c9be8f4e3b64db","observation_id":"522050a9-df17-427e-b02e-1a90539db50f","resolution":{"observed_at":"2026-08-06T22:53:19.728747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.02189","last_updated":"2025-04-06T03:12:51Z","snapshot_observed_at":"2026-08-13T18:28:46.913210Z","submitted_at":"2025-01-04T04:59:33Z","title":"A Survey of State of the Art Large Vision Language Models: Alignment, Benchmark, Evaluations and Challenges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.02189","snapshot_observed_at":"2026-08-06T22:53:19.736138Z","title":"Benchmark Eval- uations, Applications, and Challenges of Large Vision Language Models: A Survey,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.736138Z"},"links":{"cited_paper":"/paper/2501.02189","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:8ed1f915f586fc48613fac7ef6ab7ecde3064561bf054b743322e43441ddc92e","observation_id":"fb58932b-dda3-409d-b3c1-e5a7b9db7f3a","resolution":{"observed_at":"2026-08-06T22:53:19.736138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14320","last_updated":"2023-09-25T17:45:31Z","snapshot_observed_at":"2026-08-16T14:57:46.521268Z","submitted_at":"2023-09-25T17:45:31Z","title":"MUTEX: Learning Unified Policies from Multimodal Task Specifications","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14320","snapshot_observed_at":"2026-08-06T22:53:19.742757Z","title":"MUTEX: Learning Unified Policies from Multimodal Task Specifications,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.742757Z"},"links":{"cited_paper":"/paper/2309.14320","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:4ff45703b224c58597d6e88757bb3752c6847099496c5b33c58ba1c467d96cb4","observation_id":"a67e63f2-50c3-4244-b6a0-ffe57fa8d91f","resolution":{"observed_at":"2026-08-06T22:53:19.742757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:19.750227Z","title":"Vision- language model-driven scene understanding and robotic object manip- ulation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.750227Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:563673b5a6a7e3bc095b621cd4eaede4d5001919904341805a077a33c0575abe","observation_id":"e7e3ad0e-3de1-4113-b5b5-6c1193f37026","resolution":{"observed_at":"2026-08-06T22:53:19.750227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07263","last_updated":"2025-04-24T14:34:46Z","snapshot_observed_at":"2026-08-16T14:53:10.757695Z","submitted_at":"2023-10-11T07:39:42Z","title":"CoPAL: Corrective Planning of Robot Actions with Large Language Models","version":3},"cited_work":{"arxiv_id":"2310.07263","doi":"10.48550/arxiv.2310.07263","metadata_source":"pith","pith_arxiv_id":"2310.07263","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"CoPAL: Corrective Planning of Robot Actions with Large Language Models","venue":"cs.RO","work_id":"6862910b-f811-447d-9aa8-8ec2bdbd246a","year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.770138Z"},"links":{"cited_paper":"/paper/2310.07263","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:4b96f58db2506fe4e3fddb80823b46c346b495d6f3f5847ffaf0b62eef9790d7","observation_id":"f2dd8171-240b-4c6d-965e-dc82a58a169d","resolution":{"observed_at":"2026-08-06T22:53:20.119684Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00210","last_updated":"2024-11-25T21:05:42Z","snapshot_observed_at":"2026-08-16T14:05:04.248988Z","submitted_at":"2024-03-30T01:17:40Z","title":"VLM-Social-Nav: Socially Aware Robot Navigation through Scoring using Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00210","snapshot_observed_at":"2026-08-06T22:53:19.778811Z","title":"VLM-Social-Nav: Socially Aware Robot Navigation through Scoring using Vision-Language Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.778811Z"},"links":{"cited_paper":"/paper/2404.00210","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:e3a3b1227506a229c8ab7a5d06596643b538141fcd26d24d69e6aeb5c1831e0e","observation_id":"7d07ab79-1a41-4486-b734-31d71a195f24","resolution":{"observed_at":"2026-08-06T22:53:19.778811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03275","last_updated":"2023-12-06T04:02:28Z","snapshot_observed_at":"2026-08-17T07:53:38.991127Z","submitted_at":"2023-12-06T04:02:28Z","title":"VLFM: Vision-Language Frontier Maps for Zero-Shot Semantic Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03275","snapshot_observed_at":"2026-08-06T22:53:19.786689Z","title":"VLFM: Vision-Language Frontier Maps for Zero-Shot Seman- tic Navigation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.786689Z"},"links":{"cited_paper":"/paper/2312.03275","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:5311b8cade354384727f97b7c9cb72e8ddda5fc57f6b49da66606788d2898663","observation_id":"baffae8f-7b45-4d4f-8380-a6bb2ed302bc","resolution":{"observed_at":"2026-08-06T22:53:19.786689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.12403","last_updated":"2023-10-13T03:48:11Z","snapshot_observed_at":"2026-08-16T16:50:33.723741Z","submitted_at":"2022-06-24T17:59:02Z","title":"ZSON: Zero-Shot Object-Goal Navigation using Multimodal Goal Embeddings","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.12403","snapshot_observed_at":"2026-08-06T22:53:19.792331Z","title":"ZSON: Zero-Shot Object-Goal Navigation using Multimodal Goal Embeddings,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.792331Z"},"links":{"cited_paper":"/paper/2206.12403","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:480c807f1722ba707e97f104978f518b8abf591d866c6c701a70ef7709680d25","observation_id":"5fcfe77a-5771-4099-aae6-c6b2b638058f","resolution":{"observed_at":"2026-08-06T22:53:19.792331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:19.797931Z","title":"LaMI: Large Language Models for Multi-Modal Human-Robot Interaction,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.797931Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:6994e4614bb40f94d3e713d8dc3a3fd09945a4c05aa5faf0b6b1fde1ffcc12f9","observation_id":"932c51cd-ff69-4386-ae63-34c696512b8d","resolution":{"observed_at":"2026-08-06T22:53:19.797931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:19.802938Z","title":"To Help or Not to Help: LLM-based Attentive Support for Human-Robot Group Interactions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.802938Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:88a93586fd476f2f4c0571c7cb98c4fe7318a123a539314d7ba93278265fe0b9","observation_id":"406d83e2-0e16-451a-aed4-c583e86af768","resolution":{"observed_at":"2026-08-06T22:53:19.802938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2410.08792","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"VLM See, Robot Do: Human Demo Video to Robot Action Plan via Vision Language Model,","venue":"arXiv (Cornell University)","work_id":"9f922180-82ac-415a-a495-953a88209c80","year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.808029Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:10c42e299693ee6d3d5ec780ff5bb41abad9c8176758a1d51b6c3e6d5bfd4f44","observation_id":"ce2a863c-e62b-4d0f-a10e-5e61b1b19f45","resolution":{"observed_at":"2026-08-06T22:53:20.230056Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.13505","last_updated":"2024-10-11T08:58:20Z","snapshot_observed_at":"2026-08-16T13:33:03.713736Z","submitted_at":"2024-07-18T13:38:21Z","title":"Robots Can Multitask Too: Integrating a Memory Architecture and LLMs for Enhanced Cross-Task Robot Action Generation","version":2},"cited_work":{"arxiv_id":"2407.13505","doi":"10.48550/arxiv.2407.13505","metadata_source":"pith","pith_arxiv_id":"2407.13505","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Robots Can Multitask Too: Integrating a Memory Architecture and LLMs for Enhanced Cross-Task Robot Action Generation","venue":"cs.RO","work_id":"914662ba-5a37-4d48-9007-eca017db463b","year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.813106Z"},"links":{"cited_paper":"/paper/2407.13505","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:96b1ebb72f113071f8a294b3078aca81d61b187db975460f34bf857d2c1bcabe","observation_id":"8c86f276-7e64-4c68-be46-d2b17b4e4b8f","resolution":{"observed_at":"2026-08-06T22:53:20.036658Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.924153Z","title":"“Exploring large language models as a source of common-sense knowledge for robots“","venue":null,"work_id":"3dfe2497-6e47-4b47-8416-0439b589c54d","year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.818252Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:e1724dec8293bb3d1011a27065a6c3f746c4a2150e4ec04b8c810da828fd88a4","observation_id":"0cf753cb-4b92-40f3-8b1a-40e2c1f0132d","resolution":{"observed_at":"2026-08-06T22:53:20.929041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.906563Z","title":null,"venue":null,"work_id":"f9e253aa-2313-4ba5-9cfb-aaee6b64ff94","year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.823043Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:0fa55aec6eb24e5c89bdae3be6af8e4605046cc270c66de1c8c996a050323321","observation_id":"85859591-6b94-439b-b285-fd18f478b147","resolution":{"observed_at":"2026-08-06T22:53:20.911871Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.888421Z","title":null,"venue":null,"work_id":"d7f2f4cd-c227-4a09-9718-75819550d940","year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.827857Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:94ac471549bfc76bdd166950d7c976c98d2a8785d9048d75f65ea9450d649b32","observation_id":"0d7d1a81-4741-41e6-8852-3370bab662d9","resolution":{"observed_at":"2026-08-06T22:53:20.893612Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.871591Z","title":"“Quo vadis, action recogni- tion? A new model and the kinetics dataset.“ Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2017","venue":null,"work_id":"ac6b63e9-b1b9-4716-b9af-c29bf3d1a808","year":2017},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.833424Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:f3650717db2de2134279df3a261888ee459b1a066fc9578d479a14628916fc92","observation_id":"e61757eb-223c-4dd2-8d87-f9d87ac45006","resolution":{"observed_at":"2026-08-06T22:53:20.876697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-06T22:53:19.844287Z","title":"Achiam et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.844287Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:c2a929eb69f726b9949ce26a337cdf819439eee2967eba331f200cc1cd197a0d","observation_id":"29d4c07f-d1b3-4335-b8e8-f7e59a6d1aba","resolution":{"observed_at":"2026-08-06T22:53:19.844287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07073","last_updated":"2024-10-10T17:59:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-09T17:16:22Z","title":"Pixtral 12B","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07073","snapshot_observed_at":"2026-08-06T22:53:19.849314Z","title":"Agrawal et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.849314Z"},"links":{"cited_paper":"/paper/2410.07073","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:4c8a3da5e71bf0c27caa7e2930554093080c721c33dab3871b9887ebe79f5b41","observation_id":"ba1ec72c-99c8-49a7-96ef-4e90884b8175","resolution":{"observed_at":"2026-08-06T22:53:19.849314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-08-12T12:01:54.105712Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-06T22:53:19.854918Z","title":"and Li, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.854918Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:02e731ee001716de33e6f0405f5ca6c4f0913d7a54646c0d6530ae647384fdaa","observation_id":"baf6a17e-7231-4b82-9dfd-887a732ae259","resolution":{"observed_at":"2026-08-06T22:53:19.854918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-06T22:53:19.860241Z","title":"“LLaMA: Open and Efficient Foundation Language Models“ arXiv:2302.13971, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.860241Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:93b05bfec418ac723d0ca7561a4443a2da3d4d7dd8083976dcb07f6b8d9dbb8f","observation_id":"6299b6e2-5fb4-475f-a796-8ce59e463f18","resolution":{"observed_at":"2026-08-06T22:53:19.860241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.855490Z","title":"and Kembhavi, A., 2020","venue":null,"work_id":"5088cabb-64dd-4ee8-8bdf-9e86f9efc22a","year":2020},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.865807Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:0a561a6f451f20fe3d42c55e80796b461ebf726d5c2134165c4491edb6d4ae88","observation_id":"7d5f813b-6236-424a-a817-6f9a4bdba0ad","resolution":{"observed_at":"2026-08-06T22:53:20.860482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.838156Z","title":"and Chen, L","venue":null,"work_id":"810279b0-7ce9-4018-bd00-0e68c6e6c222","year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.870783Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:70d06dbd9b4883f2ef4a30ecc61c3ef825893f2420d1dc0db7d5cd4e20153295","observation_id":"05318ce4-1021-43ad-8a6d-e17ca64811f7","resolution":{"observed_at":"2026-08-06T22:53:20.843586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.817857Z","title":"and Saffiotti, A","venue":null,"work_id":"79fc87d3-d311-4aad-8451-cba3bfe9bcc9","year":2003},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.876678Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:57caf503ae47bbfc552f6833d4d439226c689df4a896e19c88148dfe7a14463d","observation_id":"6db1a9ce-b900-41af-8861-827df54ad46e","resolution":{"observed_at":"2026-08-06T22:53:20.823612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.801090Z","title":"and Ros, R","venue":null,"work_id":"e6eb3ec8-4d58-4acc-a90d-6a7bcb0a8c42","year":2011},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.883161Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:1dfd92921cdfe53664f76707223da894b9bf3818e4d7e015d27afe296fd654eb","observation_id":"23e42ec6-a853-4166-87e9-2a1f9b80d6a0","resolution":{"observed_at":"2026-08-06T22:53:20.805943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.784741Z","title":"and Deigmoeller, J","venue":null,"work_id":"b23ce85e-394d-477d-86af-b69038d7735e","year":2019},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.888879Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:97698162057f01f9268f5f80c52c31eb94f8aea2cca6ac75babfe6888cbe8807","observation_id":"40fe6b39-86f5-46bb-9b4d-9ed17c63b2aa","resolution":{"observed_at":"2026-08-06T22:53:20.789418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-06T22:53:19.893891Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.893891Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:be8304c1572d0b7fd2b061faa1f458b673f0d28a0164ca6eb1cd7ba8c6220402","observation_id":"e14c0a4f-93e7-4c41-bc1f-3dc66ed2ff6f","resolution":{"observed_at":"2026-08-06T22:53:19.893891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-06T22:53:19.903289Z","title":"Llava-onevision: Easy visual task transfer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.903289Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:5d57e3c0021abc905b22b135958c4d39532eb051ecac54bc794670a73cf844d3","observation_id":"516b5a2c-9614-4e39-8142-89ae274d8c21","resolution":{"observed_at":"2026-08-06T22:53:19.903289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.767225Z","title":"Vila: On pre-training for visual language models,","venue":null,"work_id":"e685ec60-16e4-4422-9ab1-3447750ff6f0","year":2024},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.909474Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:eee6b29ccb82d397b3e95e98ffe32781ee4b1f4c7a75830f018fda04d49b2566","observation_id":"a38c89f6-c7b7-4bf7-8390-90fe92ccbfe2","resolution":{"observed_at":"2026-08-06T22:53:20.772519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:53:20.749806Z","title":"Kr `‘uger, C","venue":null,"work_id":"1cbcad84-94f6-41e0-af40-ea46177577bb","year":2011},"citing_paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:19.917077Z"},"links":{"citing_paper":"/paper/2506.20373"},"observation_digest":"sha256:97fa9fc8b9ad770e498a774c8f5423731156d7093aeb89ea217cb1cbc3931657","observation_id":"64ac326d-5dcd-4fd6-aa71-ce4ec4b72c73","resolution":{"observed_at":"2026-08-06T22:53:20.754860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.20373","last_updated":"2025-06-25T12:36:49Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-15T21:03:56.378215Z","submitted_at":"2025-06-25T12:36:49Z","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition"},"reference_resolution":{"displayed":29,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":3,"verified_fuzzy":9},"total_outbound_references":29},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 29 of 29 outbound references and 0 inbound Pith citation observations for arXiv:2506.20373."}