{"as_of":"2026-08-15T09:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:10799cbffe701aecc92f9dc80dae0729d53860408804272a25f1eaf558cdbbad","coverage":[{"denominator":150,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T23:53:47.129799Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T16:55:22.682553Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T16:15:49.358084Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.04681","snapshot_observed_at":"2026-08-03T16:55:22.682553Z","title":"Perceiving and acting in first-person: A dataset and benchmark for egocentric human-object-human interac- tions.arXiv preprint arXiv:2508.04681, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.11393","last_updated":"2026-08-13T11:40:44Z","snapshot_observed_at":"2026-08-15T08:14:40.547955Z","submitted_at":"2025-12-12T09:07:21Z","title":"The N-Body Problem: Parallel Execution from Single-Person Egocentric Video","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T16:55:22.682553Z"},"links":{"cited_paper":"/paper/2508.04681","citing_paper":"/paper/2512.11393"},"observation_digest":"sha256:b1f407626a99a4fbaf81fee6f4f18ea9f02d872f1e0a5f284640c3cd3a008fb6","observation_id":"d714f7d3-0497-4b75-a585-5fd7ff7201f3","resolution":{"observed_at":"2026-08-03T16:55:22.682553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"cited_work":{"arxiv_id":"2508.04681","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.04681","snapshot_observed_at":"2026-07-01T16:15:49.358084Z","title":"Perceiving and acting in first-person: A dataset and benchmark for ego- centric human-object-human interactions.arXiv preprint arXiv:2508.04681, 2025","venue":null,"work_id":"b2532e91-1bb1-4d14-87f4-11e45e4a4117","year":2025},"citing_paper":{"arxiv_id":"2606.28604","last_updated":"2026-06-26T20:52:54Z","snapshot_observed_at":"2026-08-06T06:50:02.421167Z","submitted_at":"2026-06-26T20:52:54Z","title":"IMU-HOI: A Symbiotic Framework for Coherent Human-Object Interaction and Motion Capture via Contact-Conscious Inertial Fusion","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-30T00:50:08.637598Z"},"links":{"cited_paper":"/paper/2508.04681","citing_paper":"/paper/2606.28604"},"observation_digest":"sha256:e701496e7f08c9135cd1a8ad709414fec6c88c13b820aa8194fdf0063c7c9c96","observation_id":"d1c5ad6a-301d-4350-87c4-3972e9f07fb9","resolution":{"observed_at":"2026-07-01T16:15:49.359573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2508.04681/citation-record","integrity":"/paper/2508.04681/integrity","json":"/paper/2508.04681/citation-record.json","paper":"/paper/2508.04681"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.377873Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.377873Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:51ed736990c788125a9a8d63a20845f9b0b3a536c2bc8431847a36374ffc27ea","observation_id":"aa1de90f-0463-4f3c-987c-a75187a329bd","resolution":{"observed_at":"2026-08-05T23:53:36.377873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.493641Z","title":"https://eth-ait.github.io/aitviewer/","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.493641Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9fea68d332b543d4319f9a6abe186de89aa2e3cf7bf4f6e8fb55f4d26d314b2e","observation_id":"c2267eaa-8907-4e8b-a8f9-b145b8535e9d","resolution":{"observed_at":"2026-08-05T23:53:36.493641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.580669Z","title":"https://github.com/zju3dv/EasyMocap","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.580669Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:842f903392278188797ba0bbfa8fb219ca71eea4d9ae0162ba203387fdf34b3c","observation_id":"637fe4ef-1ad5-416f-835d-91cc9aee4fdd","resolution":{"observed_at":"2026-08-05T23:53:36.580669Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.644157Z","title":"https://github.com/opencv/opencv/blob/master/data/haarcascades/haarcascade_frontalface_default.xml","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.644157Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:1142d2173aa18003bee5ac020eb52d7152daafe50a2f84e476f9145d6bb76fc6","observation_id":"53604ed2-fd65-4f0c-b0d0-dfec27dc6b28","resolution":{"observed_at":"2026-08-05T23:53:36.644157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.677317Z","title":"3d human pose perception from egocentric stereo videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.677317Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:1859b35b224f9e3dc8ef9a117af55551ccff7d04eb7824304b453ec2e3f3528b","observation_id":"3afd4ab0-2388-4769-a490-b969f4fce32f","resolution":{"observed_at":"2026-08-05T23:53:36.677317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.791573Z","title":"Rgbmanip: Monocular image-based robotic manipulation through active object pose estimation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.791573Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:da01d6db53f278f4f5d4a8a95866c7faedf631684beb3a5f42feb36426358710","observation_id":"7c0eb52a-6ee3-477f-a726-e254bf10104a","resolution":{"observed_at":"2026-08-05T23:53:36.791573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.912396Z","title":"Circle: Capture in rich contextual environments","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.912396Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:fd215126438607be387396b08a5c9a7b5cb9101128b7e73d4ce0241ec6a0dc58","observation_id":"df6e3a0a-faba-4978-b8c6-1b3dd8a1e718","resolution":{"observed_at":"2026-08-05T23:53:36.912396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09598","last_updated":"2024-06-13T21:38:17Z","snapshot_observed_at":"2026-08-14T02:46:34.024598Z","submitted_at":"2024-06-13T21:38:17Z","title":"Introducing HOT3D: An Egocentric Dataset for 3D Hand and Object Tracking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09598","snapshot_observed_at":"2026-08-05T23:53:36.990477Z","title":"Introducing hot3d: An egocentric dataset for 3d hand and object tracking","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.990477Z"},"links":{"cited_paper":"/paper/2406.09598","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3b46034cca7b5433bcadffa34685a0b775edfb8ec983cc94d344acd7392c5098","observation_id":"7f94ed07-29b9-48c9-9422-497b4960be4b","resolution":{"observed_at":"2026-08-05T23:53:36.990477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.078642Z","title":"Uncertainty-aware state space transformer for egocentric 3d hand trajectory forecasting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.078642Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:6702f1abcf047a7ca362306c575f11210467b48f51603b8f6cdb28727bc8507d","observation_id":"c3be9f17-9051-40ea-98b3-2b3c435f46f6","resolution":{"observed_at":"2026-08-05T23:53:37.078642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.182153Z","title":"Gen2act: Human video generation in novel scenarios enables generalizable robot manipulation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.182153Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:cf6dd499b9decdd7b1852c859c25e819d51b991535c76cea52266809397ed2d2","observation_id":"7199a25e-d890-4589-ad57-fea792773787","resolution":{"observed_at":"2026-08-05T23:53:37.182153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.277795Z","title":"Behave: Dataset and method for tracking human object interactions","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.277795Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9a1d801037441e82e2163c8c45d86f5b35476599ba920856b6058403f4740d57","observation_id":"741097dd-6161-418c-a120-9e1924c9d250","resolution":{"observed_at":"2026-08-05T23:53:37.277795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.330804Z","title":"Bedlam: A synthetic dataset of bodies exhibiting detailed lifelike animated motion","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.330804Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:4724198067546f8d342d053aba775b3743db09feb79d2cee6d720b31206d28de","observation_id":"fc5b2d54-9aa7-446d-8358-ee8b86c909dc","resolution":{"observed_at":"2026-08-05T23:53:37.330804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.408257Z","title":"Contactpose: A dataset of grasps with object contact and hand pose","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.408257Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:102bbb5b9ec0b22f232ea819bd7f88c8ff6cb4cf7d960e4cab3e58c719c01988","observation_id":"3c868352-74a1-441c-9d95-d5f2e52f6a37","resolution":{"observed_at":"2026-08-05T23:53:37.408257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-05T23:53:37.497994Z","title":"Rt-1: Robotics transformer for real-world control at scale","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.497994Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c4fd8a3da9df80ec1888dede0c248016fdfea723b10f598d3327d54c8bd47f7e","observation_id":"23ea0b4e-5697-4c39-b686-95bcc92198fe","resolution":{"observed_at":"2026-08-05T23:53:37.497994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-05T23:53:37.605325Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.605325Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e269341eae5e4a78c4f105fa6407bb25aea7cef7cedc15e3161f0ad48f904f52","observation_id":"63beb163-f63b-4281-81d6-8cbe59e62bb1","resolution":{"observed_at":"2026-08-05T23:53:37.605325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.748118Z","title":"u tepage, Ali Ghadirzadeh, \\","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.748118Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0959f90d4178cf99b61863fedfdea3cd73d081757f21ae351566fb3d16375d2e","observation_id":"c0f2382a-e4e0-427a-a6fc-21c90bf34ea7","resolution":{"observed_at":"2026-08-05T23:53:37.748118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.811974Z","title":"Understanding hand-object manipulation with grasp types and object attributes","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.811974Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:bd412cdaede0804110565cbdfd529c2218f7aa6c381530c625125794dfb384af","observation_id":"b738690b-82b7-4577-aa62-669705241df2","resolution":{"observed_at":"2026-08-05T23:53:37.811974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.924543Z","title":"Long-term human motion prediction with scene context","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.924543Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:391f9fcfe1bae962a3f1a7ed3a3349b318c407cebe2d6d771e989fb660a01f92","observation_id":"67efd82f-7bf1-43b7-aec3-34a8ef337e8a","resolution":{"observed_at":"2026-08-05T23:53:37.924543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.044463Z","title":"A multi-sensor dataset of human-human handover","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.044463Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c95ca70622b129ab6145294875fa4019478aa5f3389b58bab9421474160a8f12","observation_id":"4528d23f-6faf-4075-b840-84a1f362cc51","resolution":{"observed_at":"2026-08-05T23:53:38.044463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.135981Z","title":"On the choice of grasp type and location when handing over an object","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.135981Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d3d4b41a0fe17728e6cf9387523e8c13c2cd8b512844fab4e916f2988729aff1","observation_id":"0dd07cb0-67c7-4e6a-a318-1aff0c879f18","resolution":{"observed_at":"2026-08-05T23:53:38.135981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.221352Z","title":"Context-aware human motion prediction, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.221352Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c4671fd6f1de27695b7370b9ec67019a091c4c984e6f66a365d1826b85705f58","observation_id":"b97bbdaa-52e9-4d73-9327-38f0ffcff497","resolution":{"observed_at":"2026-08-05T23:53:38.221352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.359155Z","title":"Towards collaborative robots as intelligent co-workers in human-robot joint tasks: what to do and who does it? In ISR 2020; 52th International Symposium on Robotics, pages 1--8","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.359155Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d3d793547800dd7608ba39c8a10bd7f33937874a6dffa9fed4932e8278b0e289","observation_id":"acd28e02-93ff-42f6-9bb9-009146e79861","resolution":{"observed_at":"2026-08-05T23:53:38.359155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.499590Z","title":"You-do, i-learn: Egocentric unsupervised discovery of objects and their modes of interaction towards video-based guidance","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.499590Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:962d4ce835a1233af74d931a4c8f7d9fd45b993b32801df0a003c319d55d61ac","observation_id":"4e870b8a-891a-4f0c-8d30-7d70ab1c02c1","resolution":{"observed_at":"2026-08-05T23:53:38.499590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.632786Z","title":"Scaling egocentric vision: The epic-kitchens dataset","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.632786Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ae4d2ccf89f0cb19f0d361b0ccad02770259cdeecfd4b777f8fe282787a2199d","observation_id":"8aa3846d-1915-4897-9826-96ed3c690a18","resolution":{"observed_at":"2026-08-05T23:53:38.632786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.729105Z","title":"Rescaling egocentric vision: Collection, pipeline and challenges for epic-kitchens-100","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.729105Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:bcbfe841e59dca150bc85fc7110d83ff85760ecd958986b040b7b5e4976b52c5","observation_id":"2b403280-8c96-4083-aab9-05cadc26cf86","resolution":{"observed_at":"2026-08-05T23:53:38.729105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16097","last_updated":"2024-05-17T15:00:55Z","snapshot_observed_at":"2026-08-13T05:16:06.103951Z","submitted_at":"2023-11-27T18:59:10Z","title":"CG-HOI: Contact-Guided 3D Human-Object Interaction Generation","version":2},"cited_work":{"arxiv_id":"2311.16097","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.16097","snapshot_observed_at":"2026-08-05T23:53:56.465466Z","title":"CG-HOI: Contact-Guided 3D Human-Object Interaction Generation","venue":"cs.CV","work_id":"dd3c5a9f-03d9-4f06-9d4c-45908ff4d9f0","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.839860Z"},"links":{"cited_paper":"/paper/2311.16097","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a7b8174be50d6a428b83f54ba9dc109023625f426e8dcdf3466a0fa098b87cea","observation_id":"7cc1fa15-4d4e-400c-aaab-c35637204c28","resolution":{"observed_at":"2026-08-05T23:53:56.623103Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03378","last_updated":"2023-03-06T18:58:06Z","snapshot_observed_at":"2026-08-14T18:47:26.721223Z","submitted_at":"2023-03-06T18:58:06Z","title":"PaLM-E: An Embodied Multimodal Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03378","snapshot_observed_at":"2026-08-05T23:53:38.955903Z","title":"Palm-e: An embodied multimodal language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.955903Z"},"links":{"cited_paper":"/paper/2303.03378","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3f032c152d5bfce9ef51afad299530af69ef9817c924506538611d578f68b573","observation_id":"443523c8-f5f6-4272-96b1-ba82e3f4812c","resolution":{"observed_at":"2026-08-05T23:53:38.955903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.065926Z","title":"Avatars grow legs: Generating smooth human motion from sparse tracking inputs with diffusion model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.065926Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:f2424e1308f23a9e5f90a46acf411c51d5407d099e5cc4167002427999f1d0cf","observation_id":"dcf469ee-718a-41ce-b577-8169e806610d","resolution":{"observed_at":"2026-08-05T23:53:39.065926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.194833Z","title":"Human preferences for robot eye gaze in human-to-robot handovers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.194833Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:39f3eb3303f66ab883dc595510e1d9a1a21bbac409beefbb5d7b655ce05bbd29","observation_id":"3621407e-7517-41e5-af35-72f97fc86ba3","resolution":{"observed_at":"2026-08-05T23:53:39.194833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.07095","last_updated":"2025-07-09T17:52:04Z","snapshot_observed_at":"2026-08-14T03:33:39.678895Z","submitted_at":"2025-07-09T17:52:04Z","title":"Go to Zero: Towards Zero-shot Motion Generation with Million-scale Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.07095","snapshot_observed_at":"2026-08-05T23:53:39.305639Z","title":"Go to zero: Towards zero-shot motion generation with million-scale data","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.305639Z"},"links":{"cited_paper":"/paper/2507.07095","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:1e8a61a87c7c5f0c8dce231896c25debdfbf6d8b1fe9b6740b84529089c74aff","observation_id":"096698dc-b191-4aaf-9133-c0261f37b320","resolution":{"observed_at":"2026-08-05T23:53:39.305639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.444849Z","title":"Arctic: A dataset for dexterous bimanual hand-object manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.444849Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:5e5b89b5bbd6fa3ccd96400a6b6f268de2dc466b52438a5695022dd389d9b7e6","observation_id":"07301309-3c82-42da-b6e6-9914b5ac4b6a","resolution":{"observed_at":"2026-08-05T23:53:39.444849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.587309Z","title":"Hold: Category-agnostic 3d reconstruction of interacting hands and objects from video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.587309Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e764773877ba9b8d2784327b65e783d67cbfafd3aa9b5acc9c51bd8816da9789","observation_id":"dee6ca9c-d63a-40b4-9876-41f032fbe3a6","resolution":{"observed_at":"2026-08-05T23:53:39.587309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.790354Z","title":"Benchmarks and challenges in pose estimation for egocentric hand interactions with objects","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.790354Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:eecdec541097c0cb38af2b8292578b660573d0ce384ae701d99e15d278ee68e8","observation_id":"7ae5b3cf-4111-4bbf-a0fe-16e696c5106b","resolution":{"observed_at":"2026-08-05T23:53:39.790354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.864859Z","title":"Social interactions: A first-person perspective","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.864859Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:2f67be42b43c10887e9f01061cabd6ab90e9d52eab8dc4920279d4251fbf10e8","observation_id":"6335ad8c-60e1-43dd-8fbb-2aa0a2c81dcf","resolution":{"observed_at":"2026-08-05T23:53:39.864859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.991773Z","title":"Three-dimensional reconstruction of human interactions","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.991773Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:56e5887e247a04dffadfb0e2e2d65cbd1bbfa9a91623ccc78a08ef6f6c8737f7","observation_id":"0eb12c38-8be0-483c-b62a-d2a8f1ca5486","resolution":{"observed_at":"2026-08-05T23:53:39.991773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.138119Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.138119Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:49c4dce56a3e9e3abfcb0beb3d79730d430ef22c6668d365382e2a9460188ebd","observation_id":"e122517d-adc9-4f43-a2fb-ee0c6eb16bb6","resolution":{"observed_at":"2026-08-05T23:53:40.138119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.326420Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.326420Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ade8dc2bc3d05d7233a444ea4eda3172eade30724ecc5c8c3a4c0c1deacab01f","observation_id":"214b7387-8c1f-40c9-a230-a1df3f6076c2","resolution":{"observed_at":"2026-08-05T23:53:40.326420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.453469Z","title":"Generating diverse and natural 3d human motions from text","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.453469Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:1978ce9dd9477de8c34dddd0cf1b1106edea4c866c930b99a41c85713a078cf7","observation_id":"c801ee8b-b65b-42f0-9bb3-28c76597692a","resolution":{"observed_at":"2026-08-05T23:53:40.453469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.557001Z","title":"Multi-person extreme motion prediction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.557001Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9db66c43552e5edb397abec121c89b6b25c5cdad46c4d8c89f2a94915115cbb3","observation_id":"e7441a50-ee9e-476d-ba80-d6d19eb6cf4c","resolution":{"observed_at":"2026-08-05T23:53:40.557001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.686731Z","title":"Interaction replica: Tracking human--object interaction and scene changes from human motion","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.686731Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d878c9a0804b0fa3cc9e7cb2505e567ec023ecef03169b20536d25cc044463e8","observation_id":"a9369fba-bce0-4115-8971-d5d5a7a3f543","resolution":{"observed_at":"2026-08-05T23:53:40.686731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.802399Z","title":"Honnotate: A method for 3d annotation of hand and object poses","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.802399Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e8813de5833386ba6bef592b9bd7298489d51e018ec6a76e247599e6e624ec08","observation_id":"86d875fc-d8c8-4c6d-9cbb-3e8d77a093fc","resolution":{"observed_at":"2026-08-05T23:53:40.802399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.944657Z","title":"Keypoint transformer: Solving joint identification in challenging hands and object interactions for accurate 3d pose estimation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.944657Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9b9d8e68cab920ba75c2fba9e62a4c307be476391253d434614143fcfef11059","observation_id":"fdf4964c-ff11-4cba-8b00-09b2d2f91658","resolution":{"observed_at":"2026-08-05T23:53:40.944657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08530","last_updated":"2024-10-11T05:02:31Z","snapshot_observed_at":"2026-08-12T22:25:44.944087Z","submitted_at":"2024-10-11T05:02:31Z","title":"Ego3DT: Tracking Every 3D Object in Ego-centric Videos","version":1},"cited_work":{"arxiv_id":"2410.08530","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.08530","snapshot_observed_at":"2026-08-05T23:53:56.111211Z","title":"Ego3DT: Tracking Every 3D Object in Ego-centric Videos","venue":"cs.CV","work_id":"2a6cce85-0162-4af3-92fd-d67d43e94f6e","year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.136974Z"},"links":{"cited_paper":"/paper/2410.08530","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:bfb920ea48379b9661bb0a531ee715fdc2a8a9da94df96a7d78ae2d2c21eab31","observation_id":"0ff40bc8-b9fe-42e7-adba-4c6dc91f5fa9","resolution":{"observed_at":"2026-08-05T23:53:56.245318Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:41.282187Z","title":"Resolving 3d human pose ambiguities with 3d scene constraints","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.282187Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:beadf10ef715a40332c7100852d26e821910cac62e6b2d7f2bf609e763955564","observation_id":"e5f1267f-bf0e-405a-8a15-8108ec6c2095","resolution":{"observed_at":"2026-08-05T23:53:41.282187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:41.447218Z","title":"Stochastic scene-aware motion prediction","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.447218Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:5b904eb0f575f5a91637a3c1f37174d235ef8372c2f1b10bae7e1a094a35bd0f","observation_id":"15d62b75-0a72-4770-a012-7f12306552ba","resolution":{"observed_at":"2026-08-05T23:53:41.447218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08858","last_updated":"2024-06-13T06:44:46Z","snapshot_observed_at":"2026-08-12T23:43:53.764034Z","submitted_at":"2024-06-13T06:44:46Z","title":"OmniH2O: Universal and Dexterous Human-to-Humanoid Whole-Body Teleoperation and Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08858","snapshot_observed_at":"2026-08-05T23:53:41.596866Z","title":"Omnih2o: Universal and dexterous human-to-humanoid whole-body teleoperation and learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.596866Z"},"links":{"cited_paper":"/paper/2406.08858","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:23006925daa7dfad2431939fb55744b8518f886df3b615a51a3df01c38f3f94e","observation_id":"0582ff3c-d9bc-45c1-a624-24c9cc183149","resolution":{"observed_at":"2026-08-05T23:53:41.596866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:41.794282Z","title":"Learning human-to-humanoid real-time whole-body teleoperation, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.794282Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:2c13cb8666bb0003947bb47a58339f442ae0e0506584965833b876e3e2f036f8","observation_id":"4b32be66-0bfe-4137-ac4b-c8c75fbb0f73","resolution":{"observed_at":"2026-08-05T23:53:41.794282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.007046Z","title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.007046Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d48b93b075e6949cd2911aa8ed919833271dc753ed488dc8202b89f993f214ea","observation_id":"2856fd24-9f52-4d4c-9065-a998b32df01a","resolution":{"observed_at":"2026-08-05T23:53:42.007046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.05973","last_updated":"2023-11-02T06:53:37Z","snapshot_observed_at":"2026-08-05T01:03:23.456778Z","submitted_at":"2023-07-12T07:40:48Z","title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.05973","snapshot_observed_at":"2026-08-05T23:53:42.151513Z","title":"Voxposer: Composable 3d value maps for robotic manipulation with language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.151513Z"},"links":{"cited_paper":"/paper/2307.05973","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a74b02cdfdf03ab9554ce96f07e353633faa850c464fb0af11a82b1bc4a98d5f","observation_id":"4311e1c5-9ae5-4490-b670-d5993dfb874a","resolution":{"observed_at":"2026-08-05T23:53:42.151513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.262006Z","title":"Intercap: Joint markerless 3d tracking of humans and objects in interaction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.262006Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:6ead843703089e02b46abdd4298c71215947150b324b76062e0553aff3ae600b","observation_id":"f19ee7a2-a298-47de-920a-c60f261c0c49","resolution":{"observed_at":"2026-08-05T23:53:42.262006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.305300Z","title":"Human3.6m: Large scale datasets and predictive methods for 3d human sensing in natural environments","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.305300Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:82bcbdfd867460cad9bd87e9a14d8ff1ca43e6f2bef2dfe325e4c381ce4435d3","observation_id":"e45659d0-a564-4095-95ac-982b45f4d7ed","resolution":{"observed_at":"2026-08-05T23:53:42.305300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.493342Z","title":"A large-scale rgb-d database for arbitrary-view human action recognition","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.493342Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:15734f520ee0027e07f522e6c1f7cb624ecba2a4e3ce943c339fa33b3afd382a","observation_id":"540747ee-8105-4d24-a3c2-00fd13e7c80c","resolution":{"observed_at":"2026-08-05T23:53:42.493342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.668418Z","title":"Affordpose: A large-scale dataset of hand-object interactions with affordance-driven hand pose","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.668418Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9dd47d456c972fed1a6b7d7298774469fbefdbc60d5488b7ddaff037f02c07ba","observation_id":"4a821793-7172-4353-8844-4f6f6468a9ca","resolution":{"observed_at":"2026-08-05T23:53:42.668418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.751703Z","title":"Avatarposer: Articulated full-body pose tracking from sparse motion sensing","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.751703Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e8c2354fa3bce5c11c853d69056dbb956859424ba5a0417b2b06ebdb10cec08d","observation_id":"c29d00bc-a3cc-4a5b-ba66-8aedf8ef1dd0","resolution":{"observed_at":"2026-08-05T23:53:42.751703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06493","last_updated":"2024-09-06T11:28:04Z","snapshot_observed_at":"2026-08-13T10:35:39.935857Z","submitted_at":"2023-08-12T07:46:50Z","title":"EgoPoser: Robust Real-Time Egocentric Pose Estimation from Sparse and Intermittent Observations Everywhere","version":3},"cited_work":{"arxiv_id":"2308.06493","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.06493","snapshot_observed_at":"2026-08-05T23:53:55.751984Z","title":"EgoPoser: Robust Real-Time Egocentric Pose Estimation from Sparse and Intermittent Observations Everywhere","venue":"cs.CV","work_id":"96b98d4b-cbd6-485e-aff2-dd06d1130a48","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.826886Z"},"links":{"cited_paper":"/paper/2308.06493","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:5e3ecc7e8371f7e87538c9b5ad116d5c1b561ce23ce6b833e0c2100a93459ffd","observation_id":"02026fc3-0c18-47a5-ba10-343bd0d8cf6c","resolution":{"observed_at":"2026-08-05T23:53:55.933190Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.954752Z","title":"Full-body articulated human-object interaction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.954752Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:5a94eb21c6c8706060303ad48b48cc55c23b15b6cb4d4f70d37fe7223aeff862","observation_id":"db12ccd5-e4f1-4f79-84d4-a32bc36b14ab","resolution":{"observed_at":"2026-08-05T23:53:42.954752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.115113Z","title":"Scaling up dynamic human-scene interaction modeling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.115113Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:4d47a1854c72f15167b01cdc1a335e13660f26e2b7ed01c7d81b75a0b4f73898","observation_id":"04876618-26f0-4a7f-8572-fd8d25a1593f","resolution":{"observed_at":"2026-08-05T23:53:43.115113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.225202Z","title":"A probabilistic attention model with occlusion-aware texture regression for 3d hand reconstruction from a single rgb image","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.225202Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:49aeb39ccb68046d765eb5f78db8f6c1aec90e4e778aa033348414cb71fc4ec3","observation_id":"934c652c-9301-4a15-83ee-91d1737144ce","resolution":{"observed_at":"2026-08-05T23:53:43.225202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12773","last_updated":"2024-10-16T17:48:50Z","snapshot_observed_at":"2026-08-12T22:21:39.180871Z","submitted_at":"2024-10-16T17:48:50Z","title":"Harmon: Whole-Body Motion Generation of Humanoid Robots from Language Descriptions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12773","snapshot_observed_at":"2026-08-05T23:53:43.249143Z","title":"Harmon: Whole-body motion generation of humanoid robots from language descriptions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.249143Z"},"links":{"cited_paper":"/paper/2410.12773","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:24667a25d8e57a31f48f7e3cdcbb05eab42a9dbaa7bb1dfc7d7a91decb55fbb9","observation_id":"272642f2-06d5-4d6c-bd35-960ce43b776d","resolution":{"observed_at":"2026-08-05T23:53:43.249143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.317534Z","title":"Epic-fusion: Audio-visual temporal binding for egocentric action recognition","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.317534Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0b9c23acf4f8f5c7392c075a589be2b47c828c7218e84be10ba7e5392d2ca6ef","observation_id":"221c6dde-b73b-41db-a401-66a69b69baf1","resolution":{"observed_at":"2026-08-05T23:53:43.317534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.13851","last_updated":"2020-11-27T17:29:48Z","snapshot_observed_at":"2026-08-08T17:41:25.099143Z","submitted_at":"2020-11-27T17:29:48Z","title":"Real-time Active Vision for a Humanoid Soccer Robot Using Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2011.13851","doi":null,"metadata_source":"pith","pith_arxiv_id":"2011.13851","snapshot_observed_at":"2026-08-05T23:53:55.456916Z","title":"Real-time Active Vision for a Humanoid Soccer Robot Using Deep Reinforcement Learning","venue":"cs.RO","work_id":"1a884a38-b66d-487a-a572-b0f901193c2b","year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.434526Z"},"links":{"cited_paper":"/paper/2011.13851","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9a5c39dcd4d90448461724d05afb1b5822f7b1b42db203eec26b1a2927960315","observation_id":"5d01fd9d-6d53-4e3d-bb7c-5a4f07b5556f","resolution":{"observed_at":"2026-08-05T23:53:55.571899Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-08-15T04:55:15.159462Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-05T23:53:43.518630Z","title":"Openvla: An open-source vision-language-action model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.518630Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:2604fcbebb45a14be05ca60d1727fd96203cd020653c3e0916f02ff4f31770e3","observation_id":"d38ec4df-1d52-4a30-bbdf-9b42b17424f1","resolution":{"observed_at":"2026-08-05T23:53:43.518630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.577785Z","title":"Dataset of bimanual human-to-human object handovers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.577785Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c5023c3294c1aa7ef792ca81366c3484de8015676bcd1f7804fc824000d42919","observation_id":"4bad86b4-1f7a-4b57-91c4-76c86cf57f97","resolution":{"observed_at":"2026-08-05T23:53:43.577785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.620869Z","title":"H2o: Two hands manipulating objects for first person interaction recognition","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.620869Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:55d3a9649359b2acf1dac563b9249a8e905cb8a447245eceb257ab79915b3b9f","observation_id":"74d306ed-8824-4528-8da0-04a41b9ab49a","resolution":{"observed_at":"2026-08-05T23:53:43.620869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.657151Z","title":"Ego-body pose estimation via ego-head pose estimation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.657151Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ee98dee0d2a7f163f662df836c680cdb203f19dcd31d73af8543ff6e94dc7c2b","observation_id":"1634a6f9-e40d-459d-8a68-83d1b510bbac","resolution":{"observed_at":"2026-08-05T23:53:43.657151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.710565Z","title":"Dngaussian: Optimizing sparse-view 3d gaussian radiance fields with global-local depth normalization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.710565Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d3fd0d19aa0d2cb408a57334aa159efd665ec8f7b1852f36131736cb505f421f","observation_id":"6ac6dd36-1ea9-4af2-94f3-aed9b7e4c3ea","resolution":{"observed_at":"2026-08-05T23:53:43.710565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.836230Z","title":"In the eye of beholder: Joint learning of gaze and actions in first person video","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.836230Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:f2f0cfb09bdd987319de520802f1fff2de8a5db794558bcb35e325450488cc62","observation_id":"a2cd7b7f-d1c1-4a31-8ee7-66062f9d93ef","resolution":{"observed_at":"2026-08-05T23:53:43.836230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.872711Z","title":"Ego-exo: Transferring visual representations from third-person to first-person videos","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.872711Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:eb9b0727489d2dda8e63d877cda2b60b3ef027df006ae7fa2883050aeb984290","observation_id":"85dd708c-dd21-4f9e-b8b9-e4a1c9408867","resolution":{"observed_at":"2026-08-05T23:53:43.872711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05684","last_updated":"2024-03-28T03:15:57Z","snapshot_observed_at":"2026-08-13T12:04:28.830591Z","submitted_at":"2023-04-12T08:12:29Z","title":"InterGen: Diffusion-based Multi-human Motion Generation under Complex Interactions","version":3},"cited_work":{"arxiv_id":"2304.05684","doi":null,"metadata_source":"pith","pith_arxiv_id":"2304.05684","snapshot_observed_at":"2026-08-05T23:53:55.224988Z","title":"InterGen: Diffusion-based Multi-human Motion Generation under Complex Interactions","venue":"cs.CV","work_id":"57f7c78c-8069-470d-9122-c2febdbd4766","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.001874Z"},"links":{"cited_paper":"/paper/2304.05684","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:10c2c43ef755c6344d6261eda6d90a3b60dbd167d3c36f342a7f92e3ff775043","observation_id":"ed506fd7-27ef-4549-bd59-405a3d77fc9e","resolution":{"observed_at":"2026-08-05T23:53:55.311121Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.00818","last_updated":"2024-01-26T15:40:29Z","snapshot_observed_at":"2026-08-14T07:06:48.333154Z","submitted_at":"2023-07-03T07:57:29Z","title":"Motion-X: A Large-scale 3D Expressive Whole-body Human Motion Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.00818","snapshot_observed_at":"2026-08-05T23:53:44.127161Z","title":"Motion-x: A large-scale 3d expressive whole-body human motion dataset","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.127161Z"},"links":{"cited_paper":"/paper/2307.00818","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:089450e854deef050fded49f3dd9dd4bbfb0e3a62dc2f327f226544aac7fa121","observation_id":"1303fdd2-dda5-47b3-ba6b-42691f636be3","resolution":{"observed_at":"2026-08-05T23:53:44.127161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.251374Z","title":"Gaussian-flow: 4d reconstruction with dynamic 3d gaussian particle","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.251374Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:fb60a819c19988bce7120f9bce3ddfd0dbe12d5bbcabcabc42f92536a74d7861","observation_id":"ef18c462-ed8c-4f1f-b2e2-19f809f7258a","resolution":{"observed_at":"2026-08-05T23:53:44.251374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.356535Z","title":"Ntu rgb+ d 120: A large-scale benchmark for 3d human activity understanding","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.356535Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:acbe3b163c3b926a3c097cb610e849e8d1c40016959ab7a2b90f138e1cba94de","observation_id":"f54d7c30-e9b5-454c-ae4b-de9ae1f6a908","resolution":{"observed_at":"2026-08-05T23:53:44.356535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.444765Z","title":"Forecasting human-object interaction: joint prediction of motor attention and actions in first person video","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.444765Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e0bf20b0487fad04f320b524a87df8b1992229845d3034870b30598bed144554","observation_id":"fd84285f-2b75-4da0-9fbf-dbea7c596e92","resolution":{"observed_at":"2026-08-05T23:53:44.444765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.519068Z","title":"Joint hand motion and interaction hotspots prediction from egocentric videos","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.519068Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:712f695bb055bd5a48f885880863211b375e0082dcd0f9050f550c4f2ae2fa7c","observation_id":"5c933f72-f545-4062-8e53-7df5675bf231","resolution":{"observed_at":"2026-08-05T23:53:44.519068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.539091Z","title":"Hoi4d: A 4d egocentric dataset for category-level human-object interaction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.539091Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:174ff1c6a5d3564f186983c790814f8243904354c983c777c1b4b6aff066bc71","observation_id":"3ca0b4e4-52a1-4498-8e77-195cdf414d08","resolution":{"observed_at":"2026-08-05T23:53:44.539091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.630172Z","title":"Taco: Benchmarking generalizable bimanual tool-action-object understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.630172Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ba20fea738b9e8178c27c8393a7d70ad7ff1460d517b5bd93c51a2b2a7ef1fa0","observation_id":"f4b8d642-93a6-492e-acf9-d1e4b928a82f","resolution":{"observed_at":"2026-08-05T23:53:44.630172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.744966Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.744966Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:6dc339ab4a199a5ecff8b2015e128acc14a5c6ba7087df812868e3498babe3ae","observation_id":"3781f457-10ca-4a54-be07-f57c3090e0ef","resolution":{"observed_at":"2026-08-05T23:53:44.744966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.817529Z","title":"Dynamics-regulated kinematic policy for egocentric pose estimation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.817529Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:10146badd19b8e70be99f413429ac774a78bd7a269a5b443349b6ca8c9fb76c4","observation_id":"040ac900-4abe-475f-85a7-3216bd152279","resolution":{"observed_at":"2026-08-05T23:53:44.817529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.945085Z","title":"Himo: A new benchmark for full-body human interacting with multiple objects","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.945085Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:351bc2d5472317170424bd6f94f24dba4264365f03c5c5f971f5e41e90d1ad7a","observation_id":"02652ef5-4919-4481-b2ee-5d9781c0527b","resolution":{"observed_at":"2026-08-05T23:53:44.945085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.081764Z","title":"Diff-ip2d: Diffusion-based hand-object interaction prediction on egocentric videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.081764Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9334f428a8a2a42c08f7a54b15e6ff8f4c343053286a328994da6a0442d8c549","observation_id":"cf2c0312-2946-4b8f-864e-5076d3ef0c92","resolution":{"observed_at":"2026-08-05T23:53:45.081764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14093","last_updated":"2026-05-01T01:50:44Z","snapshot_observed_at":"2026-08-04T06:47:25.827167Z","submitted_at":"2024-05-23T01:43:54Z","title":"A Survey on Vision-Language-Action Models for Embodied AI","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14093","snapshot_observed_at":"2026-08-05T23:53:45.175736Z","title":"A survey on vision-language-action models for embodied ai","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.175736Z"},"links":{"cited_paper":"/paper/2405.14093","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ffa716b9911a4717a07713ad51588f75af738c552f704b480d7aa2cd36283731","observation_id":"0a503427-8a8a-415e-8d53-5bb314c41e3d","resolution":{"observed_at":"2026-08-05T23:53:45.175736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.259308Z","title":"Unifying representations and large-scale whole-body motion databases for studying human motion","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.259308Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:dc7439ec4d9b084e4b5ff72b371f121bdc02bfe59dc88bebc2d2f585edcceaa6","observation_id":"a7a7c949-41e9-4dce-8267-c6515b83c5f9","resolution":{"observed_at":"2026-08-05T23:53:45.259308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.369610Z","title":"Dexvip: Learning dexterous grasping with human hand pose priors from video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.369610Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:60c14c96b40aec6d92b20aeed00e3af0bb1295dc4942bfb4bd1b57f6f8840c0a","observation_id":"59975bdf-f0ad-41d6-a98b-55029605412f","resolution":{"observed_at":"2026-08-05T23:53:45.369610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.453729Z","title":"Hoi4abot: Human-object interaction anticipation for human intention reading collaborative robots, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.453729Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:934db09f4f42474d9c4b8f924127a45e5a6cdbd038e76558f65286f96efb11aa","observation_id":"2aae58d6-a389-4889-8e72-a893b81060c9","resolution":{"observed_at":"2026-08-05T23:53:45.453729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.573456Z","title":"Eventego3d: 3d human motion capture from egocentric event streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.573456Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0fdc38a4f6f9c4179bc3984faa7ff584a7bdd4426d2615d0cc58d13bbcd7633c","observation_id":"84938642-17db-41c8-a20c-ea177fbaf462","resolution":{"observed_at":"2026-08-05T23:53:45.573456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.657160Z","title":"imapper: interaction-guided scene mapping from monocular videos","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.657160Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ef0d5930121ddeb7475024dccb2a20616e62c0c83ee7e7d0d96d2ebcb6e4ee0d","observation_id":"494b04af-b36d-4159-9380-b432773a1f3f","resolution":{"observed_at":"2026-08-05T23:53:45.657160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.749385Z","title":"Grounded human-object interaction hotspots from video","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.749385Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3bae9ae9510fa610fb57ee308b9cf11ed796c92d8052c093ddbcec01fd6711a7","observation_id":"0ddf64cf-2b30-4952-bdf7-f01f72aaadca","resolution":{"observed_at":"2026-08-05T23:53:45.749385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.722317Z","title":"Jointly learning energy expenditures and activities using egocentric multimodal signals","venue":null,"work_id":"947d016d-6ed6-4b9c-b7da-22eb8c5465a6","year":2017},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.857569Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:dac786fe92a215e6d3adcb6f03a996fca1d2986b738a3fb3fa2d4a7bf7424ca5","observation_id":"69936d59-83a1-47c4-a8df-2f5a62524333","resolution":{"observed_at":"2026-08-05T23:54:07.798225Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.582075Z","title":"You2me: Inferring body pose in egocentric video via first and second person interactions","venue":null,"work_id":"8e9fece5-c74f-42a3-923f-01122b091fa8","year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.929362Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9212015913f021893a6d98b6b16dd5655872243b44ca763f1b8492cbd72429f1","observation_id":"6150a92d-f71f-493e-93d4-937619f86f4e","resolution":{"observed_at":"2026-08-05T23:54:07.639114Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08864","last_updated":"2025-05-14T15:22:36Z","snapshot_observed_at":"2026-08-13T13:59:48.091257Z","submitted_at":"2023-10-13T05:20:40Z","title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08864","snapshot_observed_at":"2026-08-05T23:53:46.042835Z","title":"Open x-embodiment: Robotic learning datasets and rt-x models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.042835Z"},"links":{"cited_paper":"/paper/2310.08864","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3398ee10640b828c42211137bc301aa503a3f0e6ae1150f35cbdb9d7f0109262","observation_id":"33ef0a73-01ba-4479-920e-8b3bdf9b2981","resolution":{"observed_at":"2026-08-05T23:53:46.042835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.400131Z","title":"GPT -3.5 turbo fine-tuning and api updates","venue":null,"work_id":"bbd2ba3b-211f-407d-8f2d-bdf340430b06","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.142375Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ffed14b7ae8247c16f2f1586d65b7ff2f08bf1d5e2058f206f60b749bc071ce7","observation_id":"1afc4540-fead-4485-a4e8-854b5a4c8869","resolution":{"observed_at":"2026-08-05T23:54:07.487788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.184857Z","title":"Handoccnet: Occlusion-robust 3d hand mesh estimation network","venue":null,"work_id":"90a1e57f-79c9-4454-8c96-24297042dafa","year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.246943Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a5f5e7ddb7a98e2810f51ea06825b32b044a603645d177cd78eb2fd63cc8498f","observation_id":"4c35c632-a8b2-4dc0-a419-66eeda1914cd","resolution":{"observed_at":"2026-08-05T23:54:07.289841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.966379Z","title":"Reconstructing hands in 3 D with transformers","venue":null,"work_id":"6704ba4a-3547-48aa-97d0-d4f763bbf6c4","year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.352960Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ea1cc17a46fbd1bf5bb5fcb7b1ae51eeca836ca323753c87994cd0c9469904f0","observation_id":"d041e185-0ec9-4d58-aa3c-b08493b00189","resolution":{"observed_at":"2026-08-05T23:54:07.075067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06553","last_updated":"2025-07-07T05:09:32Z","snapshot_observed_at":"2026-08-13T05:04:55.541639Z","submitted_at":"2023-12-11T17:41:17Z","title":"HOI-Diff: Text-Driven Synthesis of 3D Human-Object Interactions using Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06553","snapshot_observed_at":"2026-08-05T23:53:46.456426Z","title":"Hoi-diff: Text-driven synthesis of 3d human-object interactions using diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.456426Z"},"links":{"cited_paper":"/paper/2312.06553","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:beb171501a66f6ef6781b3a3da4a758a53cf1077df1f50aebef1aa0f8c8154d3","observation_id":"8d0e308e-1883-4061-a35a-d07bf788daf1","resolution":{"observed_at":"2026-08-05T23:53:46.456426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.778148Z","title":"Action-conditioned 3d human motion synthesis with transformer vae","venue":null,"work_id":"d8dc1039-d934-444b-be59-b625cd4e1917","year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.571809Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:482cba53a29bb2e3d0c884dabebb487fdfd3651acd9a7f9591427c8bf850495c","observation_id":"d63f23e3-813a-48bf-aa21-5460f407a4e5","resolution":{"observed_at":"2026-08-05T23:54:06.852047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.532077Z","title":"The kit motion-language dataset","venue":null,"work_id":"92d000ff-7386-444e-aeb1-e3605e93c828","year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.687039Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ad9995b63b3bbbfb6ac7e425b15e2be27f8c2553d26ba32d88019956356ebaab","observation_id":"123dced8-f65b-427f-8cd9-68a3f50f93da","resolution":{"observed_at":"2026-08-05T23:54:06.629054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12259","last_updated":"2025-03-26T18:05:52Z","snapshot_observed_at":"2026-08-14T02:31:33.405994Z","submitted_at":"2024-09-18T18:46:51Z","title":"WiLoR: End-to-end 3D Hand Localization and Reconstruction in-the-wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12259","snapshot_observed_at":"2026-08-05T23:53:46.756838Z","title":"Wilor: End-to-end 3d hand localization and reconstruction in-the-wild","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.756838Z"},"links":{"cited_paper":"/paper/2409.12259","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:7520d1cffbf9d2a5408d040a094ce6811caa693cfa0720f7c27e77fd1ed3203f","observation_id":"56b393d0-dc53-4bec-9a5f-648d367cb9a3","resolution":{"observed_at":"2026-08-05T23:53:46.756838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.304769Z","title":"Mild: multimodal interactive latent dynamics for learning human-robot interaction","venue":null,"work_id":"81c64d38-24cf-4f8a-9add-1012da9136b5","year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.877540Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:7e74737e4736debbb9bd6868e47b71a03d9c8ba888185f4ec9f9d99264a6dcd3","observation_id":"5415fe11-7538-45d6-ac2a-24de01eb5990","resolution":{"observed_at":"2026-08-05T23:54:06.447546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.106318Z","title":"Moveint: Mixture of variational experts for learning human-robot interactions from demonstrations","venue":null,"work_id":"90458d95-6c1d-4f5e-a356-dbf4f16a78ed","year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.982127Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:fcafd5bb9c4134552ba87e0977f6f9aac3ba6d632d075ff01e9955a4f7c86af2","observation_id":"5241ac67-f43e-44d9-b168-caecac513f19","resolution":{"observed_at":"2026-08-05T23:54:06.202591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:05.900642Z","title":"The virtual caliper: Rapid creation of metrically accurate avatars from 3d measurements","venue":null,"work_id":"9c1c183d-22b6-4d5e-9ad3-5e0d185a7384","year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:47.129799Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:7a50127c80defc2bb95614edf528321d3e4cd85795a09bafb5d9f0759dc4964f","observation_id":"da8967d9-cbf4-48d8-af9a-09d86ffdd0a7","resolution":{"observed_at":"2026-08-05T23:54:06.000165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T00:47:19.377686Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":85,"verified_exact":5,"verified_fuzzy":10},"total_outbound_references":150},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 100 of 150 outbound references and 2 inbound Pith citation observations for arXiv:2508.04681."}