{"as_of":"2026-08-17T21:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1039daed78323efc50ac9f939a67e0157548b47c29030ca1bae84d8f7b8b157e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":16,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":16,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":16,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T05:07:05.209945Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T17:18:43.953855Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-08-17T14:16:52.244007Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":118,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:5a2d8316efcc8ce9dbe241bbc9331a8db9220a781efa5c88c88ccdac1ab03006","observation_id":"f732151a-73f6-47c9-851b-5054f3964f01","resolution":{"observed_at":"2026-05-12T20:58:59.238954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2411.10442","last_updated":"2025-04-07T09:09:39Z","snapshot_observed_at":"2026-08-13T18:24:04.904767Z","submitted_at":"2024-11-15T18:59:27Z","title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","version":2},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-05-16T09:16:17.150383Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2411.10442"},"observation_digest":"sha256:fa067df5ee2988cb8b748e781cc12d0fac0b6aa01bc56ce033600a0ef1de2e5a","observation_id":"7446d942-befe-4d0a-9ee5-efe76ad137e5","resolution":{"observed_at":"2026-05-16T09:16:17.500653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-12T18:30:41.580179Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.11507","last_updated":"2024-11-18T12:12:33Z","snapshot_observed_at":"2026-08-16T03:03:09.503436Z","submitted_at":"2024-11-18T12:12:33Z","title":"SignEye: Traffic Sign Interpretation from Vehicle First-Person View","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T18:30:41.580179Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2411.11507"},"observation_digest":"sha256:5210904f86423dfd49dd40bceffee448fc27e18a8c9c2e1a46e7239ca51b7e08","observation_id":"2a88959a-3481-4f4e-a554-dbe640691a1e","resolution":{"observed_at":"2026-08-12T18:30:41.580179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":251,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:23c19942904790ad0e2ee549362b2d254772a0cb1af62db8f2b905d236edfaa4","observation_id":"a7c9f2ef-c504-48a9-bc13-0f6fe46ade92","resolution":{"observed_at":"2026-05-10T13:23:58.122139Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-11T19:51:43.606576Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.06322","last_updated":"2024-12-09T09:18:32Z","snapshot_observed_at":"2026-08-15T15:25:43.651248Z","submitted_at":"2024-12-09T09:18:32Z","title":"LLaVA-SpaceSGG: Visual Instruct Tuning for Open-vocabulary Scene Graph Generation with Enhanced Spatial Relations","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T19:51:43.606576Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2412.06322"},"observation_digest":"sha256:c74ff6537d54acaee8c775e036a70cd65479fa2ee161caef9357e8df1fc0d43f","observation_id":"a669df6c-77bd-4c71-b69b-ea830c6f785f","resolution":{"observed_at":"2026-08-11T19:51:43.606576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-11T00:47:12.411857Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19326","last_updated":"2025-06-30T13:15:13Z","snapshot_observed_at":"2026-08-13T13:43:26.165297Z","submitted_at":"2024-12-26T18:56:05Z","title":"Task Preference Optimization: Improving Multimodal Large Language Models with Vision Task Alignment","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-11T00:47:12.411857Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2412.19326"},"observation_digest":"sha256:a51d96e7965588b77a2871e616d205e3f07bb57396c3a269fd6c626597a6764a","observation_id":"5a475ba5-0bba-44a8-8f13-ad99d42f6da7","resolution":{"observed_at":"2026-08-11T00:47:12.411857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-10T22:08:09.926810Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-14T12:32:34.328935Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"reference_index":265,"source":"pdf_text","source_observed_at":"2026-08-10T22:08:09.926810Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2501.02765"},"observation_digest":"sha256:eaa553f03718a407f7cd2a7ac8a8c0f7e0d4d2000e5012d9e996940a2df8ab1f","observation_id":"d553f152-84d0-4899-9d25-7e442a9855a9","resolution":{"observed_at":"2026-08-10T22:08:09.926810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-10T15:32:48.341347Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13921","last_updated":"2025-02-11T16:48:15Z","snapshot_observed_at":"2026-08-15T05:20:40.055405Z","submitted_at":"2025-01-23T18:59:02Z","title":"The Breeze 2 Herd of Models: Traditional Chinese LLMs Based on Llama with Vision-Aware and Function-Calling Capabilities","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-10T15:32:48.341347Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2501.13921"},"observation_digest":"sha256:a1edb074c989d9d6abd0046462dd1e54c9906bfcac71a7e812fedcbc438ba136","observation_id":"472eada9-fc76-4caf-8cc4-0cb281b81a9f","resolution":{"observed_at":"2026-08-10T15:32:48.341347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-17T09:56:52.502317Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":127,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:896771949d807348b11b53fd0e1967b187c1e7f626d88cd3fbca23e35b873618","observation_id":"4fc2bd6c-90b1-4dca-ab4e-02ab95a27224","resolution":{"observed_at":"2026-05-10T13:41:08.126330Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-16T05:07:05.209945Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.21530","last_updated":"2025-04-30T11:26:40Z","snapshot_observed_at":"2026-08-17T17:25:54.985349Z","submitted_at":"2025-04-30T11:26:40Z","title":"RoboGround: Robotic Manipulation with Grounded Vision-Language Priors","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T05:07:05.209945Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2504.21530"},"observation_digest":"sha256:b4f021b682ac433776371bc94f5fe8722f02c4084b227899f73989cfce24db78","observation_id":"c4c29b91-de0c-400b-b238-62856187c9ae","resolution":{"observed_at":"2026-08-16T05:07:05.209945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-06T15:57:04.926275Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-12T01:30:38.455355Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:04.926275Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:ce94af4e5c7dc7a0079119cce87bdf7f053a88c87910cb0356d338da59867b88","observation_id":"7c52519d-d1bd-4953-b3ef-870c2d40417b","resolution":{"observed_at":"2026-08-06T15:57:04.926275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-08-17T12:32:16.575866Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":145,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:8254f7cbdb644634a83e102cc03c1aec22de7bd2a8e4073ee20838f26b152b33","observation_id":"71525835-f3cc-46a5-aac0-c0f2ba608ae6","resolution":{"observed_at":"2026-05-10T11:58:58.860124Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-08-05T14:01:17.942135Z","title":"The all-seeing project v2: Towards general relation compre- hension of the open world","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21809","last_updated":"2025-08-29T17:43:58Z","snapshot_observed_at":"2026-08-15T18:48:00.823889Z","submitted_at":"2025-08-29T17:43:58Z","title":"VoCap: Video Object Captioning and Segmentation from Any Prompt","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-05T14:01:17.942135Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2508.21809"},"observation_digest":"sha256:402f7ee458e5b2ecac53bad95e92306da6bcac0676998f9636f200902115fb7e","observation_id":"1daf37a9-7708-467f-9ccb-7631b853dd2b","resolution":{"observed_at":"2026-08-05T14:01:17.942135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2604.18000","last_updated":"2026-04-20T09:25:30Z","snapshot_observed_at":"2026-08-03T00:50:48.601549Z","submitted_at":"2026-04-20T09:25:30Z","title":"Unmasking the Illusion of Embodied Reasoning in Vision-Language-Action Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-10T04:27:18.284698Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2604.18000"},"observation_digest":"sha256:37dc93391d1bf156a38c0aaaebeddeb7d9872d81dd1c588d4a3f6d8ed051a258","observation_id":"ddcf7caf-b1a0-46fd-a484-ab6de461b55a","resolution":{"observed_at":"2026-05-11T11:56:28.046008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":151,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:5e249b7ca2490256d42660e91fe3a0fcacbc524cf71b6f64d8e9c31d1be2c513","observation_id":"f3d9eb25-7eb4-444c-b572-cdc9af1c10d0","resolution":{"observed_at":"2026-07-03T10:48:03.033665Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","version":4},"cited_work":{"arxiv_id":"2402.19474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19474","snapshot_observed_at":"2026-07-03T17:18:43.953855Z","title":"The all-seeing project v2: Towards general relation comprehension of the open world","venue":null,"work_id":"1d280fb4-5908-49aa-890a-274a0ac40e7f","year":2024},"citing_paper":{"arxiv_id":"2606.17030","last_updated":"2026-06-17T13:54:57Z","snapshot_observed_at":"2026-08-01T21:48:31.832288Z","submitted_at":"2026-06-15T17:52:31Z","title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","version":3},"reference_index":195,"source":"arxiv_source","source_observed_at":"2026-06-27T04:19:26.332718Z"},"links":{"cited_paper":"/paper/2402.19474","citing_paper":"/paper/2606.17030"},"observation_digest":"sha256:28b2ea1ccebef48be98b710837db3e9fc2591736b6c74d0529369536300b05ef","observation_id":"e57638b4-9076-462c-bdc5-ab9d8a9314b8","resolution":{"observed_at":"2026-07-03T17:18:43.955485Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2402.19474/citation-record","integrity":"/paper/2402.19474/integrity","json":"/paper/2402.19474/citation-record.json","paper":"/paper/2402.19474"},"outbound":[],"paper":{"arxiv_id":"2402.19474","last_updated":"2024-08-23T07:20:57Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T14:14:00.076855Z","submitted_at":"2024-02-29T18:59:17Z","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 16 inbound Pith citation observations for arXiv:2402.19474."}