{"as_of":"2026-08-09T16:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7e871f50cc4137f3223e6da2876516563f5346fd8ba4c11c0201cf786d35f823","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":27,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:55:19.930713Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T19:20:06.340081Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:7c9a494bbf84badac859ae9aae02f4428edd5663bf527d16367bd6bf651a3e07","observation_id":"0e9f303b-77b4-4360-9f91-55a7dd733807","resolution":{"observed_at":"2026-05-23T06:02:37.563253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-07T11:55:19.930713Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01133","last_updated":"2025-06-01T19:33:21Z","snapshot_observed_at":"2026-08-07T11:48:13.623646Z","submitted_at":"2025-06-01T19:33:21Z","title":"From Words to Waves: Analyzing Concept Formation in Speech and Text-Based Foundation Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:19.930713Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2506.01133"},"observation_digest":"sha256:03a4a57f69f6fd78c771381e79d24e17ebe1ff837b7e5cd4ce78b6af555d5082","observation_id":"6a2b2aea-b7d0-4516-80ab-a56c46060c39","resolution":{"observed_at":"2026-08-07T11:55:19.930713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-06T22:36:14.070930Z","title":"Hu- manomni: A large vision-speech language model for human-centric video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.21277","last_updated":"2025-06-26T14:01:03Z","snapshot_observed_at":"2026-08-08T12:42:20.980285Z","submitted_at":"2025-06-26T14:01:03Z","title":"HumanOmniV2: From Understanding to Omni-Modal Reasoning with Context","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:36:14.070930Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2506.21277"},"observation_digest":"sha256:11e452e32ed9e6231cd668d8484c4918caaff1e38f58ec446fa772145ebf0a84","observation_id":"b6abf601-e5a5-40b6-998a-9d42fbc6d159","resolution":{"observed_at":"2026-08-06T22:36:14.070930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-06T22:24:36.656003Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding.arXiv preprint arXiv:2501.15111, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.21862","last_updated":"2025-06-27T02:29:58Z","snapshot_observed_at":"2026-08-09T10:29:21.701927Z","submitted_at":"2025-06-27T02:29:58Z","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","version":1},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-08-06T22:24:36.656003Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2506.21862"},"observation_digest":"sha256:a0b96d0d3e089628aec030dfde23653ef338ff8dbbd6f9d286fc80b0e513e063","observation_id":"1fb4ef1b-8bbc-4fc1-86c2-1b82f69ac1de","resolution":{"observed_at":"2026-08-06T22:24:36.656003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-06T20:25:32.881282Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human- Centric Video Understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02771","last_updated":"2025-07-03T16:34:34Z","snapshot_observed_at":"2026-08-09T00:45:22.366107Z","submitted_at":"2025-07-03T16:34:34Z","title":"Grounding Intelligence in Movement","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-06T20:25:32.881282Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2507.02771"},"observation_digest":"sha256:0ed772c198fed36d00ca80a7fa227f1c9bea33d8f7c537f75d42533fe2f0f767","observation_id":"2d8ab3f4-bcb6-4714-a34a-933d1bc3c8e0","resolution":{"observed_at":"2026-08-06T20:25:32.881282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-06T17:39:41.688863Z","title":"Humanomni: A large vision-speech lan- guage model for human-centric video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10300","last_updated":"2025-07-14T14:04:14Z","snapshot_observed_at":"2026-08-08T22:29:22.885986Z","submitted_at":"2025-07-14T14:04:14Z","title":"FaceLLM: A Multimodal Large Language Model for Face Understanding","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T17:39:41.688863Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2507.10300"},"observation_digest":"sha256:b30f1134eb821f89d19683f53504b447a26b1d73b58581509a1f5778247c148d","observation_id":"e079e223-3733-428a-8926-77c04bf46313","resolution":{"observed_at":"2026-08-06T17:39:41.688863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-06T05:49:55.526816Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.01178","last_updated":"2025-08-02T03:33:47Z","snapshot_observed_at":"2026-08-07T21:57:57.399342Z","submitted_at":"2025-08-02T03:33:47Z","title":"Advancing the Foundation Model for Music Understanding","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-06T05:49:55.526816Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2508.01178"},"observation_digest":"sha256:151cbbe2b10e556de083e65afea7ae8bd4e843a179650324ea9539f50d5d2245","observation_id":"100cee19-04d4-461f-b53b-523d9ca1c7ac","resolution":{"observed_at":"2026-08-06T05:49:55.526816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-06T05:02:24.540778Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.02429","last_updated":"2025-08-04T13:49:03Z","snapshot_observed_at":"2026-08-08T01:53:05.709859Z","submitted_at":"2025-08-04T13:49:03Z","title":"Multimodal Large Language Models for End-to-End Affective Computing: Benchmarking and Boosting with Generative Knowledge Prompting","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T05:02:24.540778Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2508.02429"},"observation_digest":"sha256:59dbb05103c4396ea465837bcca2d5dd0b7c4020487dc743839e35d4b7273910","observation_id":"fa56c0e6-2f13-4c61-8dc4-f259ad81b243","resolution":{"observed_at":"2026-08-06T05:02:24.540778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2508.10016","last_updated":"2026-05-22T12:46:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-06T16:17:29Z","title":"Training-Free Multimodal Large Language Model Orchestration","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-19T00:12:39.834892Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2508.10016"},"observation_digest":"sha256:c5a21779c9150bee84fd5c79b511c140113121c00f7f152836291490a09b936f","observation_id":"4402c2e6-7863-41c9-b114-689858367479","resolution":{"observed_at":"2026-05-19T00:12:54.107948Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2508.10016","last_updated":"2026-05-22T12:46:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-06T16:17:29Z","title":"Training-Free Multimodal Large Language Model Orchestration","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-25T08:02:15.950975Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2508.10016"},"observation_digest":"sha256:b52ef827b16fdbb966b5a785051d3ced8aa9e95d7a234078d15fd172e2100553","observation_id":"909bd532-5a03-4c10-b716-673949aa9b66","resolution":{"observed_at":"2026-05-25T08:05:30.643788Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2604.00013","last_updated":"2026-04-12T03:30:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-03-10T12:48:41Z","title":"C2F-Thinker: Coarse-to-Fine Reasoning with Hint-Guided Reinforcement Learning for Multimodal Sentiment Analysis","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-15T13:51:40.334057Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2604.00013"},"observation_digest":"sha256:2f97ba2ed36bf0f0c7b24fb1d65d038d78aac611a13c9fb11a174003ba1df854","observation_id":"11cec45c-71e6-466f-97e3-23c5c3c0950d","resolution":{"observed_at":"2026-05-15T13:55:53.282103Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2604.05079","last_updated":"2026-04-06T18:30:50Z","snapshot_observed_at":"2026-08-02T23:48:40.440218Z","submitted_at":"2026-04-06T18:30:50Z","title":"SVAgent: Storyline-Guided Long Video Understanding via Cross-Modal Multi-Agent Collaboration","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T20:20:08.590407Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2604.05079"},"observation_digest":"sha256:2bbf56707f0193f8c14b6d32d849a572bd1114a023acdaf699a13b50af2d1f1f","observation_id":"4062d7e7-479f-4d9f-a260-29c62f8a83aa","resolution":{"observed_at":"2026-05-10T22:00:48.970344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2604.15823","last_updated":"2026-04-17T08:22:14Z","snapshot_observed_at":"2026-07-06T23:03:16.345488Z","submitted_at":"2026-04-17T08:22:14Z","title":"Watching Movies Like a Human: Egocentric Emotion Understanding for Embodied Companions","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T08:49:33.107658Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2604.15823"},"observation_digest":"sha256:439978d667bf600122da29bf5bb465050db7f6dad8d145f0001a59623db2ce8d","observation_id":"95265503-a77e-4605-b5a6-f35c0d1754c7","resolution":{"observed_at":"2026-05-10T08:53:04.276506Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2604.16617","last_updated":"2026-04-17T18:13:27Z","snapshot_observed_at":"2026-08-02T22:28:35.019925Z","submitted_at":"2026-04-17T18:13:27Z","title":"AVRT: Audio-Visual Reasoning Transfer through Single-Modality Teachers","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T08:02:53.574120Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2604.16617"},"observation_digest":"sha256:c4c241ac658e1d2ac8f2ec09ecccfc1709e71d08d86a6cc806b4559c4299c7d0","observation_id":"3f0a8955-3999-4a65-b62d-a7cc1225d133","resolution":{"observed_at":"2026-05-10T09:18:32.134430Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2604.18460","last_updated":"2026-04-20T16:16:36Z","snapshot_observed_at":"2026-07-06T23:05:17.639285Z","submitted_at":"2026-04-20T16:16:36Z","title":"Learning Invariant Modality Representation for Robust Multimodal Learning from a Causal Inference Perspective","version":1},"reference_index":283,"source":"arxiv_source","source_observed_at":"2026-05-10T04:32:29.428080Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2604.18460"},"observation_digest":"sha256:8a54bcd599fe2531463377243392b0e74beab1244d03c5e29a101f2d80aa08e3","observation_id":"a746a167-4cf3-491d-857e-f4a5e5deb2b7","resolution":{"observed_at":"2026-05-11T11:51:03.220651Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2605.07593","last_updated":"2026-05-08T11:06:43Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T11:06:43Z","title":"TraceAV-Bench: Benchmarking Multi-Hop Trajectory Reasoning over Long Audio-Visual Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T01:53:01.939765Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2605.07593"},"observation_digest":"sha256:96eb52b6cf651224ddba01a2677f6bec981767d07276311a504d65789b00ddd9","observation_id":"5bcc8013-624b-4b6c-b3b1-2f639fee0d58","resolution":{"observed_at":"2026-05-11T04:15:56.047112Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2605.15764","last_updated":"2026-05-15T09:24:41Z","snapshot_observed_at":"2026-07-06T23:27:01.213106Z","submitted_at":"2026-05-15T09:24:41Z","title":"GRASP: Learning to Ground Social Reasoning in Multi-Person Non-Verbal Interactions","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-20T18:49:18.815456Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2605.15764"},"observation_digest":"sha256:535629a263c846a490f47b3903e6217c2f24bea01c3430fa8a574312d564659d","observation_id":"0c2ce232-45d3-46e7-92f1-58edd413481d","resolution":{"observed_at":"2026-05-20T18:53:39.002635Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2605.18018","last_updated":"2026-05-18T08:09:37Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:09:37Z","title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","version":1},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-05-20T12:10:54.874012Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2605.18018"},"observation_digest":"sha256:abddb8aa47d3aebe6968b980c0cb4f26c3197c50d18de16108b362a5001d4ced","observation_id":"fa6cc457-39d6-4973-b9a0-195c6e5ec81d","resolution":{"observed_at":"2026-05-20T12:13:16.342785Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2605.21417","last_updated":"2026-05-24T12:32:06Z","snapshot_observed_at":"2026-07-06T23:31:51.443407Z","submitted_at":"2026-05-20T17:12:55Z","title":"Ordering Matters: Rank-Aware Selective Fusion for Blended Emotion Recognition","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-21T05:18:29.720630Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2605.21417"},"observation_digest":"sha256:68a8caccb32ce4cf7dc22b7fbe428d039de72782587d7e4317dd0f5b362d87ac","observation_id":"4ff8a9ec-fe22-4656-9a19-5af71dfaf5c6","resolution":{"observed_at":"2026-05-21T05:19:39.200505Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2605.21417","last_updated":"2026-05-24T12:32:06Z","snapshot_observed_at":"2026-07-06T23:31:51.443407Z","submitted_at":"2026-05-20T17:12:55Z","title":"Ordering Matters: Rank-Aware Selective Fusion for Blended Emotion Recognition","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-30T17:06:26.561698Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2605.21417"},"observation_digest":"sha256:551e4dfea0aa94e240f58afa10eeed5a5689fb2815f64c958a0c00d58bbdf29d","observation_id":"430b824a-7ecb-463d-b4af-de283cf666fe","resolution":{"observed_at":"2026-06-30T17:14:57.327465Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2606.05896","last_updated":"2026-06-22T18:29:42Z","snapshot_observed_at":"2026-07-06T23:45:51.910263Z","submitted_at":"2026-06-04T09:03:43Z","title":"Resonant Minds: Closed-Loop Social Avatars with Theory of Mind","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-28T02:04:39.753443Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2606.05896"},"observation_digest":"sha256:d566cd9146a76324cb56d4b24f9d421c086e845d7e0b133ce67262804f18682c","observation_id":"85f127db-a05d-4fc2-868a-471b1e4a238a","resolution":{"observed_at":"2026-07-02T12:36:56.204673Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2606.07643","last_updated":"2026-06-01T19:12:09Z","snapshot_observed_at":"2026-08-04T07:32:54.888878Z","submitted_at":"2026-06-01T19:12:09Z","title":"AVI-Bench: Toward Human-like Audio-Visual Intelligence of Omni-MLLMs","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-28T14:36:53.295540Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2606.07643"},"observation_digest":"sha256:e7944401b9b78684af8680c6c0c2bdf1d030579dd2df990147882769188e27bb","observation_id":"2b0317bc-baa6-490e-815f-2d0073361aed","resolution":{"observed_at":"2026-07-01T23:06:21.372188Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2606.20970","last_updated":"2026-06-18T22:17:18Z","snapshot_observed_at":"2026-08-01T20:10:55.070907Z","submitted_at":"2026-06-18T22:17:18Z","title":"CogniRoute: Learning to Route Social Evidence in Omni-Modal Models","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-06-26T17:37:11.371892Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2606.20970"},"observation_digest":"sha256:e58c03653e51cf1dceb231c946617fd8c80229d732e6eaed956992041067ca7d","observation_id":"69e6726e-d2c1-4fdf-91df-1ca81034a101","resolution":{"observed_at":"2026-07-04T03:49:30.398553Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2606.24286","last_updated":"2026-06-23T08:06:58Z","snapshot_observed_at":"2026-08-07T04:14:24.900720Z","submitted_at":"2026-06-23T08:06:58Z","title":"AVOC: Enhancing Hour-Level Audio-Video Understanding in Omni-Modal LLMs via Retrieval-Inspired Token Compression","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-26T00:16:56.174638Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2606.24286"},"observation_digest":"sha256:9da478bfc01498332cba8e1a775d3f9fe17e4016f7404423cef0068997ca8969","observation_id":"a89b4cd4-0ccc-4af1-9a1c-c3661f978015","resolution":{"observed_at":"2026-07-04T16:49:57.283536Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":"2501.15111","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-07-04T19:20:06.340081Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding","venue":null,"work_id":"f7355d17-2b25-4606-b51d-e07f909d400f","year":2025},"citing_paper":{"arxiv_id":"2606.25325","last_updated":"2026-07-20T13:20:39Z","snapshot_observed_at":"2026-08-02T10:17:41.360200Z","submitted_at":"2026-06-24T02:43:26Z","title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-06-25T21:31:38.450382Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2606.25325"},"observation_digest":"sha256:34a15b2c381bb1de44c661d103274622fcf01f6f33efd9b827c6616d8cd6eb27","observation_id":"8789fd0b-e0c6-4ee2-b437-9aeea8698f35","resolution":{"observed_at":"2026-07-04T19:20:06.342275Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-02T01:25:09.040422Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14683","last_updated":"2026-07-16T07:44:42Z","snapshot_observed_at":"2026-08-08T07:22:18.542280Z","submitted_at":"2026-07-16T07:44:42Z","title":"InCarEmo: A Multimodal Dataset for In-Cabin Emotion Recognition and Driver State Monitoring","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T01:25:09.040422Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2607.14683"},"observation_digest":"sha256:778dd28a0117ac333f9dbd0177e8da13fab52124e099078c6d0dac0c2d794230","observation_id":"9152cc76-dd74-4f33-b421-a8b6edea5666","resolution":{"observed_at":"2026-08-02T01:25:09.040422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15111","snapshot_observed_at":"2026-08-01T20:07:41.111549Z","title":"Humanomni: A large vision-speech language model for human-centric video understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16742","last_updated":"2026-07-18T10:04:18Z","snapshot_observed_at":"2026-08-07T21:51:22.400147Z","submitted_at":"2026-07-18T10:04:18Z","title":"Multi-Dimensional Quality Assessment for AI-Generated Human-Centric Videos: Dataset and Model","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T20:07:41.111549Z"},"links":{"cited_paper":"/paper/2501.15111","citing_paper":"/paper/2607.16742"},"observation_digest":"sha256:7e5541cda7a990d5fcdbc403751e1f04d1677181f231350afe5582ff2237bdf5","observation_id":"d47f81cc-cea3-4ad9-a62b-73d38c8f2d38","resolution":{"observed_at":"2026-08-01T20:07:41.111549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.15111/citation-record","integrity":"/paper/2501.15111/integrity","json":"/paper/2501.15111/citation-record.json","paper":"/paper/2501.15111"},"outbound":[],"paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T11:56:57.969083Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 27 inbound Pith citation observations for arXiv:2501.15111."}