{"as_of":"2026-08-14T14:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:62bf325de9d2e12c1f85270f6bfa0f94b4634c470288e5d20d5c5a6220c125fa","coverage":[{"denominator":95,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":95,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:33:54.320561Z","state":"measured"},{"denominator":99,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":99,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:24:36.541729Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-23T06:02:37.644844Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":"2501.07978","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fa- cial dynamics in video: Instruction tuning for improved fa- cial expression perception and contextual awareness","venue":null,"work_id":"f0aef9b3-ec8d-4b60-981d-68951a751ecc","year":2025},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-08-12T23:41:31.711979Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:c4197c3304b8f839899703cc99b8646d2702d9412592a9847a55d99e9a5f4647","observation_id":"dfbf3f28-6334-4a74-834c-57a2e45d06e1","resolution":{"observed_at":"2026-05-23T06:02:37.649306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":"2501.07978","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fa- cial dynamics in video: Instruction tuning for improved fa- cial expression perception and contextual awareness","venue":null,"work_id":"f0aef9b3-ec8d-4b60-981d-68951a751ecc","year":2025},"citing_paper":{"arxiv_id":"2503.09158","last_updated":"2026-05-09T06:28:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T08:33:46Z","title":"FaVChat: Hierarchical Prompt-Query Guided Facial Video Understanding with Data-Efficient GRPO","version":6},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T00:21:51.621582Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2503.09158"},"observation_digest":"sha256:3494d8b0b98bf75da7ae7b3addf5e6b66a270115c82839a953c7d9e094a45f18","observation_id":"9ba63ce5-3dbf-4d8c-90ca-ef657731fb73","resolution":{"observed_at":"2026-05-23T00:22:18.796437Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-08-06T22:24:36.541729Z","title":"Facial dynamics in video: Instruction tuning for improved facial expression perception and contextual awareness.arXiv preprint arXiv:2501.07978, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.21862","last_updated":"2025-06-27T02:29:58Z","snapshot_observed_at":"2026-08-13T09:08:08.645501Z","submitted_at":"2025-06-27T02:29:58Z","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","version":1},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-08-06T22:24:36.541729Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2506.21862"},"observation_digest":"sha256:d30b11705c966ff741db65817c7550cfb1b0a376f14b7e03c13f5702df180ac7","observation_id":"2f1f05b9-e8db-4b09-a634-0d7783dd5351","resolution":{"observed_at":"2026-08-06T22:24:36.541729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"cited_work":{"arxiv_id":"2501.07978","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07978","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fa- cial dynamics in video: Instruction tuning for improved fa- cial expression perception and contextual awareness","venue":null,"work_id":"f0aef9b3-ec8d-4b60-981d-68951a751ecc","year":2025},"citing_paper":{"arxiv_id":"2605.18018","last_updated":"2026-05-18T08:09:37Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:09:37Z","title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-05-20T12:10:54.874012Z"},"links":{"cited_paper":"/paper/2501.07978","citing_paper":"/paper/2605.18018"},"observation_digest":"sha256:8bb6080e724451199fba2d2166c0cab4dc8675a0cef0c4c991110c1b6ea95108","observation_id":"260b9555-6759-4738-8c89-c4f26d70f407","resolution":{"observed_at":"2026-05-20T12:13:16.336074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.07978/citation-record","integrity":"/paper/2501.07978/integrity","json":"/paper/2501.07978/citation-record.json","paper":"/paper/2501.07978"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.884278Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.884278Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b3868b291f8e2503a0a7ee695188109f072b6b1953a86dcc2cf6f1ea27b2eb57","observation_id":"cc93ae32-cc76-4acc-ba60-cbec4923ba3e","resolution":{"observed_at":"2026-08-10T20:33:53.884278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.889385Z","title":"Emotion recognition in speech using cross- modal transfer in the wild, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.889385Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:91b27bba7093ea3c541b517e670eba59f67d9b8c621691787166901bcaaded06","observation_id":"4abc8b02-edbf-4fe1-a5c3-b75fdb296c4a","resolution":{"observed_at":"2026-08-10T20:33:53.889385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.893802Z","title":"Claude-3.5, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.893802Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:bb6f3d5c6af366d1089aef1b3d723f77154b9643723adc284dd4f6cf1f615e42","observation_id":"cdd82447-b527-47b6-8bcc-f1a8e953315f","resolution":{"observed_at":"2026-08-10T20:33:53.893802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-10T20:33:53.903232Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.903232Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:1205c9673a275436d4fdb6b524a8bf121a1d38dd6e6d070d09c664cbb9cfccfe","observation_id":"d6f6185c-f02d-4cf3-a693-a66927114450","resolution":{"observed_at":"2026-08-10T20:33:53.903232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.907800Z","title":"Collecting highly paral- lel data for paraphrase evaluation","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.907800Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:3fa49f1ddff605e212b57e2457448a04e064cabe7577ed9ec30a05344e71b2b4","observation_id":"8cbe17d3-6f5a-421c-ba14-585c11660183","resolution":{"observed_at":"2026-08-10T20:33:53.907800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02157","last_updated":"2025-06-24T07:42:09Z","snapshot_observed_at":"2026-08-12T23:30:31.638008Z","submitted_at":"2024-07-02T10:55:43Z","title":"FineCLIPER: Multi-modal Fine-grained CLIP for Dynamic Facial Expression Recognition with AdaptERs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02157","snapshot_observed_at":"2026-08-10T20:33:53.913411Z","title":"Finecliper: Multi-modal fine-grained clip for dynamic facial expression recognition with adapters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.913411Z"},"links":{"cited_paper":"/paper/2407.02157","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:93ba61bd7b0fd69df0e8c660a7cd65a2056a3ba50d54bf8c1ebf645b5ad03f0d","observation_id":"14417cbe-13db-43dc-8f44-377172cdc49b","resolution":{"observed_at":"2026-08-10T20:33:53.913411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04325","last_updated":"2024-06-06T17:58:54Z","snapshot_observed_at":"2026-08-12T23:48:30.486852Z","submitted_at":"2024-06-06T17:58:54Z","title":"ShareGPT4Video: Improving Video Understanding and Generation with Better Captions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04325","snapshot_observed_at":"2026-08-10T20:33:53.918678Z","title":"Sharegpt4video: Improving video understand- ing and generation with better captions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.918678Z"},"links":{"cited_paper":"/paper/2406.04325","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:037368356808d341e2724488add849ba5250a1272e962770ed116dee87e379e3","observation_id":"7ade2e87-b39e-4fb9-a3b9-2a3bacd370d0","resolution":{"observed_at":"2026-08-10T20:33:53.918678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.923207Z","title":"Stcam: Spatial-temporal and channel attention module for dynamic facial expression recognition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.923207Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a012a2d93edc1bb7a1a3adf0b265dcca65a42641b08145803c8228f119a5963f","observation_id":"a70c08be-1453-477b-a074-daa4cfb9a060","resolution":{"observed_at":"2026-08-10T20:33:53.923207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.14238","snapshot_observed_at":"2026-08-10T20:33:53.927345Z","title":"Internvl: Scaling up vision foundation mod- els and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.927345Z"},"links":{"cited_paper":"/paper/2312.14238","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:f81231c015497ade981c891660ba7ca906e8c64135ada0c904ce2f8f0074661a","observation_id":"058f14fb-70b5-4ece-a08c-26afc70ea1c0","resolution":{"observed_at":"2026-08-10T20:33:53.927345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-10T20:33:53.930933Z","title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video- llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.930933Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:7813ce4d4abd2f9267e309af331d4ae6e7924d8e1e5a6260aa99e94ee14fd9c8","observation_id":"8d653a64-b7f1-4921-926f-8b741118b8ba","resolution":{"observed_at":"2026-08-10T20:33:53.930933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.934690Z","title":"Transface: Calibrating trans- former training for face recognition from a data-centric per- spective, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.934690Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:da91db4bfa538c4717ea48959a9f4c5c48beb88749f85465ea6953dc7a9724d3","observation_id":"184a51db-02cc-49ff-a15b-11671cca3d01","resolution":{"observed_at":"2026-08-10T20:33:53.934690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.937932Z","title":"Diffusionrig: Learning personal- ized priors for facial appearance editing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.937932Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:78a93bb45f1487dad153906fcff8335dabf427d1587467ed5aef54d1310d9656","observation_id":"70be547e-fdad-44e1-8463-20e460e9a722","resolution":{"observed_at":"2026-08-10T20:33:53.937932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16420","last_updated":"2024-01-29T18:59:02Z","snapshot_observed_at":"2026-08-05T03:42:54.599829Z","submitted_at":"2024-01-29T18:59:02Z","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16420","snapshot_observed_at":"2026-08-10T20:33:53.942983Z","title":"Internlm-xcomposer2: Mastering free-form text-image composition and compre- hension in vision-language large model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.942983Z"},"links":{"cited_paper":"/paper/2401.16420","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:feda528d59930f4698ea3fda2544813f9caca97a13d69aaeedfa93280519d5c0","observation_id":"868dbd0f-f384-43a1-8d93-af5d185986ac","resolution":{"observed_at":"2026-08-10T20:33:53.942983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.947008Z","title":"Strongsort: Make deep- sort great again","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.947008Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a86cbb3b3f2dfc868d18cfded18c50f7bf9716e1601a76d734877415a20c86e1","observation_id":"6236c81b-5727-4fb4-8b60-8bc7774588a9","resolution":{"observed_at":"2026-08-10T20:33:53.947008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.950534Z","title":"EmoCLIP: A Vision-Language Method for Zero-Shot Video Facial Ex- pression Recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.950534Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a82ed15f8b97912a5947f7c7df6787d15c5587603c1b01af62a31019d82be9be","observation_id":"17472c72-cec0-40a2-99b2-84c691b507e7","resolution":{"observed_at":"2026-08-10T20:33:53.950534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-10T20:33:53.954454Z","title":"Video-mme: The first-ever compre- hensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.954454Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:5611110561dd6cc38b692fa36a18d79793557a2354015da63ea632e25e649a52","observation_id":"8e20eb0f-a12a-465a-9847-5d95dbc2c3b4","resolution":{"observed_at":"2026-08-10T20:33:53.954454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.15010","last_updated":"2023-04-28T17:59:25Z","snapshot_observed_at":"2026-08-12T23:40:42.885633Z","submitted_at":"2023-04-28T17:59:25Z","title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.15010","snapshot_observed_at":"2026-08-10T20:33:53.958465Z","title":"Llama-adapter v2: Parameter-efficient vi- sual instruction model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.958465Z"},"links":{"cited_paper":"/paper/2304.15010","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:0960bc9d521c6805f97e25d3b176ede357c00c069cbc15cf7fc46196b1682a94","observation_id":"2ee3927a-96fb-42f8-b2fd-123ecbc2904c","resolution":{"observed_at":"2026-08-10T20:33:53.958465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.963217Z","title":"Music Emotion Recognition: Toward new, robust standards in personalized and context-sensitive ap- plications","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.963217Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:52ff76f1606274883c3133fd9faf67f77bc33f84fcb7f02418f3e8cfa9c85eec","observation_id":"289d0ca9-4987-45a4-9795-01f08b933ada","resolution":{"observed_at":"2026-08-10T20:33:53.963217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.967225Z","title":"LoRA: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.967225Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:2ee3ee1dced305677da8183db5c09ec04040026b10c3cb4825d7c0e52a0a1ad2","observation_id":"b69d2990-0a6e-4675-a16e-24c27a8da406","resolution":{"observed_at":"2026-08-10T20:33:53.967225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.11760","last_updated":"2020-11-10T21:49:14Z","snapshot_observed_at":"2026-08-06T11:42:54.284269Z","submitted_at":"2020-11-10T21:49:14Z","title":"Multimodal Pretraining for Dense Video Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2011.11760","snapshot_observed_at":"2026-08-10T20:33:53.972371Z","title":"Multimodal pretraining for dense video cap- tioning","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.972371Z"},"links":{"cited_paper":"/paper/2011.11760","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:6a73a9cc68d87915b6eb3e4950a216b6a430a28739cc0368709e1f2fce274ce9","observation_id":"b7602365-cb26-4dfd-ba34-4fa8ff056ba9","resolution":{"observed_at":"2026-08-10T20:33:53.972371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13250","last_updated":"2024-05-16T12:23:05Z","snapshot_observed_at":"2026-08-13T04:14:17.025561Z","submitted_at":"2024-02-20T18:58:54Z","title":"Video ReCap: Recursive Captioning of Hour-Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13250","snapshot_observed_at":"2026-08-10T20:33:53.977080Z","title":"Video recap: Recursive captioning of hour-long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.977080Z"},"links":{"cited_paper":"/paper/2402.13250","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:f20bd6ad9a71778df0ca30ed8c6f515f63fc931f8b44b27a317539a047d53b73","observation_id":"4d501ff5-cfb1-45a9-8ef6-2707817343ae","resolution":{"observed_at":"2026-08-10T20:33:53.977080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.982395Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.982395Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9dbbc773f5830b93010a69b95a47d39ba258492ff3ca3ef73810a8d12cbdc0b2","observation_id":"33d0f7e5-2d02-4c8f-9663-7bed3fa6c57e","resolution":{"observed_at":"2026-08-10T20:33:53.982395Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.986831Z","title":"Dfew: A large-scale database for recognizing dynamic facial expres- sions in the wild, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.986831Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:86460ea2653afe84d46d90ade8f80412da5354c8602738183710150fbfda654c","observation_id":"0d82b7f6-3584-472c-9619-29f1ddd9bf3a","resolution":{"observed_at":"2026-08-10T20:33:53.986831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.08046","last_updated":"2024-04-05T15:21:09Z","snapshot_observed_at":"2026-08-13T05:25:45.984864Z","submitted_at":"2023-11-14T10:11:36Z","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.08046","snapshot_observed_at":"2026-08-10T20:33:53.991893Z","title":"Chat-univi: Unified visual representation em- 9 powers large language models with image and video under- standing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.991893Z"},"links":{"cited_paper":"/paper/2311.08046","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9536b083ed0ebf4b304fc1b2fe3500b9391117c28dbcd1899f0bf3980b3f4fc2","observation_id":"ac506810-02a2-41f5-af59-04498bbaecb6","resolution":{"observed_at":"2026-08-10T20:33:53.991893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:53.996900Z","title":"Expression, affect, action unit recognition: Aff-wild2, multi-task learning and arcface, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:53.996900Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8c3a2d5dc303f65f01dfd1d6b4c6a47db52d4bed4b71cbb00fdcce34bef6fa7a","observation_id":"b19a64f9-dbfe-445d-8e6b-174955e70bb1","resolution":{"observed_at":"2026-08-10T20:33:53.996900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.000691Z","title":"Afew-va database for valence and arousal estimation in-the-wild","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.000691Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:cce49fc3ae851ac29737b2e0cc8c7f5e68cf9bd8e4c26aac1decd0c09961546b","observation_id":"49941daa-11fa-4191-86d9-ffe12cf973be","resolution":{"observed_at":"2026-08-10T20:33:54.000691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.005726Z","title":"Dense-captioning events in videos","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.005726Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:860ae271d35c0f3e2f1bab27bd8d2348cc33b66d68e8121b68816c085bd334c7","observation_id":"61377989-453c-4780-bcc3-f74d63bdc9c6","resolution":{"observed_at":"2026-08-10T20:33:54.005726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.426199Z","title":"Context-aware emotion recognition net- works","venue":null,"work_id":"42ffb5a3-fcda-42b7-b7f6-c3788f1098e0","year":2019},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.016846Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b463f3d9f4081d446de5063580df972335993ff2ffa69ec8433930273447a07d","observation_id":"b64a0f55-59a7-4ea3-b690-8dfe4fe2cbb8","resolution":{"observed_at":"2026-08-10T20:33:55.430420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.411498Z","title":"Llava-next: What else influences visual instruction tun- ing beyond data?, 2024","venue":null,"work_id":"213cd81d-41d5-421a-a420-1da739b20a2d","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.021962Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:cc8d6be56236a3cba72c82166a2fd8780adb1e6a430231fbff5e1029f353dd17","observation_id":"c2632afb-5b1a-4883-b728-12af4c6b4f4e","resolution":{"observed_at":"2026-08-10T20:33:55.416512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-10T20:33:54.026558Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.026558Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e1b3699092de069f3eb3aa586a825a08e34b12353beaf9ab091873d80e287494","observation_id":"7cff97f2-83c7-460b-8ba4-38f3988fdfee","resolution":{"observed_at":"2026-08-10T20:33:54.026558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-08-13T00:09:23.835117Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07895","snapshot_observed_at":"2026-08-10T20:33:54.031175Z","title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.031175Z"},"links":{"cited_paper":"/paper/2407.07895","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e98b75d27ab2e1e5775f8a013798b5da6859e3f9fd67dee6ff363c18b3ec1407","observation_id":"e26760f4-3fcf-4b4e-ac56-e16b76841d14","resolution":{"observed_at":"2026-08-10T20:33:54.031175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.035535Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.035535Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a401e3a375df0565ddad4fae188f46ee3c18b96e25c5fc03e2274e1e2fb0eab6","observation_id":"7f8c1543-a945-4910-af7e-409144b7f88f","resolution":{"observed_at":"2026-08-10T20:33:54.035535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T20:33:54.040489Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.040489Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8b2bbda125e565cde03c1b539e1afe669ba6537dc0952c770dfc7940c807f59b","observation_id":"2e33504e-cdb5-4ea3-9099-cc2a3bfd6e84","resolution":{"observed_at":"2026-08-10T20:33:54.040489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.391841Z","title":"Dual-sti: Dual-path spatial-temporal interac- tion learning for dynamic facial expression recognition","venue":null,"work_id":"4380bc23-76d7-45a2-89b3-eeb215afc379","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.045144Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:080701de29c81385f9c899a485fb01ab286bc6fe006d1dc4bd9c5705b010c0e9","observation_id":"9094410c-9abc-4c79-bf0a-6f01cc769b6e","resolution":{"observed_at":"2026-08-10T20:33:55.395528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.380721Z","title":"Facial affective behavior analysis with instruction tuning, 2024","venue":null,"work_id":"dfffeb26-277d-4c0d-b6fa-30315ced2bed","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.049977Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:337f4b5052350510c5db875bcaa0ffd7f5bc7744b37ca75306b022868e32a6fc","observation_id":"b5f712da-2f25-4616-844c-628f256cccd3","resolution":{"observed_at":"2026-08-10T20:33:55.384319Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.368796Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"d3932651-cd36-4165-a6e3-83d36968931f","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.054027Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b340041f00720725a8841ca08495f895efb65db0f2bf1e111238a53a4b4fda28","observation_id":"88837485-f367-4190-a78b-1cc280f44245","resolution":{"observed_at":"2026-08-10T20:33:55.373429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.352855Z","title":"Photomaker: Customizing realistic human photos via stacked id embedding","venue":null,"work_id":"b8a1b969-8583-49b6-aabb-c30b5f2cd349","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.058317Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e9c8e3d2907e8a6e74cbfb43bd41489f979293374efa17fce9be704c2d3e3ca6","observation_id":"231af9a6-5224-4f1f-9aa0-310260037293","resolution":{"observed_at":"2026-08-10T20:33:55.358412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-10T20:33:54.063000Z","title":"Video-llava: Learning united visual represen- tation by alignment before projection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.063000Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b95c6d76f171e4c27ca97bc28dd531c9e3b3f47e974b303beda08018104bf3b5","observation_id":"fa8073b5-3bd1-43cd-963c-2284534b242b","resolution":{"observed_at":"2026-08-10T20:33:54.063000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.337580Z","title":"Saanet: Siamese action-units attention network for improving dynamic facial expression recogni- tion","venue":null,"work_id":"09e02673-8a28-4223-b1db-e8ce0405cefb","year":2020},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.068366Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:081869b0836e0d5eebd21231a0d29160eaeb7ccdddc7d50daba8ca377a24dddf","observation_id":"c2d2eb12-c3d7-4f98-85e0-64962d28268b","resolution":{"observed_at":"2026-08-10T20:33:55.342396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-08-13T06:40:28.574929Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-10T20:33:54.072952Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.072952Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:711704f760ef2ed8f0d3b63f9f448073d489d7da2c43c6b302442181abddcba2","observation_id":"b1ae5ed9-7d2a-466e-baf4-923d799a0625","resolution":{"observed_at":"2026-08-10T20:33:54.072952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.079056Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.079056Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:da52cb11eccb6b103ef2e56580684967b69671cc30fee1d5f1bcfdf65f41c12a","observation_id":"deaaf7f4-3e1b-4922-a469-7e7c51f095d5","resolution":{"observed_at":"2026-08-10T20:33:54.079056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08268","last_updated":"2025-02-03T21:47:31Z","snapshot_observed_at":"2026-08-14T09:16:47.522119Z","submitted_at":"2024-02-13T07:47:36Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08268","snapshot_observed_at":"2026-08-10T20:33:54.083552Z","title":"World model on million-length video and language with ringattention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.083552Z"},"links":{"cited_paper":"/paper/2402.08268","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c3c7e26ac2fbaee81cd06941a4b7d6df65101addce5b3ce64e9d8fd39833a431","observation_id":"2a8b8f70-e38a-45c8-8359-777e250eb065","resolution":{"observed_at":"2026-08-10T20:33:54.083552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15785","last_updated":"2024-06-27T12:05:48Z","snapshot_observed_at":"2026-08-13T10:03:47.236571Z","submitted_at":"2023-09-27T16:58:35Z","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.15785","snapshot_observed_at":"2026-08-10T20:33:54.087918Z","title":"One for all: Video conversation is fea- sible without video instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.087918Z"},"links":{"cited_paper":"/paper/2309.15785","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:3474cff3e3d9a681817aef4564313ee00cb25281a18d87cd9444069624e60ff2","observation_id":"c6196b34-0323-4b0c-a59a-c4796f0303b6","resolution":{"observed_at":"2026-08-10T20:33:54.087918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00308","last_updated":"2024-03-30T10:11:26Z","snapshot_observed_at":"2026-08-13T00:40:59.749269Z","submitted_at":"2024-03-30T10:11:26Z","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00308","snapshot_observed_at":"2026-08-10T20:33:54.092524Z","title":"St-llm: Large language models are effective tem- poral learners","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.092524Z"},"links":{"cited_paper":"/paper/2404.00308","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e0b3dca91e507e051aecfbc63b2d5ddb38d7c8c3fb8e4cdbae6248f091ea6362","observation_id":"4d075e73-3a34-4aa7-874c-120ee1be9681","resolution":{"observed_at":"2026-08-10T20:33:54.092524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.317001Z","title":"Mafw: A large-scale, multi-modal, compound affective database for dynamic facial expression recognition in the wild, 2023","venue":null,"work_id":"09685e6b-d522-4d3f-893e-6a4c2e4c8546","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.098492Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c45cfe5e0fae2606cfc99f1a7eb36d5ce9421f05eb2373bc23757081e81b7892","observation_id":"803469f6-deef-4297-b87b-92833edf75ad","resolution":{"observed_at":"2026-08-10T20:33:55.321731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.305593Z","title":"DamoFD: Digging into backbone de- sign on face detection","venue":null,"work_id":"37a3b952-f17f-4090-8baa-ef331d1c52b9","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.102402Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:4c51ef2b7aadab8fb520875e7fa3454b9ca9889c9f902bd472be061d277cb263","observation_id":"e8f43fe3-bbf1-4cc1-bce9-5e410716a276","resolution":{"observed_at":"2026-08-10T20:33:55.309781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.294021Z","title":"Livingstone and Frank A","venue":null,"work_id":"05d57c00-64d1-4529-800b-a3ee5db3f76e","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.109723Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e317e80d513185947850c201859e130a634a63cd68c874e0c7a80a0b4de0f076","observation_id":"1a21361f-515c-4376-a7f1-c07d45c73d82","resolution":{"observed_at":"2026-08-10T20:33:55.297610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.281581Z","title":"Cohn, Takeo Kanade, Jason Saragih, Zara Ambadar, and Iain Matthews","venue":null,"work_id":"27715c5a-109f-4d61-8c27-2aadaa705023","year":2010},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.114235Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:7825701c0caee05d4d194d4ef1ccddc6a0d0b121456e4f8d8facd44305bdb432","observation_id":"c9ccb828-d05c-4b47-ae5e-5b61b57bade5","resolution":{"observed_at":"2026-08-10T20:33:55.286051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.268278Z","title":"The extended cohn- kanade dataset (ck+): A complete dataset for action unit and emotion-specified expression","venue":null,"work_id":"4cc08af6-88b2-4234-98eb-dead70d2e3b4","year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.118502Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:15cbb02d41daacb5738b6a16da0e1fb94583b41c07ad3e16219e066756976227","observation_id":"2d697a4f-8b27-4320-b587-e8742ea614d5","resolution":{"observed_at":"2026-08-10T20:33:55.273230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.253957Z","title":"Learning multi-dimensional edge feature- based au relation graph for facial action unit recognition","venue":null,"work_id":"64f61333-c68f-4a5b-84db-f698d90b2305","year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.123016Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:6955107bd67dbe4d020ddd97fe9a5333ddd9ff04ac42fb5b30d1b772fe5d9875","observation_id":"b5c470e0-abcf-40aa-ba94-8c820facbbd1","resolution":{"observed_at":"2026-08-10T20:33:55.257974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-10T20:33:54.127444Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.127444Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:336a3c995cf210c2caf107731c2af62b9385e06298b4e2163928fa8047187edc","observation_id":"d0e643a4-7903-4f7c-91ee-b39f6e6d84e9","resolution":{"observed_at":"2026-08-10T20:33:54.127444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.242003Z","title":"Video-chatgpt: Towards detailed video 10 understanding via large vision and language models, 2024","venue":null,"work_id":"117f505e-6b26-4217-a4ae-09a3c7b73769","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.131730Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c461967cbae13bc0319b7c65a1430b4233ff4d41809f2a0ba915a40547d475ab","observation_id":"8ffd7c41-e2d5-4641-a30c-47452385392f","resolution":{"observed_at":"2026-08-10T20:33:55.246121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.135546Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.135546Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:3646288547f9d5bb8efafae35fff901849a5a6138e507af4733a63838274ff6a","observation_id":"e82a982a-ec88-45ab-9a48-d0a347527cc1","resolution":{"observed_at":"2026-08-10T20:33:54.135546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.222997Z","title":"The importance of emotional regulation in mental health","venue":null,"work_id":"6132470b-0272-42fc-a6cf-9e10333cf85b","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.139183Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:95690056803dcef431f14779c8aeadece7c9ee45e34e3b00a83c45e4bbfbaeaf","observation_id":"710122ae-8fba-4ba8-aa6f-3f5913cf3f7b","resolution":{"observed_at":"2026-08-10T20:33:55.227238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12960","last_updated":"2025-03-10T17:08:19Z","snapshot_observed_at":"2026-08-13T00:50:05.841672Z","submitted_at":"2024-03-19T17:58:04Z","title":"FaceXFormer: A Unified Transformer for Facial Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12960","snapshot_observed_at":"2026-08-10T20:33:54.143445Z","title":"Facexformer: A unified transformer for fa- cial analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.143445Z"},"links":{"cited_paper":"/paper/2403.12960","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:dd8b6418c6bcdeb77cd949fde5a6b9090db92277558774a845d25ba00e90c26e","observation_id":"719aa5fe-2e0a-4e13-9845-7ea75642b62f","resolution":{"observed_at":"2026-08-10T20:33:54.143445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.211652Z","title":"Repre- sentation learning and identity adversarial training for facial behavior understanding, 2024","venue":null,"work_id":"c765be70-4c0a-4ec5-97ac-28955d1ba679","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.148000Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:2b482df678fd2afb52c15cb2b4b847488fee616c6b6a647e383d09e493db7ef8","observation_id":"9f37c381-ba5c-4eb6-8586-9ccfaf984da2","resolution":{"observed_at":"2026-08-10T20:33:55.215683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.198643Z","title":null,"venue":null,"work_id":"e3785db6-fed0-41b0-8262-81bb22984a74","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.152472Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:fe9c55c346c65e96068038ee9c81321243260461e2adb80de88104ad50e488cd","observation_id":"d662422b-4de8-4bca-bc29-6399216d3fe9","resolution":{"observed_at":"2026-08-10T20:33:55.202907Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.185831Z","title":"Gpt-4v(ision) system card","venue":null,"work_id":"20f6d859-4610-495f-9930-bec31e7468f1","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.156778Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:bb98f45b373ade5e0b9b5c79a65f4aee023a0fb38fadb295b366eee202918857","observation_id":"f6f9678f-c6e2-4ba8-b2e7-9df8a433944e","resolution":{"observed_at":"2026-08-10T20:33:55.189876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.174542Z","title":"Gpt-4 technical report, 2023","venue":null,"work_id":"b9f90daa-3c2e-4588-b377-2094993be61b","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.160875Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:4b2d8c12bcef07fd3be197dd9554d722e5f7c39168d8b3237f0b9f907259b19b","observation_id":"feebdaca-2d58-4d68-b35f-94a596c1ce9d","resolution":{"observed_at":"2026-08-10T20:33:55.178240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.161739Z","title":"Gpt-4o system card, 2024","venue":null,"work_id":"aef30f49-f526-4bd4-a9f5-21eb4f8a00c4","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.167146Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a963a2d11a8e39f3cc78ddfde4b9a807f2f8b11387a21c1815cbb2b8fd1f700b","observation_id":"68ab66d2-4688-4f8d-8a79-199d155881f5","resolution":{"observed_at":"2026-08-10T20:33:55.166879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.148400Z","title":"Digihuman: A con- versational digital human with facial expressions","venue":null,"work_id":"8ea23ae4-3110-436b-aae0-3dd8ee403771","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.171845Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e50849fabceb0642717f8d121435e43c21fe9d835a05ffb04a29bc68fa687703","observation_id":"fed2f313-d31c-47f1-a744-9a6b3e0308f9","resolution":{"observed_at":"2026-08-10T20:33:55.152984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.136614Z","title":"Pantic, M","venue":null,"work_id":"32e709de-8d95-42c6-a626-fc33901e8655","year":2005},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.176751Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c0754b45bf3a23660de9835fcebb57a24c0f9e407b14af43caa7e2729a2473f9","observation_id":"1d57b4a1-9157-4f8f-a53c-d32938b2a18e","resolution":{"observed_at":"2026-08-10T20:33:55.140334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.181114Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.181114Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c773bbf13d03deb0c246c233bfa598f4db5087f0f06e0808ac1daf04b447e7c1","observation_id":"5af1b47d-f09a-4c75-9ece-bce449cfd8f5","resolution":{"observed_at":"2026-08-10T20:33:54.181114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.118423Z","title":"Movie description","venue":null,"work_id":"fc6ebef7-cc24-492c-9230-dc34f5cdd816","year":2017},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.186009Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c501372b265c1222da72f3ad1b7c69405e629b3f589b0e4ec75fb4f82daa7014","observation_id":"113794d7-32e1-42f4-adb7-8d05ccabfe0a","resolution":{"observed_at":"2026-08-10T20:33:55.121966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.107629Z","title":"Multi-view dynamic facial action unit detection","venue":null,"work_id":"ebefc33d-af95-4c8a-a543-be51ba7d1b88","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.190865Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:98a6a98c7d308381f3bb20fd0cb33d79c07a1b2cb8fc4349c0032239bca07a80","observation_id":"d423450c-b4bf-45a3-a3a2-63e59e6853fb","resolution":{"observed_at":"2026-08-10T20:33:55.111532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.093500Z","title":"Deep adaptive attention for joint facial action unit detection and face alignment","venue":null,"work_id":"0a377fc1-aeac-4496-9269-dcac6c487b77","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.195507Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:60b056dc2e6cb5cddf79cce52658b3e8033db9d855b712f790e601ba787d4436","observation_id":"adc5adc1-c310-4d96-94c4-3e30f01709c8","resolution":{"observed_at":"2026-08-10T20:33:55.098590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.080982Z","title":"Driver’s emotion and behavior classification system based on internet of things and deep learning for advanced driver assistance system (adas)","venue":null,"work_id":"34494831-7f9c-44c3-9102-db82ed84f75b","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.200363Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:f4798687ccf565ba3a3f637109583d333f72a5c41bf54d90b77de60aedb5c2f9","observation_id":"d57afeef-1816-4ffa-ab42-5dd4bc8d113b","resolution":{"observed_at":"2026-08-10T20:33:55.084813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.070258Z","title":"Gemini: A family of highly capable multi- modal models, 2024","venue":null,"work_id":"74eee480-5326-4af6-9aac-2de481c40bca","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.204861Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:25ae5ae829e1041d63c3a85940a99bc19fdc54e6301ebe3e1c9fb3d11bb4bdca","observation_id":"e9a31929-d21b-4728-9114-4d844a9aa896","resolution":{"observed_at":"2026-08-10T20:33:55.073989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.060456Z","title":"Qwen2.5: A party of foundation models, 2024","venue":null,"work_id":"360e6452-83ca-4535-8ea1-d27e14629222","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.209214Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:827ac2f10370add21a7941ad358f311114f920c841acc569d5afd25d7bb1953a","observation_id":"2954039a-fbb4-45d5-a87f-b04dd8aaa782","resolution":{"observed_at":"2026-08-10T20:33:55.063870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.050036Z","title":"Induced disgust, hap- piness and surprise: an addition to the mmi facial expres- sion database","venue":null,"work_id":"52b22bbe-f8a2-4466-9995-7155c75eb003","year":2010},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.213735Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9e017b29f94b0eacde390947ed69b6f57aba0c4a6c71786c30385fd9c92e4634","observation_id":"e2ee8a76-9ba8-42df-b627-373ccf0dfbdd","resolution":{"observed_at":"2026-08-10T20:33:55.053978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.038000Z","title":"Cider: Consensus-based image description evalua- tion","venue":null,"work_id":"848b0105-3e1a-4942-bc00-0b924cde8f2a","year":2015},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.218449Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:8ecd7a936c7a11055660bdd4b31243fbdc074cd8b28d13dd42b388cc602d241b","observation_id":"96b16d77-d1c1-42b2-9937-6c6580196ac4","resolution":{"observed_at":"2026-08-10T20:33:55.042966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.024743Z","title":"A survey on the pipeline evolution of facial capture and tracking for digital humans","venue":null,"work_id":"087ee612-73cb-47e0-abe6-7cbb26338b28","year":1917},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.222398Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:fc6f96b55d67971965c9c2caeca7b2717c68ee43962a85721e3dfc1a1b59903a","observation_id":"fe9dd5c5-1566-4bc3-808d-f1c7e92a8f29","resolution":{"observed_at":"2026-08-10T20:33:55.028702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.013998Z","title":"Gross, Kristina H ¨o¨ok, Regan Mandryk, and Petr Slovak","venue":null,"work_id":"dac742cd-dc5b-475c-9521-98accb48f591","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.226871Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:c697fe1cc4c56fc54910dd2a85933aa1e2cbfb07809c3bc0c9878d6839c6e955","observation_id":"f278228a-cd22-4c69-906e-39655cb90a70","resolution":{"observed_at":"2026-08-10T20:33:55.017717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:55.003526Z","title":"Tarsier: Recipes for training and evaluating large video description models, 2024","venue":null,"work_id":"52d26679-9b51-4354-acec-d2558477530d","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.231149Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:caf3d111625f76a3776554bff1b16828ded979908065d24540035a5217b4623c","observation_id":"0b0fa2dc-26a1-483e-b33e-485eb0196ed9","resolution":{"observed_at":"2026-08-10T20:33:55.007121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-10T20:33:54.235456Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.235456Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:d47babb1d91f8f2934ae7cf47aeb1c5012adf4fbba691028d80c953619b0f463","observation_id":"329bd536-d2db-4793-a525-e6978bf8566a","resolution":{"observed_at":"2026-08-10T20:33:54.235456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.992904Z","title":"Vatex: A large-scale, high- quality multilingual dataset for video-and-language research","venue":null,"work_id":"39e8ac35-5ad1-49f8-85af-507dadcf31a1","year":2019},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.239911Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e87b6aad45271fea74a3827973c0fdc09dd4a57e287aca54a8e5dee820223e2b","observation_id":"f7cb10f3-2000-433b-95c5-bdfcf0a0d9a0","resolution":{"observed_at":"2026-08-10T20:33:54.996676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.858284Z","title":"Ferv39k: A large-scale multi-scene dataset for fa- cial expression recognition in videos, 2022","venue":null,"work_id":"1e70381d-a8ff-4241-980c-1044912c2d1e","year":2022},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.244178Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:a04ec964d3384441e0a9fc6382d5e1d368143f8575cbcf3eaea0498ebd2b2871","observation_id":"8365d89e-3c8f-480b-913e-31a5ee19eee5","resolution":{"observed_at":"2026-08-10T20:33:54.985798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-13T00:47:41.087976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-10T20:33:54.248879Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.248879Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:934e5cebccc66fe1780f84e0b9c4fa0b970dbb5b8c90aa29bc7fe5e985580be4","observation_id":"17a4cd68-123f-4ab6-a9b6-59c5b94b9864","resolution":{"observed_at":"2026-08-10T20:33:54.248879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.253320Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.253320Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:618bcddda56af3394f20380c907fee2b4399b6274c7fba15006cf20eb2424f5b","observation_id":"78a57c11-57c5-49b9-a415-014a2697bd64","resolution":{"observed_at":"2026-08-10T20:33:54.253320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.838544Z","title":"Pllava : Parameter-free llava extension from images to videos for video dense captioning, 2024","venue":null,"work_id":"729e22ca-2acc-4d45-b097-fab5ae936613","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.258279Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:27a158142d52d98d2b0be3cef30190d119c041f69000e7b29989884a4d75ba20","observation_id":"b9baf577-4213-4296-898f-e4561352466e","resolution":{"observed_at":"2026-08-10T20:33:54.842118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.262867Z","title":"xgen-mm (blip-3): A family of open large multimodal models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.262867Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:4d9bb248d2a0618598307868b2b6f6edbc88507843abf9c8bb4581a0fc7e8ab6","observation_id":"b86eb911-55bb-4a55-ad71-570e44686c28","resolution":{"observed_at":"2026-08-10T20:33:54.262867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-08-13T08:24:19.015641Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-10T20:33:54.270705Z","title":"mplug-owl: Modularization empowers large language models with multimodality","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.270705Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:05066c1aa25c6c8887dee1f75116e7d760327a65e939710863566ab541e1ed47","observation_id":"a80bd366-e3b8-4c07-bc84-872b9aa404ef","resolution":{"observed_at":"2026-08-10T20:33:54.270705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.811947Z","title":"mplug- owl2: Revolutionizing multi-modal large language model with modality collaboration, 2023","venue":null,"work_id":"a93893c5-c855-47e9-8c2d-f9855faad9be","year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.274065Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:bebcbf78e89530913d081100c54a40ef0981b597c6fdf902b5ef25f2c1fbbf21","observation_id":"7120bbcc-1364-4dbf-9b61-41dc9cd72a46","resolution":{"observed_at":"2026-08-10T20:33:54.814951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.802725Z","title":"Spatio-temporal convolutional features with nested lstm for facial expression recognition.Neurocomputing, 317: 50–57, 2018","venue":null,"work_id":"3f8777c7-379e-4e02-b4b7-bf4db8a853d4","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.277803Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:4f8629afce2ef15b5108cbfd85f3b6d264762c5c4b028b519784a2a647b781fd","observation_id":"99cbcbe2-93a1-4092-83f4-6ed80486ba15","resolution":{"observed_at":"2026-08-10T20:33:54.805828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.791938Z","title":"Auformer: Vision transformers are parameter-efficient facial action unit detectors","venue":null,"work_id":"2887f477-3f95-4f2f-8fd9-b88d25cb3940","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.281283Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e40509e471bb0a506d1f6a583deaef76c8fa0a45e14cfeb98c494490d99c07e2","observation_id":"1d5ac5bb-0e4d-495e-9633-9bcc71042d74","resolution":{"observed_at":"2026-08-10T20:33:54.796041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-08-13T15:50:38.254753Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-10T20:33:54.285100Z","title":"Video-llama: An instruction-tuned audio-visual language model for video un- derstanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.285100Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e4bd228242802a8ec3bb66895d65a329dad57517588576948c2c14a702610718","observation_id":"262d4719-76df-40e9-88b2-d0e486e2a57e","resolution":{"observed_at":"2026-08-10T20:33:54.285100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15105","last_updated":"2023-03-27T11:13:50Z","snapshot_observed_at":"2026-08-13T12:15:46.016083Z","submitted_at":"2023-03-27T11:13:50Z","title":"Vision Transformer with Quadrangle Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15105","snapshot_observed_at":"2026-08-10T20:33:54.289229Z","title":"Vi- sion transformer with quadrangle attention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.289229Z"},"links":{"cited_paper":"/paper/2303.15105","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:b368b1f49bc1ab63d60b9342c390aa26012ce0ab402669209d81d7556edcd833","observation_id":"1ffd1b68-9ee3-4d03-8634-e20268023f95","resolution":{"observed_at":"2026-08-10T20:33:54.289229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.782213Z","title":"Cohn, Shaun Canavan, Michael Reale, Andy Horowitz, Peng Liu, and Jeffrey M","venue":null,"work_id":"18cbc492-a5ef-486b-80b2-3e99ad98c629","year":2014},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.294736Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:9d3434b755fde7fc329808a31231cecb3adce5aa6c80046f0fa1358d5c4a6707","observation_id":"9812e6b7-9531-4f92-aa79-f3b95bdc06a8","resolution":{"observed_at":"2026-08-10T20:33:54.785326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.771101Z","title":"Llava- next: A strong zero-shot video understanding model, 2024","venue":null,"work_id":"535a8468-c43a-47be-b100-cf75a24d8a03","year":2024},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.298057Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:e03780143c14531eff41e4819cb229b3fe57fdf36c2c5ceb400cb919dbc82222","observation_id":"ca82cb7d-5634-4986-921c-5a23b8072967","resolution":{"observed_at":"2026-08-10T20:33:54.774625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.759763Z","title":"Facial expression recognition from near- infrared videos","venue":null,"work_id":"279adfd5-ace0-4bf2-ae94-db72e9cdc4c0","year":2011},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.301613Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:0a8671dd433e17b16ea5c82d340bd2176087bd76d85ee937544a767de42be8e1","observation_id":"e265d731-5a5a-48a6-beac-ca464787be71","resolution":{"observed_at":"2026-08-10T20:33:54.764001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-08-12T23:41:31.711979Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05067","snapshot_observed_at":"2026-08-10T20:33:54.304945Z","title":"Llava-octopus: Unlocking instruction-driven adaptive projector fusion for video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.304945Z"},"links":{"cited_paper":"/paper/2501.05067","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:18a744425b326ae3cbac46ec3cb6849e564b05feaaec00f226968395a92503e2","observation_id":"eace1683-3784-4d4d-a46a-932b0a4d921c","resolution":{"observed_at":"2026-08-10T20:33:54.304945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.747361Z","title":"Deep region and multi-label learning for facial action unit detec- tion","venue":null,"work_id":"d4a6c64a-0d37-4491-856a-9344ae3f71c7","year":2016},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.308759Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:2d365ea6ef66137aa9d862251187f5c5df719e8cd914205c287a7808cb7481c9","observation_id":"14e9c2b1-9476-4ab0-b54c-3afe638013e0","resolution":{"observed_at":"2026-08-10T20:33:54.751577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-10T20:33:54.312278Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.312278Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:2723930ced692076c2fb604a0a9be598c18ccf19a4b0376ac68c705eb22a6d73","observation_id":"6d9ec7d2-adf8-4084-9f4a-740fce5b0c11","resolution":{"observed_at":"2026-08-10T20:33:54.312278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:33:54.732859Z","title":"Towards automatic learning of procedures from web instructional videos","venue":null,"work_id":"1b03852d-3ad5-4ade-9fd6-786c22fa8c04","year":2018},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.316060Z"},"links":{"citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:ce47333ba4b66f8b879192fef6fb6fe17dab8f31dbe4fc381f4442746644e17c","observation_id":"2276db77-201b-497d-a22f-600d3dd25b63","resolution":{"observed_at":"2026-08-10T20:33:54.739302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-10T20:33:54.320561Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.320561Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:7dde0ccb29d9610015235debd570043bf4c5e63277a917661e5f915a2c52f421","observation_id":"30e8f1e5-610c-4d29-a881-238bc6134e5d","resolution":{"observed_at":"2026-08-10T20:33:54.320561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T08:09:35.736153Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness"},"reference_resolution":{"displayed":95,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":52,"verified_exact":0,"verified_fuzzy":43},"total_outbound_references":95},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 95 of 95 outbound references and 4 inbound Pith citation observations for arXiv:2501.07978."}