{"as_of":"2026-08-12T16:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:02f81b18dea1dc7bbadfba580ef07ce23caa8b7b1d7e40bdd80348bb10cec60f","coverage":[{"denominator":94,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":94,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:07:27.714660Z","state":"measured"},{"denominator":94,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":94,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.21015/citation-record","integrity":"/paper/2507.21015/integrity","json":"/paper/2507.21015/citation-record.json","paper":"/paper/2507.21015"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.592232Z","title":"Society of mind","venue":null,"work_id":null,"year":1986},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.592232Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ce80560aa741180f57e5059bf9a4651e8e81a79b87f60ada84dbc8e68c14ed75","observation_id":"ef8048ad-0f81-48e2-95bd-6228b7193da3","resolution":{"observed_at":"2026-08-06T13:07:26.592232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.661748Z","title":"Emotion recognition in human-computer interaction","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.661748Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f0cc267f98c47f8792145f9e5a8062153a922a1dbb8d8024623220e06990bcb3","observation_id":"83c82526-bf1a-4ee5-a186-383f493b7c67","resolution":{"observed_at":"2026-08-06T13:07:26.661748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.741061Z","title":"An overview of emotion in artificial intelligence","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.741061Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:719b44ad2b9d80d7d9b351c705612b5ee6b4fe06e5c219876b141f69a4e887c5","observation_id":"f491c3aa-26ad-479e-9029-d58d3660f428","resolution":{"observed_at":"2026-08-06T13:07:26.741061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.793886Z","title":"Deep facial expression recognition: A survey","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.793886Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:e4a9a540f13f4b6b7a12c75e6fda101c3e933a63f74011d1c4864f293e7567fd","observation_id":"e807f6d5-3610-46d3-b508-83399eee5a45","resolution":{"observed_at":"2026-08-06T13:07:26.793886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.873360Z","title":"A survey on facial emotion recognition techniques: A state-of-the-art literature review","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.873360Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8b5fb12db1f3d6e7e9a5e9543621ac18e9f169d91c1b31b80db179839d2f99ff","observation_id":"1dba552a-29c9-4621-8fd8-d36930e6142a","resolution":{"observed_at":"2026-08-06T13:07:26.873360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:26.940239Z","title":"Understanding deep learning techniques for recognition of human emotions using facial expressions: A comprehensive survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:26.940239Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9ec05613cd14cc5e8510a854a9299ce1e270bde0fab99ee74524cbe33bcea2e3","observation_id":"a0818c7a-b929-4123-a7ab-abe8debb2fcc","resolution":{"observed_at":"2026-08-06T13:07:26.940239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.056038Z","title":"Facial micro-expressions: An overview","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.056038Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9a04722c36e9250490cc702db5d4efb234a88848664707501b7d9adc522d2198","observation_id":"1a786f0b-de19-4653-b951-d668d2a57ee0","resolution":{"observed_at":"2026-08-06T13:07:27.056038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.176560Z","title":"A model of the perception of facial expressions of emotion by humans: Research overview and perspectives","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.176560Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:b83cf3936d056d1bccd827266889b4bd0e5c5ac2e605b2bf547f13deccd2c5b3","observation_id":"2c240e3f-d6e6-4c7d-abab-3ca7006dbceb","resolution":{"observed_at":"2026-08-06T13:07:27.176560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.230927Z","title":"Deep learning for human affect recognition: Insights and new developments","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.230927Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:bcaa35a39d41d0c82bfaf49344af394e13044abbb738d43cb01173e50cdae3e0","observation_id":"d97ce718-b408-409d-902e-c023dde91c0f","resolution":{"observed_at":"2026-08-06T13:07:27.230927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.295239Z","title":"A review of affective computing: From unimodal analysis to multimodal fusion","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.295239Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:52e18adc713600b49e8716e3082d912da24c23f83a437da477a9ed78850e826d","observation_id":"dd551dc5-478a-40a1-ac56-ad3a6f194767","resolution":{"observed_at":"2026-08-06T13:07:27.295239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.299794Z","title":"An argument for basic emotions","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.299794Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:256f85f94c68cb6ff064fedbc13c4d07ac9d2ff68d12f64d23732635d34667a1","observation_id":"a13612c8-151f-425a-bf05-fd6c844a98a8","resolution":{"observed_at":"2026-08-06T13:07:27.299794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.304830Z","title":"A circumplex model of affect","venue":null,"work_id":null,"year":1980},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.304830Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:a8eb2eed423eb20eef5120117dce78b5da5626eadf4be6639600bd52f0098c78","observation_id":"e43912a0-793e-476a-ab88-55e0bc60b856","resolution":{"observed_at":"2026-08-06T13:07:27.304830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01495","last_updated":"2025-05-07T13:05:04Z","snapshot_observed_at":"2026-08-10T07:07:35.791983Z","submitted_at":"2024-10-02T12:45:09Z","title":"OV-MER: Towards Open-Vocabulary Multimodal Emotion Recognition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01495","snapshot_observed_at":"2026-08-06T13:07:27.309698Z","title":"Open-vocabulary multimodal emotion recognition: Dataset, metric, and benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.309698Z"},"links":{"cited_paper":"/paper/2410.01495","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:4aca3a92a5de3eda00a29697cf0d47c90f29ace4c1abb974c1eaf72a8a63dac6","observation_id":"b75452e2-5d5f-492f-b21f-126db4a70919","resolution":{"observed_at":"2026-08-06T13:07:27.309698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-06T13:07:27.314585Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.314585Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8c9bd0bbe2b79b4d2c26923ed7c362a9d79d8f2ceccc94c898a01b712294de19","observation_id":"8450f0b3-3ee7-4861-a546-d95f9703d13b","resolution":{"observed_at":"2026-08-06T13:07:27.314585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.320211Z","title":"Self-report captures 27 distinct categories of emotion bridged by continuous gradients","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.320211Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9af0318b86d091b0f94277104849b370f852cc1f38863c0e542e1abf33abf108","observation_id":"3f4eb40c-0771-4fde-b84e-45e9ef730be2","resolution":{"observed_at":"2026-08-06T13:07:27.320211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.324876Z","title":"The language of emotion","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.324876Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:c675e789682938ec7fe0a849aa68463ee98120949fc4b32774506cab72a58468","observation_id":"8e58cf6f-20f1-4b0b-bed8-4ba023332c5e","resolution":{"observed_at":"2026-08-06T13:07:27.324876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.186200Z","title":"The role of language in emotion: Predictions from psychological constructionism","venue":null,"work_id":"df1658be-f5fa-4a4b-a173-b6f9c1f20c5f","year":2015},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.329266Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0fb30e4f8fef5f2ffe899d6dda84fc0e14cda8b24359a9b53f42f66fbe410fae","observation_id":"474ccc17-737c-4a19-b8e1-b08742ecb011","resolution":{"observed_at":"2026-08-06T13:07:29.190794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.170017Z","title":"Describe your facial expressions by linking image encoders and large language models","venue":null,"work_id":"f40598b9-dc32-4b7d-aaef-916bec08925e","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.333575Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ee717da6cba08ff0630cf7d0a34a77b3cede7987c1da6cdaee27333661768001","observation_id":"613824fc-1400-4cf0-ae0d-3f896e0f5bce","resolution":{"observed_at":"2026-08-06T13:07:29.174981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.153389Z","title":"Facial affective behavior analysis with instruction tuning","venue":null,"work_id":"c25d6924-36aa-49cd-b45c-aaa15c5f9611","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.337588Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:e783aa9e1242c82ad9ecca01d3154c77c878d35218a818a250b8aed449807f64","observation_id":"6747419b-2eb9-4c31-8d2e-fbce8c61250f","resolution":{"observed_at":"2026-08-06T13:07:29.158768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.342458Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.342458Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:071a34da96c19683f88621cf511863ee2695dc931bc32e6be8ea341ff31ea57c","observation_id":"00e97ee9-0119-45ec-8c4d-9cfcaf6414f9","resolution":{"observed_at":"2026-08-06T13:07:27.342458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.128004Z","title":"Emoclip: A vision-language method for zero-shot video facial expression recognition","venue":null,"work_id":"b856731f-ae95-48f6-8cce-f7eb58107f49","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.346762Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5cd24d66eebdb902c2c3ed2ad92f4e61b2aada03821bc88de64877945b0e273f","observation_id":"27759557-c862-47df-b6a3-0975c4adcc68","resolution":{"observed_at":"2026-08-06T13:07:29.132614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.113077Z","title":"Flip-80m: 80 million visual-linguistic pairs for facial language-image pre-training","venue":null,"work_id":"7859a851-e8bf-4f21-9daa-cf1679292132","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.350761Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:10a48e95b79425b54c80358792312ee8c5a3a3a1c7c9705d3ea34ffa15e27cf4","observation_id":"63b2af78-2d5c-48e6-a36e-d6e31ff541ff","resolution":{"observed_at":"2026-08-06T13:07:29.117499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.097446Z","title":"Enhancing zero-shot facial expression recognition by llm knowledge transfer","venue":null,"work_id":"550e0e6a-ebb7-4faf-a9ca-54a329981acd","year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.355611Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:fb99bc1624ed27473ed8aee9785e33550a15f44a7c10d5f6a13b36a46c5dbd49","observation_id":"af6ee1a3-6628-434e-8d7d-7bacdc929a0f","resolution":{"observed_at":"2026-08-06T13:07:29.102546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.360923Z","title":"Facexbench: Evaluating multimodal llms on face understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.360923Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:70493f0141ef74dd39286f3a8f2d180a5f7a92bdc3da3d31fec410bfab684c44","observation_id":"796b78a5-1e88-4291-95cf-94103d0cd1ba","resolution":{"observed_at":"2026-08-06T13:07:27.360923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.365402Z","title":"Face-human-bench: A comprehensive benchmark of face and human understanding for multi-modal assistants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.365402Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8d37f77482dd2aff3b71760a1ea7e10edb246741aae00058d7ce12d50327fc10","observation_id":"101f4a02-2524-4b60-84a9-2fee07ea7bc7","resolution":{"observed_at":"2026-08-06T13:07:27.365402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.370480Z","title":"Gpt-4v with emotion: A zero-shot benchmark for generalized emotion recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.370480Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:7dd38be569ccf48706229c5a94b4f013a071d9f80c91417ebab1478d56780661","observation_id":"9093d8a4-4c21-4974-bc0d-27f203cb14d5","resolution":{"observed_at":"2026-08-06T13:07:27.370480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-06T13:07:27.375535Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.375535Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:17ed85462b2152b973878c5f1f2947f36d25791ec7d498764847c1c25112dc1f","observation_id":"3b546bfb-8e31-4df9-9027-375b17c4b38a","resolution":{"observed_at":"2026-08-06T13:07:27.375535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.071546Z","title":"Occlusion aware facial expression recognition using cnn with attention mechanism","venue":null,"work_id":"d7a8f3da-8418-46df-bba8-b7701dfb94f6","year":2018},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.385225Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:02b9cedf73114e68147aed5ae95087d2d07ac75e2e1c1ce6f6b1aeac8a052404","observation_id":"0d4882c9-0f73-42ab-ba71-9383e160156d","resolution":{"observed_at":"2026-08-06T13:07:29.075874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.055875Z","title":"Region attention networks for pose and occlusion robust facial expression recognition","venue":null,"work_id":"35c1b493-580b-42db-83a2-dbb598dda76a","year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.390540Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9a510733959c2bf66ed489ca5563eb27cd6e3b54fa696d23aecc2ee62c706936","observation_id":"c26d7ed2-1a9b-4b0f-beab-c05733330d07","resolution":{"observed_at":"2026-08-06T13:07:29.061244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.040029Z","title":"Learning deep global multi-scale and local attention features for facial expression recognition in the wild","venue":null,"work_id":"3ba27e78-6747-44a2-a506-2fdd4265db2f","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.395215Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0a6947e19574f8e6305c747cd82cd4307f4da5824eec51edf828b6c230b72956","observation_id":"430b144b-c4bd-4e05-800f-4dd2e61dbb53","resolution":{"observed_at":"2026-08-06T13:07:29.044843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.024850Z","title":"Robust lightweight facial expression recognition network with label distribution training","venue":null,"work_id":"8967b7aa-f1de-4414-b473-ab1eb4490c06","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.399715Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:3cddb39b2caf754db469a1045e769a68443f47e378ba667d3071a45448abbe6b","observation_id":"7e04cd3e-a95e-4130-a8b5-01fc21efedf3","resolution":{"observed_at":"2026-08-06T13:07:29.029929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:29.009538Z","title":"Facial expression recognition with visual transformers and attentional selective fusion","venue":null,"work_id":"97ec0a49-af22-451a-86df-b021f6011508","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.407722Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:200b42f301fde0000a793b50f398a51189554818e5e5689095ec1aeebafaa823","observation_id":"80dd1b37-590b-4991-b264-f22b09697a1a","resolution":{"observed_at":"2026-08-06T13:07:29.014568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.993449Z","title":"Transfer: Learning relation-aware facial expression representations with transformers","venue":null,"work_id":"eeff6080-adbb-49fb-8175-f65da44163cf","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.411851Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:735a255a8d663837485a16147e43d1373cf6e19083e4d291bf9748e6f2271e73","observation_id":"c1ea6df3-43d5-425c-a3df-94839e7e39de","resolution":{"observed_at":"2026-08-06T13:07:28.998550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.974742Z","title":"Poster: A pyramid cross-fusion transformer network for facial expression recognition","venue":null,"work_id":"3723528d-839f-4b15-a3f3-8ef5dad37784","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.415946Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f29b8804b13a799a7032afdd2f26586badde698d290e77cd5377d09aaf37dcae","observation_id":"4dbb4180-0854-43b3-a52d-afac2bfb917e","resolution":{"observed_at":"2026-08-06T13:07:28.979206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.957667Z","title":"Svfap: Self-supervised video facial affect perceiver","venue":null,"work_id":"0175a2c1-37d6-42ff-baab-2da226d4da33","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.420289Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5ed3f054ee103b44741339ad61e6ec3da51734c30384a674828c6605758a6d41","observation_id":"334c6763-05d5-4a4d-b5c1-ded126b9604a","resolution":{"observed_at":"2026-08-06T13:07:28.963118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.938456Z","title":"Poster++: A simpler and stronger facial expression recognition network","venue":null,"work_id":"e8dd462d-78f7-4075-b6fa-efbc981f1dcf","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.424789Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:fcc9b5148e0af0fb02f0d1442a06cbe9d0113d6b049c43bea57e8362bd615792","observation_id":"b831ece9-cd9f-4e18-9f18-1e7875b8a1ed","resolution":{"observed_at":"2026-08-06T13:07:28.943246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.921667Z","title":"Reliable crowdsourcing and deep locality-preserving learning for expression recognition in the wild","venue":null,"work_id":"02c0d7bb-f36a-4028-b954-018a9ac574bf","year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.431081Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:4190cbed03a912e4c718a237bd93ce35ab9cc110f770c24fea8bfa6d1c9781ad","observation_id":"5d933e14-0c5a-4771-8e24-f997f8411e73","resolution":{"observed_at":"2026-08-06T13:07:28.927537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.904804Z","title":"Suppressing uncertainties for large-scale facial expression recognition","venue":null,"work_id":"1bba0894-94ee-4781-b5ea-a57cec950a7e","year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.435495Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:39fcd1dfce9cae44eaac0f5265a191d7cb56dae82e52ef0678b3521cbf3201f6","observation_id":"d4b4681a-434e-4c5f-89b7-9ff5626202a1","resolution":{"observed_at":"2026-08-06T13:07:28.910444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.886278Z","title":"Relative uncertainty learning for facial expression recognition","venue":null,"work_id":"46116927-d49d-4606-b174-4ed549c024d6","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.439812Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0089195d30f36ca52f5f5ceb7318cb93e49caa683bbc41c44fd65a7f63d557a4","observation_id":"d9df4532-1db3-4df1-9c92-b1a861982c28","resolution":{"observed_at":"2026-08-06T13:07:28.892265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.868839Z","title":"Learning emotion representations from verbal and nonverbal communication","venue":null,"work_id":"746a5111-8979-4a26-99f6-0537c1e65fbb","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.444071Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:306176b28514069a4aabbbbf04a0866a03f7e4a107c02d06edeff359b3775bf6","observation_id":"ec11e3d6-aea9-44c9-9e48-c902aa3a5b2a","resolution":{"observed_at":"2026-08-06T13:07:28.874331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.853063Z","title":"Collecting large, richly annotated facial-expression databases from movies","venue":null,"work_id":"13041206-54c2-4249-ab55-f4cc0a16c834","year":2012},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.448368Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9aeaeccd3d06e8d69ca2c1e5fe6b68c1d25141655696ad4041047cf6029345ec","observation_id":"668b0632-e627-4b26-ac6b-6ece05f3dba8","resolution":{"observed_at":"2026-08-06T13:07:28.857788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.837472Z","title":"Training deep networks for facial expression recognition with crowd-sourced label distribution","venue":null,"work_id":"7d76a70c-39d2-4a8a-ba1d-1fdde4986c4a","year":2016},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.453976Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:021ba6fb25c3768defe9783a4b7c61e543a2f780268744a25937f1ade0973bc9","observation_id":"53205689-aba6-4690-b331-ea5483dfff10","resolution":{"observed_at":"2026-08-06T13:07:28.842244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.821293Z","title":"Affectnet: A database for facial expression, valence, and arousal computing in the wild","venue":null,"work_id":"3bcc0b4d-2c01-47b1-a8ca-eac3110e0151","year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.458964Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:2de836b0db5125482cc79e383555324484faf8b0320775407151ad01276c5497","observation_id":"d9730ec3-50fe-4659-b3d4-074cc7ab87f4","resolution":{"observed_at":"2026-08-06T13:07:28.825975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.805419Z","title":"Dfew: A large-scale database for recognizing dynamic facial expressions in the wild","venue":null,"work_id":"f008b22d-952f-4f4c-b6b5-e4f590a14e14","year":2020},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.463896Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8076322dde183aba3d2ac9f755c258b638126fb92a21b4a5cddc57b06113170e","observation_id":"2b92e18b-5709-42bd-8711-922081a8989c","resolution":{"observed_at":"2026-08-06T13:07:28.810550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.789228Z","title":"Mafw: A large-scale, multi-modal, compound affective database for dynamic facial expression recognition in the wild","venue":null,"work_id":"d37623f4-33e2-4d6f-bedd-076a2dbc73b6","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.468537Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:804997d1b18278df6ec3ae30c75b72588d1bf4ab0ec0fe407ecb4af047798727","observation_id":"e5d8ca84-2fa9-4848-bfc6-bfea2b3c7214","resolution":{"observed_at":"2026-08-06T13:07:28.794142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.771854Z","title":"Ferv39k: A large-scale multi-scene dataset for facial expression recognition in videos","venue":null,"work_id":"bc946e7e-a3de-41c0-b533-136a205408a1","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.472829Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ce39a5bc615abf3683399b6d6dbaca118dbea32a73b3312819bcc1bebf29f61a","observation_id":"859685ba-3115-451b-8b6d-2e6b1be7dad7","resolution":{"observed_at":"2026-08-06T13:07:28.777832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1811.07770","last_updated":"2019-12-13T23:44:20Z","snapshot_observed_at":"2026-08-11T22:36:33.782982Z","submitted_at":"2018-11-11T01:57:15Z","title":"Aff-Wild2: Extending the Aff-Wild Database for Affect Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.07770","snapshot_observed_at":"2026-08-06T13:07:27.477365Z","title":"Aff-wild2: Extending the aff-wild database for affect recognition","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.477365Z"},"links":{"cited_paper":"/paper/1811.07770","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:48dcc5e9225eb4d437ba2a1a41c76950ae6a608468bf5544561f4769392fe0bd","observation_id":"a6b0f5c3-de36-4929-8dee-e00c386c8737","resolution":{"observed_at":"2026-08-06T13:07:27.477365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.754395Z","title":"Deep affect prediction in-the-wild: Aff-wild database and challenge, deep architectures, and beyond","venue":null,"work_id":"f19540e3-ece7-4eef-afed-b06f66e67083","year":2019},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.482389Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:6f72f89429cebb867c5f34fdf9dadb3ee6171411678f6d0b3a15dfe74201a462","observation_id":"45a3adf5-cf83-46dc-adcf-6d8fe902d8c9","resolution":{"observed_at":"2026-08-06T13:07:28.760052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.737863Z","title":"Compound facial expressions of emotion","venue":null,"work_id":"3f5afb87-f902-4d0a-944a-e9129d52f9c9","year":2014},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.486532Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:d20ee741b3fb927d3f67fdf2903ed38052eb1aa663321d5d225fd02ffdd03804","observation_id":"68a872fe-c5e7-4beb-b4dc-a0bd6f360814","resolution":{"observed_at":"2026-08-06T13:07:28.742804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.722448Z","title":"Configural information in facial expression perception","venue":null,"work_id":"b584882b-4532-4440-b6be-3f0f9f96fd9d","year":2000},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.491038Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:edff26be510545406f842a8971273afb3c101d825a233662c09762a6b892a4e4","observation_id":"bfcf1974-de71-402c-b1eb-b7261af5129e","resolution":{"observed_at":"2026-08-06T13:07:28.727222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.707820Z","title":"Parts and wholes in expression recognition","venue":null,"work_id":"4ea69ae0-1945-4853-8bf1-383a08eb3917","year":2000},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.496626Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:85ae459988467500bb27c15065c70fd7521e45572a83725d6053ca0b13985976","observation_id":"9c4b2c25-bfde-4039-a2a1-ca4eb145d015","resolution":{"observed_at":"2026-08-06T13:07:28.712447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.692053Z","title":"Mixed emotions: Holistic and analytic perception of facial expressions","venue":null,"work_id":"bc7095a6-c5f9-4127-b29a-f6dd2ca3712c","year":2012},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.501743Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:08e3bae781df70a3591b7b6790c330576e2b3d66fd3a51cedc1c5a3e21a72a70","observation_id":"736555e8-e085-4135-9928-40a153ecaa55","resolution":{"observed_at":"2026-08-06T13:07:28.696994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.675969Z","title":"The role of facial movements in emotion recognition","venue":null,"work_id":"88f0b455-e941-443a-8757-238ddced6e95","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.506388Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8ec4885b4bab43442967d332fd7b26b4acad9f2fcfcd665846c8026bcc9a6b42","observation_id":"c168dbc8-1d94-4a55-baaf-e90bf4d73d38","resolution":{"observed_at":"2026-08-06T13:07:28.680641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.510794Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.510794Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:d90f2954c6c60d422c3a4f010b097a4feb50a7f0dbdcd4b923fda33467f07706","observation_id":"cc17857f-4354-4c3a-afae-b0d9837b3146","resolution":{"observed_at":"2026-08-06T13:07:27.510794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.515559Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.515559Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:7439d30372a423832541383907f30952928045d75b5fd70b47ce9484f5e2ffff","observation_id":"45b1d21e-5166-4104-b63c-945dd671881e","resolution":{"observed_at":"2026-08-06T13:07:27.515559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.519813Z","title":"Reproducible scaling laws for contrastive language-image learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.519813Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f07c0eba794f142f5d7678fdea7f925087023f84dd29cfb20dc13c49f5962a55","observation_id":"543b6e97-02fa-484a-8e72-b90f188ba575","resolution":{"observed_at":"2026-08-06T13:07:27.519813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.524336Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.524336Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:bdbcc0075243bb9074696356702613ec6940a45f6ce5d9f0e98976929b88bfa8","observation_id":"16766737-fb2d-4040-aa01-201cb94a8304","resolution":{"observed_at":"2026-08-06T13:07:27.524336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15389","last_updated":"2023-03-27T17:02:21Z","snapshot_observed_at":"2026-07-06T15:08:34.018146Z","submitted_at":"2023-03-27T17:02:21Z","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15389","snapshot_observed_at":"2026-08-06T13:07:27.528659Z","title":"Eva-clip: Improved training techniques for clip at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.528659Z"},"links":{"cited_paper":"/paper/2303.15389","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:e374631bf5a5a75bbd202f85dc80b37bf92dce8debe5864575d07418cc2e67dd","observation_id":"a58c779b-930f-4c26-a463-bf3d2c91a56a","resolution":{"observed_at":"2026-08-06T13:07:27.528659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.616868Z","title":"Demystifying clip data","venue":null,"work_id":"a06b5de1-25ba-46c7-908e-69976bf0bfec","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.534253Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:527f5c4519d0a7ec141d0243b53d23b83b40aaaf80a7f9ff9da10730b77a998f","observation_id":"e367f1a0-e1a6-41ea-b20d-c15f86c97ebf","resolution":{"observed_at":"2026-08-06T13:07:28.621180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.601141Z","title":"Dreamlip: Language-image pre-training with long captions","venue":null,"work_id":"fb52c6e3-847a-4253-83d4-b0bf37596cce","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.539781Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:6c8d3dea6d942437e29aa22bdaea5614aef118af858e26f747169397a2435446","observation_id":"296cdcf5-98d0-4dd3-a98d-6a1461eba2ee","resolution":{"observed_at":"2026-08-06T13:07:28.605700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00740","last_updated":"2025-03-29T12:57:07Z","snapshot_observed_at":"2026-08-10T02:03:26.217451Z","submitted_at":"2024-04-30T01:19:18Z","title":"Modeling Caption Diversity in Contrastive Vision-Language Pretraining","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00740","snapshot_observed_at":"2026-08-06T13:07:27.544588Z","title":"Modeling caption diversity in contrastive vision-language pretraining","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.544588Z"},"links":{"cited_paper":"/paper/2405.00740","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:be511df78ffe6acd022a0743fc497c0eab524d73b6f4265a9c3bf6429e89a595","observation_id":"172bed96-dc4c-4ace-b307-44b8a6db6d9e","resolution":{"observed_at":"2026-08-06T13:07:27.544588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.585280Z","title":"Improving fine-grained understand- ing in image-text pre-training","venue":null,"work_id":"8a160f06-44b3-41ab-8beb-0660a0c8dcb9","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.549311Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:0edcbe97a7122bfb18e050747eb88fec672f3a8cbdd8fe354ef08be030c933b7","observation_id":"96e16bd9-f574-4353-8097-d51cf7a79905","resolution":{"observed_at":"2026-08-06T13:07:28.590425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.554743Z","title":"General facial representation learning in a visual-linguistic manner","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.554743Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ccaff28e5f8ef2e066c562a844b9b5a5530a997877831be288dd4a507a73d02c","observation_id":"2a162f3a-c265-4461-bf56-c969dce59c59","resolution":{"observed_at":"2026-08-06T13:07:27.554743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T13:07:27.559162Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.559162Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:81b1d33f022138df037cf1af2ac744dda52bbff3eeb3e58ea434f44ffce6ccb5","observation_id":"d1da1679-01f1-4495-bf09-5f9664c2e4ca","resolution":{"observed_at":"2026-08-06T13:07:27.559162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.564018Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36:34892–34916, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.564018Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:c1e773264a79225786b6b0501f157bd8e105b979087d7f4c9c568f9aaece34e9","observation_id":"62611581-b192-4cca-832b-9a3429ff96ba","resolution":{"observed_at":"2026-08-06T13:07:27.564018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.548980Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":"13504f6d-b6bd-4179-882e-e9a0215f7470","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.568575Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:66610c1b9e9c380250aad991bce3ab7025d142b1a39e3d31debf5187e02a753c","observation_id":"18510aaf-16b3-425d-8e0a-92146c5d651d","resolution":{"observed_at":"2026-08-06T13:07:28.553765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.573362Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.573362Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:d9cab1bdec2ba6cf46c6564926a8e37f1905d7930a9ee5808f9ecc2d9931714e","observation_id":"a7fe66eb-2834-4485-abe6-a6200aecf3bf","resolution":{"observed_at":"2026-08-06T13:07:27.573362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-09T21:25:20.369782Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-06T13:07:27.577897Z","title":"Qwen technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.577897Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8ef08cf72b1b4555c5cf49df25cc2c85495e390321b3f3d9968b537168da98ea","observation_id":"0846c629-f9e1-45ee-94ff-fbfa015973c2","resolution":{"observed_at":"2026-08-06T13:07:27.577897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.522421Z","title":"Expllm: Towards chain of thought for facial expression recognition","venue":null,"work_id":"f9803b51-0efd-4cc5-a2f5-fca5b9b22270","year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.583046Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:b4efa646db60c2d274418ff9b2baa21f35dade07f18f2a42d98e21915674ac93","observation_id":"b32689e9-eb62-4a25-9e74-442c05fcc889","resolution":{"observed_at":"2026-08-06T13:07:28.527127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11424","last_updated":"2024-08-21T08:28:40Z","snapshot_observed_at":"2026-08-09T13:16:40.260855Z","submitted_at":"2024-08-21T08:28:40Z","title":"EMO-LLaMA: Enhancing Facial Emotion Understanding with Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.11424","snapshot_observed_at":"2026-08-06T13:07:27.587777Z","title":"Emo-llama: Enhancing facial emotion understanding with instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.587777Z"},"links":{"cited_paper":"/paper/2408.11424","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:515b20e7d1cd4c87eab7936d41009c94cf0356684556ae547df9aa62342d116f","observation_id":"91706ae5-009f-4165-9389-758e3f05e0f2","resolution":{"observed_at":"2026-08-06T13:07:27.587777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.505529Z","title":"Emotion-llama: Multimodal emotion recognition and reasoning with instruction tuning","venue":null,"work_id":"8b40948f-c1ee-4880-99fc-d686c3181e32","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.592827Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:bd79b405330b47697c1095aa99632c32f97f844dc555f5a918ba424680e3653c","observation_id":"ad3f9e85-5b32-4f95-bb8f-4e6f97606552","resolution":{"observed_at":"2026-08-06T13:07:28.510694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16566","last_updated":"2025-05-07T13:20:08Z","snapshot_observed_at":"2026-08-10T12:13:11.254896Z","submitted_at":"2025-01-27T23:18:39Z","title":"AffectGPT: A New Dataset, Model, and Benchmark for Emotion Understanding with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16566","snapshot_observed_at":"2026-08-06T13:07:27.597574Z","title":"Affectgpt: A new dataset, model, and benchmark for emotion understanding with multimodal large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.597574Z"},"links":{"cited_paper":"/paper/2501.16566","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:62d9154d576fae6f5b6b0a0596ce45f43fe4d71f15120f62cdaf046ddc50e6a5","observation_id":"9ff15cd6-87af-4345-9f2f-47c7635f8d50","resolution":{"observed_at":"2026-08-06T13:07:27.597574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.05379","last_updated":"2025-03-10T07:11:14Z","snapshot_observed_at":"2026-08-07T21:57:26.595540Z","submitted_at":"2025-03-07T12:46:42Z","title":"R1-Omni: Explainable Omni-Multimodal Emotion Recognition with Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.05379","snapshot_observed_at":"2026-08-06T13:07:27.602296Z","title":"R1-omni: Explainable omni-multimodal emotion recognition with reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.602296Z"},"links":{"cited_paper":"/paper/2503.05379","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:86f8796b6afed30f81be6ab82c6317c30410389ebfc61a542d1e8ebf90dffcf4","observation_id":"dcd2a90e-57dd-493c-898a-6e66e0d1b598","resolution":{"observed_at":"2026-08-06T13:07:27.602296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.487200Z","title":"Generative adversarial network for text-to-face synthesis and manipulation with pretrained bert model","venue":null,"work_id":"988fa545-ac9e-4216-a648-38e0f1c12ab6","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.607503Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:a067016114da9d1d9576594c35076b221f05038fe8687d8e588658a2198e1bfc","observation_id":"372f1b4f-0d2c-4d48-9411-8e24d1f8fb4e","resolution":{"observed_at":"2026-08-06T13:07:28.493528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.611815Z","title":"Tedigan: Text-guided diverse face image generation and manipulation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.611815Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:f60b1e97be61c0e061988f3e1ed8be25ee4cfdce70acd0fdbe345300a2b5e03d","observation_id":"d620aab3-8df8-4240-bf39-ae514f932d14","resolution":{"observed_at":"2026-08-06T13:07:27.611815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.458103Z","title":"Talk-to-edit: Fine-grained facial editing via dialog","venue":null,"work_id":"19d9ab30-aa07-4e1e-9a3d-4d6c1f58249d","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.616335Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:b254200bcb635dde94b557ee5559d6070885eb248d64468aebeb68ed749f7531","observation_id":"bd60d693-e329-40fb-874e-ffc93d0969ba","resolution":{"observed_at":"2026-08-06T13:07:28.463190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08515","last_updated":"2024-07-12T01:19:33Z","snapshot_observed_at":"2026-08-02T05:39:29.403958Z","submitted_at":"2024-07-11T14:00:14Z","title":"15M Multimodal Facial Image-Text Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08515","snapshot_observed_at":"2026-08-06T13:07:27.621712Z","title":"15m multimodal facial image-text dataset","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.621712Z"},"links":{"cited_paper":"/paper/2407.08515","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:bcb479531b46995e080b96a928e5fe2dd21dff366548004c1f126597399eea42","observation_id":"c1535cd7-1311-4ab3-a765-abb76371da81","resolution":{"observed_at":"2026-08-06T13:07:27.621712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08824","last_updated":"2025-03-25T10:14:57Z","snapshot_observed_at":"2026-08-10T04:02:41.432736Z","submitted_at":"2024-03-09T11:16:09Z","title":"Computational Analysis of Stress, Depression and Engagement in Mental Health: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08824","snapshot_observed_at":"2026-08-06T13:07:27.626612Z","title":"Measuring non-typical emotions for mental health: A survey of computational approaches","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.626612Z"},"links":{"cited_paper":"/paper/2403.08824","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:facfe62cc4736561520f977f6ea140027126ae3bde93338f7f9f145f79ed25ff","observation_id":"100cbd00-c651-4570-a040-0f104650aa21","resolution":{"observed_at":"2026-08-06T13:07:27.626612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.443019Z","title":"Recognizing emotion from facial expressions: psychological and neurological mechanisms","venue":null,"work_id":"53e2b549-ac6a-4839-ad5c-b9e3c24c8610","year":2002},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.632099Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:a64e5dfcba09d03b1804efea705d8efde18bd53f4f464425d2feabba8a6dde56","observation_id":"b5516e37-14cc-4f24-95d8-7281e94524eb","resolution":{"observed_at":"2026-08-06T13:07:28.447457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-06T13:07:27.636802Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.636802Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:7476548e90d0a11f12396943091b6a30f764c73f21495c78d167568e3fddaaad","observation_id":"c4942f20-be8d-43ad-9381-17fd57fdd3f2","resolution":{"observed_at":"2026-08-06T13:07:27.636802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:27.641491Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.641491Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:4d9bc64ae0379481fc7d46bb94c83dd716e9fa56ad5f69f62b97cf1bb476d468","observation_id":"49e8119b-58e9-439e-9a72-0efcf4c17de3","resolution":{"observed_at":"2026-08-06T13:07:27.641491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.415813Z","title":"Facial action coding system","venue":null,"work_id":"0f4f2493-e5b4-4f79-96b6-5dcf9c3324d5","year":1978},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.646353Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:99c9148f779b6ef795502810e97270b9676fab8cee4bcadb23257241fc3071e4","observation_id":"316b960e-eb7b-47f3-b8bd-6f6fc7c5b4d3","resolution":{"observed_at":"2026-08-06T13:07:28.420990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.399917Z","title":"Softclip: Softer cross-modal alignment makes clip stronger","venue":null,"work_id":"3e5fd32b-3256-4e59-a4dc-cc1121b0d4a7","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.651724Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:14fe3a95ced003256452e6f3074b6765d45dd21a0ed66598a83585d131805f92","observation_id":"00c31e87-2361-43ec-9cb8-440841c6c4aa","resolution":{"observed_at":"2026-08-06T13:07:28.404940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.382871Z","title":"Cwcl: Cross-modal transfer with continuously weighted contrastive loss","venue":null,"work_id":"c4e294e2-01af-470c-9c9a-da81d393ef65","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.658642Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:eb1cb3d9ac7d74cb3a441616bbcca063f6dcda8a78fed2b430fd0669f76ab668","observation_id":"c92ff191-88ce-47ea-a195-157a86b31f93","resolution":{"observed_at":"2026-08-06T13:07:28.388027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.365515Z","title":"Generalizable facial expression recognition","venue":null,"work_id":"ac5e2668-2326-4c20-bbcf-32ab7cb95bc3","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.664046Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:617f45867160a1f58e01560de95b712c4275b8b62bc13ae0e89c8462225e3e89","observation_id":"c7333be1-82c7-4c11-ab1e-5c7e57932a2d","resolution":{"observed_at":"2026-08-06T13:07:28.371188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.346402Z","title":"Flava: A foundational language and vision alignment model","venue":null,"work_id":"3841f7cf-b2c1-4d6d-8f42-31e66e9b1dd0","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.669757Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:80912541f791a0430e47c8775643792f2704df261e60b4d662a46fca3c3607bc","observation_id":"c6b01c4a-bc33-4570-bd7d-25213998ac48","resolution":{"observed_at":"2026-08-06T13:07:28.351256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20717","last_updated":"2024-10-28T04:19:32Z","snapshot_observed_at":"2026-08-09T23:23:01.453631Z","submitted_at":"2024-10-28T04:19:32Z","title":"Face-MLLM: A Large Face Perception Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.20717","snapshot_observed_at":"2026-08-06T13:07:27.675297Z","title":"Face-mllm: A large face perception model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.675297Z"},"links":{"cited_paper":"/paper/2410.20717","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:9f1a0885e1e95444ef7107717be3f0b07f3cf33eeb2303b779ae5ba5efb1e6e8","observation_id":"37e65246-652e-4bcb-92b5-e08e324a1669","resolution":{"observed_at":"2026-08-06T13:07:27.675297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14786","last_updated":"2025-02-20T18:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-20T18:08:29Z","title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14786","snapshot_observed_at":"2026-08-06T13:07:27.680305Z","title":"Siglip 2: Multilingual vision-language encoders with improved semantic understanding, localization, and dense features","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.680305Z"},"links":{"cited_paper":"/paper/2502.14786","citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:8e16da081ae2f8bf9469cb37126c1a48adcb759fd647a61498b292aca00c67a9","observation_id":"38ee7e1c-45b8-44fe-a6f2-2dbb1596612e","resolution":{"observed_at":"2026-08-06T13:07:27.680305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.327547Z","title":"Learn from all: Erasing attention consistency for noisy label facial expression recognition","venue":null,"work_id":"6a282ab9-b61c-4938-89f9-fcb1e0b8d3b7","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.686607Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:dad46190b8891a6f8fba910d9ec32b7ebc80b360e9b0694ef8f3a45da5f448fc","observation_id":"4d20077b-717a-4e68-84af-bfacd77fd4f5","resolution":{"observed_at":"2026-08-06T13:07:28.333222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.308993Z","title":"Latent-ofer: Detect, mask, and reconstruct with latent vectors for occluded facial expression recognition","venue":null,"work_id":"63e735e2-78fa-40f6-9e92-7a89261abce4","year":2023},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.692832Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:5eaae2b1b925ea001fae1b1f10ddd68c3c16a8965766dc48b2c5fd78d95bfc31","observation_id":"0c7ce87b-05cb-44d0-a49c-09dfc5bd7168","resolution":{"observed_at":"2026-08-06T13:07:28.314024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.288809Z","title":"From static to dynamic: Adapting landmark-aware image models for facial expression recognition in videos","venue":null,"work_id":"9164280e-c3d2-46b4-8984-b6412ac7e02e","year":2024},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.698844Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:34f50c020c7b98fadf2c744efb1fbb5850575e0bab8dfd9dcfbb69b2118b1585","observation_id":"0ddd7f09-5bdb-4044-92eb-3384ca30156a","resolution":{"observed_at":"2026-08-06T13:07:28.295881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.272405Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"9ba6373e-18f8-493b-88ba-bb60c51ca05d","year":2021},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.705145Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:ee59be5646694988b5d76cc2e424be5d6199905f2b317a34b131a7911d3dcb0a","observation_id":"84411ab7-998a-4dea-b93e-156964029dea","resolution":{"observed_at":"2026-08-06T13:07:28.277285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.256283Z","title":"Expanding language-image pretrained models for general video recognition","venue":null,"work_id":"ca51a836-7950-4017-a47d-313018b9cd45","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.709988Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:454e25f9abb74eda4cf2a9dbffb9e8c04ca8b0c4710ec8df70b47121f9d732e1","observation_id":"bb9fa76a-8359-49e1-a6b9-59488c37731a","resolution":{"observed_at":"2026-08-06T13:07:28.261598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:07:28.236841Z","title":"Learning to prompt for vision-language models","venue":null,"work_id":"1c4e9b3c-37d7-4a3e-a75a-77c9a4b60be4","year":2022},"citing_paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-06T13:07:27.714660Z"},"links":{"citing_paper":"/paper/2507.21015"},"observation_digest":"sha256:be856249397092ced70af57dd48ae378c229266520c255f625a94ad81b088bf5","observation_id":"30c11dac-bc73-4cf0-8cae-94b246a1bbc4","resolution":{"observed_at":"2026-08-06T13:07:28.243764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21015","last_updated":"2025-07-28T17:28:08Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T20:11:38.562500Z","submitted_at":"2025-07-28T17:28:08Z","title":"Learning Transferable Facial Emotion Representations from Large-Scale Semantically Rich Captions"},"reference_resolution":{"displayed":94,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":43,"verified_exact":0,"verified_fuzzy":51},"total_outbound_references":94},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 94 of 94 outbound references and 0 inbound Pith citation observations for arXiv:2507.21015."}