{"as_of":"2026-08-23T07:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:43cf8f23b0f2b42d8ee5061bcc7343c4840abe251609c7d6fba1fdfd00e1dc5d","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:45:35.630510Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:45:35.485977Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-15T17:45:35.776294Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"cited_work":{"arxiv_id":"2507.20568","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.20568","snapshot_observed_at":"2026-08-15T17:45:35.776294Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","venue":"cs.CV","work_id":"9a09f67e-a63e-4a58-817e-f85162c811f8","year":2025},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.485977Z"},"links":{"cited_paper":"/paper/2507.20568","citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:84d17c52f75a946b2b03cdf37d89def8925d68b30ee1b0306f845d3d45b70a1b","observation_id":"d7300e3d-3c19-4e37-be91-70c578710184","resolution":{"observed_at":"2026-08-15T17:45:35.783472Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.20568/citation-record","integrity":"/paper/2507.20568/integrity","json":"/paper/2507.20568/citation-record.json","paper":"/paper/2507.20568"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"cited_work":{"arxiv_id":"2507.20568","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.20568","snapshot_observed_at":"2026-08-15T17:45:35.776294Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","venue":"cs.CV","work_id":"9a09f67e-a63e-4a58-817e-f85162c811f8","year":2025},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.485977Z"},"links":{"cited_paper":"/paper/2507.20568","citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:84d17c52f75a946b2b03cdf37d89def8925d68b30ee1b0306f845d3d45b70a1b","observation_id":"d7300e3d-3c19-4e37-be91-70c578710184","resolution":{"observed_at":"2026-08-15T17:45:35.783472Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.221053Z","title":null,"venue":null,"work_id":"aa6bf836-a411-4762-bbba-4b2ae6697673","year":null},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.491335Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:18f8557f5c25aff3c32edc1df00823ce2d8684a1bb30a7939a5cd7e0d05c5a50","observation_id":"db2fb6d1-8225-4060-bed0-ef4f8476012b","resolution":{"observed_at":"2026-08-15T17:45:36.225921Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.206744Z","title":"#$%ℒ!\"#$%→𝓛𝒑𝒄GTCodeTalkerℒ!","venue":null,"work_id":"e79b002a-fc0e-405a-8c57-9a413b8f4ea6","year":1950},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.496691Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:06398bb758c946ec8a615aa86085c6d964a3558130b05240a0e5545aee0f61a6","observation_id":"047c2a33-7b04-4828-84de-1b5fc666d10f","resolution":{"observed_at":"2026-08-15T17:45:36.210958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.177191Z","title":null,"venue":null,"work_id":"ef4b62d1-d9a2-4121-8910-e2f67ac5bf33","year":null},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.506448Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:a884eada76dac121978e319b63be38c6fc0a8ac1dd057f83d88527d0d672d9ce","observation_id":"956cd412-625a-4618-b331-a7ab6d662c36","resolution":{"observed_at":"2026-08-15T17:45:36.181886Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.162602Z","title":null,"venue":null,"work_id":"20fa7192-a984-4339-89af-220b9b0f6c55","year":2023},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.511020Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:bbe0d697ab313fba2d9f5791be886f81202b9ea2bf461657e8c0df93566efbcd","observation_id":"8aae4533-9797-4bc7-8470-d309a64cff60","resolution":{"observed_at":"2026-08-15T17:45:36.167531Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.079033Z","title":"Exploring phonetic context-aware lip-sync for talking face generation,","venue":null,"work_id":"8a541fca-078c-4cfc-b79b-574d44d433af","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.539133Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:071aeb3b14b9397d297f0b77e45b3dbc60b90294af62bc102c2013dc8e25818e","observation_id":"c4a617cb-e7ca-41d0-a549-7fc0b4409cc1","resolution":{"observed_at":"2026-08-15T17:45:36.083697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.148123Z","title":"Synthesizing obama: learning lip sync from audio,","venue":null,"work_id":"bdca508a-76ab-4f9d-b9b9-1ac7237d627e","year":2017},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.515693Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:015fa8c7389fe14560920f36cdd9735e0344f02a87082acfdbff4305e14fe892","observation_id":"abcdf14f-23d5-4c66-8e62-afbe2713349c","resolution":{"observed_at":"2026-08-15T17:45:36.153001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.133980Z","title":"Deep video portraits,","venue":null,"work_id":"8b931aa7-4d3e-4804-b1d1-9b95e211d211","year":2018},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.520317Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:e94a4266fbb6da37ad75fa6848db132881800ce25a654e22dab1ba42bb1241d9","observation_id":"281d1bb8-dc16-4d0c-988a-0ec96439fb6c","resolution":{"observed_at":"2026-08-15T17:45:36.138564Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.118914Z","title":"Analyzing visible articulatory movements in speech production for speech-driven 3d facial an- imation,","venue":null,"work_id":"322f3162-fde3-4d17-abda-3d4397aa2133","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.524948Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:063dcbf967c321b3d40ca26040d771e60d330065e3ac75412037f78d70eddfcb","observation_id":"3d569118-45f7-47f4-8e0e-9c4fded09d18","resolution":{"observed_at":"2026-08-15T17:45:36.124024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.105807Z","title":"Enhancing speech-driven 3d facial animation with audio-visual guidance from lip reading expert,","venue":null,"work_id":"c2d19180-fec7-4955-bb6c-f0f1e21b19be","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.530146Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:409352270245385f68f377c1b6a2847de7f889c3bdb1d8f3904d1e2bef894722","observation_id":"38f8da23-4427-4472-b12c-df362411b339","resolution":{"observed_at":"2026-08-15T17:45:36.109965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.092813Z","title":"Facial anima- tion based on context-dependent visemes,","venue":null,"work_id":"86994777-b063-4fbe-87dd-a40cab2b9b44","year":2006},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.534585Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:f0d53327f378862adf24f6135a858d53bc0c23f89e96acc6c0232889bf9819c4","observation_id":"647d4152-b926-4f56-ae51-284aae3cbd8a","resolution":{"observed_at":"2026-08-15T17:45:36.096812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.003530Z","title":"Unitalker: Scaling up audio-driven 3d facial animation through a unified model,","venue":null,"work_id":"83f38073-57f9-42f2-ae90-deb061c74e3a","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.566477Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:f521b8ffe437bf133e78b647198ae620678e59fef9c9abd88e6ca2ce88de9e6e","observation_id":"6b77d196-78f9-45de-88fc-2b1a280cf131","resolution":{"observed_at":"2026-08-15T17:45:36.008142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.064506Z","title":"Lip read- ing sentences in the wild,","venue":null,"work_id":"5496a684-b42c-49f6-b9db-6c1b8a165273","year":2017},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.544036Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:a79c3c78edec8d2a0e4cd679f4cd01a2fdfad68cf87817b6b02f296ca371323b","observation_id":"97cedecc-4e53-468d-923f-c89c565b91fa","resolution":{"observed_at":"2026-08-15T17:45:36.069176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.048694Z","title":"Deep audio-visual speech recognition,","venue":null,"work_id":"1d91acf3-1bc4-4a3e-87c5-7895db45ac7f","year":2022},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.548644Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:cda0efe8a4b078171ade44b4d110f8cd3bb530211b847ec00b8870ffb6244d1d","observation_id":"51ae8322-fe94-462d-9e41-9390c7981a42","resolution":{"observed_at":"2026-08-15T17:45:36.054399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.191787Z","title":null,"venue":null,"work_id":"b5d53ea7-669a-41f1-9fe1-1ad054d681ac","year":null},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.501654Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:f0dcc7cc60622ed79ffbc59351be53f6de479e298d74a1ce8d0628817ef9178a","observation_id":"f361292e-702a-46cd-9196-c630f5a5199a","resolution":{"observed_at":"2026-08-15T17:45:36.196634Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.033912Z","title":"Probabilistic speech-driven 3d facial motion synthe- sis: New benchmarks methods and applications,","venue":null,"work_id":"6344f13e-2147-4b8a-9e2a-3e9241c94897","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.553252Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:c8688236117b8352ef9b43a07f7a068cab34957a7cbec4c1791b82673121116c","observation_id":"f29a4a06-2f51-48da-b39f-00694507d7d1","resolution":{"observed_at":"2026-08-15T17:45:36.038828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.557461Z","title":"Speech-driven 3D face animation with composite and regional facial movements,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.557461Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:dc55a921377a02f5f2a9ae52337b90329a162149bf1513bfd4b1a7c38aad477e","observation_id":"570587a7-fb89-49c8-92ea-4cb9a0163987","resolution":{"observed_at":"2026-08-15T17:45:35.557461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:36.018698Z","title":"Facetalk: Audio- driven motion diffusion for neural parametric head models,","venue":null,"work_id":"a1766ad9-cef3-4b27-8ac7-a3a54f86137c","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.561971Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:364227e3a2c7a3d5f930d290f4a058e1f1f5ad355b680c14ed40a8c8c08d8d00","observation_id":"a8239471-f431-49e6-b46e-6fb96b681880","resolution":{"observed_at":"2026-08-15T17:45:36.023731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.988708Z","title":"Imitator: Personalized speech-driven 3d facial an- imation,","venue":null,"work_id":"b8b9769c-763b-477b-b5fb-d7fa6644998f","year":2023},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.570812Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:b498ad7e2efb8b7cff74ad54faacbc92ea9f933841b8d95d8c92a2220b2f319f","observation_id":"008b8ba6-1138-4f11-a6b8-456525349d6f","resolution":{"observed_at":"2026-08-15T17:45:35.993664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.973183Z","title":"Diffposetalk: Speech-driven stylistic 3d facial animation and head pose generation via diffusion models,","venue":null,"work_id":"b8096cbc-44ff-4ebc-85b0-a3f541e98518","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.575188Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:7ac3df3fbccfbba179ff36c0b03cfd319f7fce2bc64a26710afcb4589dd52596","observation_id":"11a8720e-2c32-4f60-ae48-7f824516bb18","resolution":{"observed_at":"2026-08-15T17:45:35.978204Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.957973Z","title":"Capture, learning, and synthesis of 3D speaking styles,","venue":null,"work_id":"cdf337b2-7b24-42df-b540-9b8d40e94047","year":2019},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.580064Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:ae8a1e85fa735fd5624e7385eafaeee835f569a281e85cc1a1e234e22af01d14","observation_id":"f8ffe7a1-0ade-434b-8332-8fa83d621e30","resolution":{"observed_at":"2026-08-15T17:45:35.962723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.942730Z","title":"Faceformer: Speech-driven 3d facial animation with transformers,","venue":null,"work_id":"cc168e7a-526c-421c-960d-8ed5aa59d691","year":2022},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.584596Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:e09c4634c3b9f12895bf80958bde407ff231903ca2a2f2f787cdd669d0abe42b","observation_id":"9837ac46-e866-4b39-85ff-4b4e2d54f90e","resolution":{"observed_at":"2026-08-15T17:45:35.947700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.928039Z","title":"Codetalker: Speech-driven 3d facial animation with discrete mo- tion prior,","venue":null,"work_id":"4e1e0221-2c5d-4ec1-9e54-fe210e822fac","year":2023},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.588904Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:299acddf9bf0dc0f8c12a170548a80f2eeea6c38150a896a58818ff7564941c4","observation_id":"d749cc8e-29d3-4700-9848-e6198ace2169","resolution":{"observed_at":"2026-08-15T17:45:35.933005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.913411Z","title":"Selftalk: A self-supervised commutative training dia- gram to comprehend 3d talking faces,","venue":null,"work_id":"dd0221fa-f9c1-47fb-b6aa-cf3fe364742d","year":2023},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.593483Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:4c79b1a1b0db7e72994ee6eb611c47aa43333cac0e79fc1bfaed1ad04efd8224","observation_id":"ce76801b-a5ab-4bac-9abd-b4ce16aae40e","resolution":{"observed_at":"2026-08-15T17:45:35.918032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.897608Z","title":"Scantalk: 3d talking heads from unregistered scans,","venue":null,"work_id":"c129d049-faa4-414b-9523-8171999e8ea5","year":2024},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.597893Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:c28f4172541d427ea2212ba0c741f8abc44bb6aaa42e5340889f2ddb5045bcd6","observation_id":"7be29f90-dbe6-4bf6-ab14-5e51943c703a","resolution":{"observed_at":"2026-08-15T17:45:35.902652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.881448Z","title":"A 3-d audio-visual corpus of affective communication,","venue":null,"work_id":"f195e95b-ad56-490b-9f4d-23ed842b0aea","year":2010},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.602287Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:3e34809e7ded7fa54388172ee430067e0d59415afd34cd27fbbcfe014c7d4559","observation_id":"981113a7-8324-417f-83c8-7adc892fed3f","resolution":{"observed_at":"2026-08-15T17:45:35.886180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.11243","last_updated":"2023-06-26T17:43:18Z","snapshot_observed_at":"2026-08-16T16:43:59.403256Z","submitted_at":"2022-07-22T17:55:39Z","title":"Multiface: A Dataset for Neural Face Rendering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.11243","snapshot_observed_at":"2026-08-15T17:45:35.606701Z","title":"Multiface: A dataset for neural face rendering,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.606701Z"},"links":{"cited_paper":"/paper/2207.11243","citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:236b0a92aa0f37bdd63414249d9845bcd3f9cc14a4e1f751a4751cb6ebf22c13","observation_id":"2de44def-1030-4529-829a-76b9f29a463b","resolution":{"observed_at":"2026-08-15T17:45:35.606701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.866309Z","title":"Lipreading using temporal convolutional networks,","venue":null,"work_id":"f208857a-c8b8-4dfe-aca5-05d5641e0248","year":2020},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.610868Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:6ab60e04042cba2bcf10bbcd2f12449fed2c2f106ce15eb4b36c104b35067286","observation_id":"e815e14c-2df2-429b-b0a4-7afae147e3ae","resolution":{"observed_at":"2026-08-15T17:45:35.870903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.850502Z","title":"Masked lip-sync prediction by audio-visual contextual exploitation in transformers,","venue":null,"work_id":"6f420161-1c3b-4a4b-b886-2152be51628a","year":2022},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.614971Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:f31465468ff4f769044991a9adea428a190a587ee6fa0e90b2f978e54c27c7c6","observation_id":"6b751b62-ce00-4e3f-acde-d95e31bd34c1","resolution":{"observed_at":"2026-08-15T17:45:35.855570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.835110Z","title":"Modeformer: Modality- preserving embedding for audio-video synchronization using transformers,","venue":null,"work_id":"89576d69-a96e-4c71-a6f3-d05bd7adeab3","year":2023},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.618807Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:8273cfefc7b0e46ab5323a8f87b25cc71fcbb29e21bf7d1705a2defe67ff9dcc","observation_id":"1d6bdaf6-a479-48d3-b251-a2e305e04489","resolution":{"observed_at":"2026-08-15T17:45:35.840070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.820141Z","title":"Learning a model of facial shape and expression from 4d scans,","venue":null,"work_id":"f1137c64-7ade-4275-b5fb-7a739d74db07","year":2017},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.622850Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:1f23ebb1eed76da204a9a38c9de1555d676ad5714100be38fab124f39e897c4c","observation_id":"61cc1179-f847-495c-b8c2-d11433c891a7","resolution":{"observed_at":"2026-08-15T17:45:35.824951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.626840Z","title":"Toward accurate dynamic time warping in linear time and space,","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.626840Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:26da81bc6c5c243ab4833dfb07eee06ddbff1eb084612f66049c9f7ab8b34023","observation_id":"b39e061b-615b-45f9-8098-3dfd57a84051","resolution":{"observed_at":"2026-08-15T17:45:35.626840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:45:35.794692Z","title":"Meshtalk: 3D face animation from speech using cross-modality disentanglement,","venue":null,"work_id":"778d13be-5ea0-49d3-9934-ec82b060c771","year":2021},"citing_paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T17:45:35.630510Z"},"links":{"citing_paper":"/paper/2507.20568"},"observation_digest":"sha256:604680269f6dc55d7ae2785e9bad7492c1de529773be65624ef14d2e68182aed","observation_id":"55323e5f-53f6-4a8d-955a-c9d874e33566","resolution":{"observed_at":"2026-08-15T17:45:35.799699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.20568","last_updated":"2025-08-11T10:06:46Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T17:38:58.612364Z","submitted_at":"2025-07-28T07:04:50Z","title":"Learning Phonetic Context-Dependent Viseme for Enhancing Speech-Driven 3D Facial Animation"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":25},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 1 inbound Pith citation observation for arXiv:2507.20568."}