{"as_of":"2026-08-15T04:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:df298cdd18a7036d3233a8d2938012dab179fba87d9bef97a1a2ffc661e3b29d","coverage":[{"denominator":63,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":63,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:18:44.636368Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-14T09:16:37.881212Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-10T07:52:13.672691Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"cited_work":{"arxiv_id":"2506.05890","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05890","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"611ecb18-4b9c-40f0-b5c1-7fbd35dfb81f","year":2025},"citing_paper":{"arxiv_id":"2604.16987","last_updated":"2026-08-06T06:53:47Z","snapshot_observed_at":"2026-08-11T03:44:47.099788Z","submitted_at":"2026-04-18T13:22:42Z","title":"DVAR: Adversarial Multi-Agent Debate for Video Authenticity Detection","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T07:51:04.580617Z"},"links":{"cited_paper":"/paper/2506.05890","citing_paper":"/paper/2604.16987"},"observation_digest":"sha256:5bfea069077a913cb5656f1ea866728d80ef1a9308dce356e7dbb8e4a9792c32","observation_id":"05daabf8-1f49-4c32-ab61-1d6c0584ba93","resolution":{"observed_at":"2026-05-10T07:52:13.673965Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05890","snapshot_observed_at":"2026-07-14T09:16:37.881212Z","title":"arXiv preprint arXiv:2506.05890 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10787","last_updated":"2026-07-12T14:25:04Z","snapshot_observed_at":"2026-08-14T13:47:53.862539Z","submitted_at":"2026-07-12T14:25:04Z","title":"Detecting AI-Generated Video: A Vision-Language Dual-View Survey","version":1},"reference_index":147,"source":"arxiv_source","source_observed_at":"2026-07-14T09:16:37.881212Z"},"links":{"cited_paper":"/paper/2506.05890","citing_paper":"/paper/2607.10787"},"observation_digest":"sha256:34b298ba24f17b6b51f22772bde0c95b4d6ccb390b6bcaf4d69921271e86cbc4","observation_id":"54b0d352-79e3-4e54-8b40-06164f84c7f7","resolution":{"observed_at":"2026-07-14T09:16:37.881212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.05890/citation-record","integrity":"/paper/2506.05890/integrity","json":"/paper/2506.05890/citation-record.json","paper":"/paper/2506.05890"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:58.871737Z","title":"Open- domain, content-based, multi-modal fact-checking of out- of-context images via online resources","venue":null,"work_id":"01a4253d-d8cc-4ca6-861a-aab03f61b1b1","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:37.239990Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:264366856c26f8d0f262d752b928141ca5b64a27935c42ddf2d71233cfc706a8","observation_id":"f2f83a14-1a38-4827-a901-d0dab17856a9","resolution":{"observed_at":"2026-08-07T10:18:59.062729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:58.576544Z","title":"Exposing the deception: Uncover- ing more forgery clues for deepfake detection","venue":null,"work_id":"949deac3-0e9b-438e-9650-7053171e3333","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:37.340649Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:7503cc1af39c49a3e103471493c4ca6de006075e40e386a4e6491cac261b02c3","observation_id":"60c62935-ceee-4f73-b3e8-72de03e1a8a3","resolution":{"observed_at":"2026-08-07T10:18:58.706920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:58.304857Z","title":"Aligned and non-aligned double jpeg detection using convolutional neural networks.Jour- nal of Visual Communication and Image Representation, 49: 153–163, 2017","venue":null,"work_id":"6a7aef01-225e-4ac2-9dee-fcc1cf40f940","year":2017},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:37.464100Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:33b637870ccb5b94719e0ba1c44613e53cd999418f577293056a329cab46057b","observation_id":"42e2ac0f-08f3-4cdc-829e-6d357210cbd7","resolution":{"observed_at":"2026-08-07T10:18:58.442126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:58.000517Z","title":"Audio-visual person-of-interest deep- fake detection","venue":null,"work_id":"b60a7d5f-0134-48f5-ab0b-ade233702c46","year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:37.662591Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:43c38bf58e853f5a2464e74e649fded4fde3995a4af17cb31aaff882209e5f7d","observation_id":"d0a3c8e5-0c5a-4ba2-8743-a643fbf60faf","resolution":{"observed_at":"2026-08-07T10:18:58.120854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T10:18:37.831155Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale.arXiv preprint arXiv:2010.11929, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:37.831155Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:9041cbedb4be42a705582d8cf7044183997306af1361cd692b46382ae178374a","observation_id":"ad2d48c9-c86e-4d21-b8db-fa344cb04b23","resolution":{"observed_at":"2026-08-07T10:18:37.831155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:57.652775Z","title":"An empirical study of training end-to-end vision-and-language transformers","venue":null,"work_id":"6945995a-c4f9-4f99-b25e-43a01e34233c","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:37.995634Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:ba29e7a09647d6fc07d5861c88a716cc92f798e88a37106b18c18d73aaa06656","observation_id":"0d95b32e-1f85-4866-b4f6-b7c4b9f5b0cc","resolution":{"observed_at":"2026-08-07T10:18:57.823323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:38.187795Z","title":"Generative adversarial networks.Commu- nications of the ACM, 63(11):139–144, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:38.187795Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:720c8f0224b9e9adcd53ff69096fe07033ca8d909c5ab5858f2708886eb7b370","observation_id":"58b722c5-067f-46dc-8c29-22bbdbc0562d","resolution":{"observed_at":"2026-08-07T10:18:38.187795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:57.311774Z","title":"Delving into the local: Dynamic in- consistency learning for deepfake video detection","venue":null,"work_id":"17bbb000-9399-42a1-a83b-3882ec939833","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:38.364308Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:eb178c933ff462004b75ff3589deb14a14595131f37e4b66d637b4ef4c2ad463","observation_id":"1185772c-8877-4b83-a132-30e03275b865","resolution":{"observed_at":"2026-08-07T10:18:57.485540Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:38.475145Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:38.475145Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:9af8a1697871f2f4dc82daedf92371a33f41df9d0d69463302bce978820c36fd","observation_id":"c696d886-90a5-4f49-965f-b9d4a04a3e7c","resolution":{"observed_at":"2026-08-07T10:18:38.475145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:38.579509Z","title":"Momentum contrast for unsupervised visual rep- resentation learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:38.579509Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:beceb0dbf84ff217e88624e18bfbe210185cbb2ad6974c6c27a6285ed8ab10d1","observation_id":"6000fa62-6f22-4acf-932e-9351cf837ce9","resolution":{"observed_at":"2026-08-07T10:18:38.579509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:57.019111Z","title":"Detection of fake images via the ensemble of deep representations from multi color spaces","venue":null,"work_id":"c5227cbd-9ffa-4965-96f0-1bdc908fbff6","year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:38.755404Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:67054d00eaebcfb235f01581ce5868e91fcbd53ed5b30b8b7726a5602ef4fc10","observation_id":"4c4a0c62-ec3a-4e68-86d8-103aebb361e7","resolution":{"observed_at":"2026-08-07T10:18:57.148313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:38.952761Z","title":"Denoising dif- fusion probabilistic models.Advances in neural information processing systems, 33:6840–6851, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:38.952761Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:3eda8abd386f74e5305ef08100ae522ec98012861ee3117199128ce1c525c3d2","observation_id":"d23ec748-075b-42b7-9909-793e446e71ab","resolution":{"observed_at":"2026-08-07T10:18:38.952761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:39.097255Z","title":"Fighting fake news: Image splice detection via learned self-consistency","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.097255Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:85d82368df161c13fd455c0fb54de767c26e920714a6c9ad4781b35164343ccc","observation_id":"a5a54910-9491-44c9-a7a3-9e26b20fe89b","resolution":{"observed_at":"2026-08-07T10:18:39.097255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:56.520070Z","title":"Bihpf: Bilateral high- pass filters for robust deepfake detection","venue":null,"work_id":"93a2c45f-eeb8-484a-b7fb-f9b98f9573d7","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.234620Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:727e494b3ae9fef0205ecfa499914cd6ab59ec424d3c7ae57a3ab03935eeb966","observation_id":"27364305-05c9-4d16-94cf-dd41334d18ea","resolution":{"observed_at":"2026-08-07T10:18:56.793365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:56.216091Z","title":"Multimodal fusion with recurrent neural networks for rumor detection on microblogs","venue":null,"work_id":"eee3e4c8-0dce-4ab1-b577-52dea55dc060","year":2017},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.344331Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:a0ae01533c650604df1577f519bbe51e974d42cd35cc4124058866c7d9597002","observation_id":"afdc16d7-8c86-4166-ae62-cccfc2119eaf","resolution":{"observed_at":"2026-08-07T10:18:56.320475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:55.803101Z","title":"Countering malicious deepfakes: Survey, battleground, and horizon.International journal of computer vision, 130(7):1678–1734, 2022","venue":null,"work_id":"446f1021-5144-48be-ab26-b092a1f16197","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.434488Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:50af45b6e9d4ae49d492d27f8a7d3650f2a968b4a385186804b70213088ea8ad","observation_id":"4ee5331d-2816-4c4e-9b47-71a1ad43db46","resolution":{"observed_at":"2026-08-07T10:18:55.992383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:55.430837Z","title":"Bert: Pre-training of deep bidirectional trans- formers for language understanding","venue":null,"work_id":"efbce1d2-88c1-4f76-863d-4f4cacbdfc9a","year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.576824Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:9a045db64d8465b906a89a4a0f61ac2c304efb945a9ba63aac679291b88cc94e","observation_id":"0422e08a-b539-4045-8ba7-b218564626c2","resolution":{"observed_at":"2026-08-07T10:18:55.636659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:55.104032Z","title":"Mvae: Multimodal variational autoencoder for fake news detection","venue":null,"work_id":"aabeba6c-1560-4e97-8c6b-907fd0b262d0","year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.675659Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:00059565f626f86f87bccdbc7f21efc27d4a9ba80ea13236d27eaeaa56bd818f","observation_id":"d4f3a157-cbcf-4bc3-8c6d-b537b34cd1ed","resolution":{"observed_at":"2026-08-07T10:18:55.237008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:39.799083Z","title":"Vilt: Vision- and-language transformer without convolution or region su- pervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.799083Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:a37898d33cbc8c876354cb58c1893e2ccc63e7812e545c9209d55a35c6762cae","observation_id":"9f0607bc-c543-4ac5-82f4-61adc4829e22","resolution":{"observed_at":"2026-08-07T10:18:39.799083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:54.638321Z","title":"Align before fuse: Vision and language representation learn- ing with momentum distillation.Advances in neural infor- mation processing systems, 34:9694–9705, 2021","venue":null,"work_id":"d46304b3-be32-4a4b-bf23-bfec5f65c5ed","year":2021},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.880919Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:6ecf5f54cc0033c0c67f5d4d90dcefc194d8e02ede0f9b6cdd3ccb0725075e83","observation_id":"239427a4-4f94-4b89-9bc5-40179039fc5b","resolution":{"observed_at":"2026-08-07T10:18:54.808047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:39.979138Z","title":"Frequency-aware discriminative feature learning supervised by single-center loss for face forgery detection","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:39.979138Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:73f28989fe1d651cde521d58bd1d0bfe53b0ab1f81cac278d4e5fe0857982839","observation_id":"4cd99fa1-cb0d-493e-b010-06fec3a461ba","resolution":{"observed_at":"2026-08-07T10:18:39.979138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:54.270010Z","title":"Towards multimodal dis- information detection by vision-language knowledge inter- action.Information Fusion, 102:102037, 2024","venue":null,"work_id":"847d6a20-afa3-4858-80bb-bd1d2bb40873","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.106193Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:91736c1942535ed4ce9eba3ed7fa6d4257606902e03fe4b3a0289714610329d1","observation_id":"ff95316e-3534-4086-bbc8-3a722dce76c8","resolution":{"observed_at":"2026-08-07T10:18:54.445033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:54.030139Z","title":"Unified frequency-assisted trans- former framework for detecting and grounding multi-modal manipulation.International Journal of Computer Vision, pages 1–18, 2024","venue":null,"work_id":"36e413c6-8b4b-41c8-81a8-439142b29eee","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.238908Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:2603b480d1bb5941d5e1d49bb82f1975048b7e3c0126f2ee9ffbbb6a8bc567ae","observation_id":"9355d2e6-4487-4009-a35f-15e3bb31cd4e","resolution":{"observed_at":"2026-08-07T10:18:54.105717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:53.760953Z","title":"Fka-owl: Ad- vancing multimodal fake news detection through knowledge- augmented lvlms","venue":null,"work_id":"4e3c3979-e4f2-4b1e-9ea5-74436da874b3","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.295054Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:7aaee3f1ed8fbca5e1c9155300300e217b55fd84a91801aa10a94d477c3425d6","observation_id":"c920d64e-cdda-4e64-aed0-08eb777b9c96","resolution":{"observed_at":"2026-08-07T10:18:53.880072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.11692","last_updated":"2019-07-26T17:48:29Z","snapshot_observed_at":"2026-07-31T22:31:37.910868Z","submitted_at":"2019-07-26T17:48:29Z","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.11692","snapshot_observed_at":"2026-08-07T10:18:40.387139Z","title":"Roberta: A robustly optimized bert pretraining approach.arXiv preprint arXiv:1907.11692, 2019","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.387139Z"},"links":{"cited_paper":"/paper/1907.11692","citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:6ad3e6f70c1fec5c1f8af6a08ab842dcc980a5c15e14d15c4a84e63bf4affd84","observation_id":"4c5d42fa-e832-4903-91ca-18e4a75ff05b","resolution":{"observed_at":"2026-08-07T10:18:40.387139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-14T20:13:52.872565Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-07T10:18:40.495276Z","title":"Fixing weight decay regularization in adam.arXiv preprint arXiv:1711.05101, 5,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.495276Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:908e6503716ff5fcaa3ebc73af699467521b9ff4d3e2ba4ac2f67482cb46b6a1","observation_id":"0b29f08c-bcfd-4cbd-aae0-749cdad8f362","resolution":{"observed_at":"2026-08-07T10:18:40.495276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.05893","last_updated":"2021-09-21T20:38:45Z","snapshot_observed_at":"2026-08-13T19:40:53.985009Z","submitted_at":"2021-04-13T01:53:26Z","title":"NewsCLIPpings: Automatic Generation of Out-of-Context Multimodal Media","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.05893","snapshot_observed_at":"2026-08-07T10:18:40.532883Z","title":"Newsclip- pings: Automatic generation of out-of-context multimodal media.arXiv preprint arXiv:2104.05893, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.532883Z"},"links":{"cited_paper":"/paper/2104.05893","citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:d880c1557fe0a9e3c19ad8992c1262bd0e1392f9bde6a39d1ef46cb624a83f30","observation_id":"eac1e57d-cd67-43e9-8f44-19b19e01545e","resolution":{"observed_at":"2026-08-07T10:18:40.532883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:40.621841Z","title":"Gener- alizing face forgery detection with high-frequency features","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.621841Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:db040ccfd68c8506898be81b3ad6da1d582df74a9a4b1d427e4c7c0e3e39f279","observation_id":"ae2dbec0-0502-470d-9265-2537b94e76b2","resolution":{"observed_at":"2026-08-07T10:18:40.621841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:53.366050Z","title":"Forensic similarity for digital images.IEEE Transactions on Information Forensics and Security, 15:1331–1346, 2019","venue":null,"work_id":"769f73ad-2860-4a30-b666-18feb683169c","year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.697435Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:0584c367146c0713bb9773b7157c27ebe32715e1268ad219cdd5130441f1ff8f","observation_id":"199439c5-c9b7-4fee-9128-153fdd84d8db","resolution":{"observed_at":"2026-08-07T10:18:53.536049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:53.076189Z","title":"Exposing fake images with forensic similarity graphs.IEEE Journal of Selected Topics in Signal Processing, 14(5):1049–1064, 2020","venue":null,"work_id":"de0d1ef9-6fe1-45d1-9559-2aded0c3149c","year":2020},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.766710Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:5470d4467f2645fff9df204afe1ca6d803cb8bb66d71f9e273d7e8b4ccc830ab","observation_id":"9ccabdf8-cae1-48a7-b544-6a7346f70b67","resolution":{"observed_at":"2026-08-07T10:18:53.190529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:52.756428Z","title":"Detecting gan- generated imagery using saturation cues","venue":null,"work_id":"47da0d6b-f577-4e33-bc51-14ff727bb6cc","year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.860129Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:cee319a7f0015980e41cc06f9865fd890106b9f75779995fd815b69584bddb4b","observation_id":"a1d086f7-0a36-404c-be13-806965adc08c","resolution":{"observed_at":"2026-08-07T10:18:52.921658Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:52.496816Z","title":"Hierarchical frequency-assisted interactive networks for face manipulation detection.IEEE Transac- tions on Information Forensics and Security, 17:3008–3021,","venue":null,"work_id":"c4cf7b34-f008-449f-8015-46d12ad323a8","year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.921150Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:4070a2eb158c1e8f283a612ea4468592aa94073dc8dce0930d1c6846a7af5970","observation_id":"d29bd27b-4214-4c47-bc32-9bb90d755c0b","resolution":{"observed_at":"2026-08-07T10:18:52.625585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:52.185581Z","title":"F 2 trans: High-frequency fine-grained transformer for face forgery detection.IEEE Transactions on Information Forensics and Security, 18:1039–1051, 2023","venue":null,"work_id":"225b84d9-c328-454b-89d0-fa1a9b3a916a","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:40.996545Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:21747bb8512caabdf50dba9684951f4844f4255c4ab2be9eed2ca727713fc1e3","observation_id":"50e3180f-ae26-4384-8b97-743a4f0dc78c","resolution":{"observed_at":"2026-08-07T10:18:52.316311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:51.840972Z","title":"Self-supervised distilled learning for multi-modal mis- information identification","venue":null,"work_id":"fa9987b3-fe34-4290-a424-c2bbc934c872","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.123516Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:1a40b49071982aa7d2eaf314617441f0917a1948e268ac766fbf5d51b0cd51d7","observation_id":"38cfb14e-fb75-42a3-ba0d-c21f5bb0ac4e","resolution":{"observed_at":"2026-08-07T10:18:52.036092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:51.566613Z","title":"Laa-net: Localized artifact attention network for quality-agnostic and generalizable deepfake de- tection","venue":null,"work_id":"ee6135e3-f761-4272-a936-92d6c97514e8","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.221985Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:df5f892df8351d87f31e8b77516508e81884e8a017ca174e34d9ab62dbc7dcd9","observation_id":"bc997ab0-0ad1-4fda-9419-250d41568d1b","resolution":{"observed_at":"2026-08-07T10:18:51.710322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:51.222201Z","title":"Capsule-forensics: Using capsule networks to detect forged images and videos","venue":null,"work_id":"8eaa3bb1-a50a-47a4-a282-2da9299f42e6","year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.367279Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:c4ee87c84ad7c150eaf2eb87553c778f8b9d34dc2a8856fb98ad699b40c6f7ce","observation_id":"51f35dd8-4296-4669-b3bd-d9ae3f0273b4","resolution":{"observed_at":"2026-08-07T10:18:51.358053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:50.900299Z","title":"Avff: Audio-visual feature fusion for video deepfake detection","venue":null,"work_id":"d22714e3-35cb-4148-9bce-127c7907cb19","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.468866Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:742041272e7495f6f523a41e804b6ed50b3768093e698c933049a1063532d533","observation_id":"e98aec79-ff82-4d2e-b9f5-af43548ddf1b","resolution":{"observed_at":"2026-08-07T10:18:51.072219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:41.581303Z","title":"Deepfake generation and detection: A benchmark and survey.arXiv preprint arXiv:2403.17881,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.581303Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:2410fe21b7c625e1412cadff3d3c3f6cfb41e834e729e4c384f989624cc40d65","observation_id":"7432cb68-bdd2-49ca-80d5-6f89e62cdaab","resolution":{"observed_at":"2026-08-07T10:18:41.581303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:50.559339Z","title":"Deepfake text detec- tion: Limitations and opportunities","venue":null,"work_id":"2055e5c7-afd9-49e7-ac11-2d39b9d45774","year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.735845Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:86c9e5ba2210f0eafb8593e6ed371e7cc8cb9de4eb806030d68cb1025e2f0aea","observation_id":"5526bf16-58e2-4781-8175-0b499c0c639e","resolution":{"observed_at":"2026-08-07T10:18:50.734399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:50.315243Z","title":"Thinking in frequency: Face forgery detection by min- ing frequency-aware clues","venue":null,"work_id":"750eb591-f8ed-4bca-b798-e3b26a85b135","year":2020},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.815141Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:9ffac56fde2a49d5124cb5fe8cd905aed19d928943d7fc13824735fdedd9450e","observation_id":"21070ce8-15bb-4b43-ac06-b934df56e9d4","resolution":{"observed_at":"2026-08-07T10:18:50.444620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:41.933471Z","title":"Language models are unsu- pervised multitask learners.OpenAI blog, 1(8):9, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:41.933471Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:b5f138fe54c43f00f39f75bf6b67d295b170e3126c90af564a9cf7470a8bfc94","observation_id":"1a1b8a5d-a546-4fdd-9638-f1e14f4ad041","resolution":{"observed_at":"2026-08-07T10:18:41.933471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:42.060265Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.060265Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:ac7e68f6553b2e9b8cfab1675e0732a2f604f3116a5f7e5ad70f066b26082f98","observation_id":"4280bb5b-4405-4b2e-80a8-a4272de1ec33","resolution":{"observed_at":"2026-08-07T10:18:42.060265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:50.022491Z","title":"Detecting and grounding multi-modal media manipulation","venue":null,"work_id":"b5ac4c5b-ce8d-47bf-a6de-1b0b7e1473ba","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.151803Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:18072fcf33d410bd0814ca5d084230558ee427f23871aaf559c864244c240807","observation_id":"a65bb5da-e787-4581-aeef-0fc3f30bb119","resolution":{"observed_at":"2026-08-07T10:18:50.139866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:49.811440Z","title":"Detecting and grounding multi-modal media manip- ulation and beyond.IEEE Transactions on Pattern Analysis and Machine Intelligence, 2024","venue":null,"work_id":"d1f32388-0eed-4969-87c1-3e26c75937b7","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.261517Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:75fe6b55e272967f22f6ac70518d02b817ada0ec9c4af13db337af267a16a143","observation_id":"e26c4480-7a02-4fc2-af9f-6ac5c329770d","resolution":{"observed_at":"2026-08-07T10:18:49.907688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:42.357691Z","title":"Learning on gradients: Generalized arti- facts representation for gan-generated images detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.357691Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:65bcfc4b88c9b926baa9dad39c853a690786d46d6db3f5a044936679dffd0127","observation_id":"48c38a3b-9e2b-47c5-9ad7-876da3e9ac45","resolution":{"observed_at":"2026-08-07T10:18:42.357691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:49.517134Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":"42989e45-cb8b-40b5-9510-88f317329b1e","year":2017},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.450291Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:5d4f7e89fd3247f0c079ae17573185fcea6edb34d21eb79ad17b4aef3ed2d665","observation_id":"4ef2c1f0-61b1-4d8c-b6cf-a415e2b5b243","resolution":{"observed_at":"2026-08-07T10:18:49.673865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:49.158348Z","title":"Exploiting modality- specific features for multi-modal manipulation detection and grounding","venue":null,"work_id":"d4efa6d7-d5c7-405a-af5f-e7b9319e1dde","year":2024},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.545481Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:7fb9f29bf58f57fb0584905169c4840d4f0abccb44f57f7e8740a2f6a28976f0","observation_id":"7a615101-fa26-4e6a-89b1-05a6c96d7887","resolution":{"observed_at":"2026-08-07T10:18:49.325867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:48.911196Z","title":"Noise based deepfake de- tection via multi-head relative-interaction","venue":null,"work_id":"0bffa5be-0f1a-472f-8ff5-7e6b6e32c9b1","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.647953Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:76ef22b43b651f39356c991aeb105d9b07bf86e6587bc29bf1c4e42f7c36d749","observation_id":"1f7176c0-c327-4fc8-827a-8c41dbe46276","resolution":{"observed_at":"2026-08-07T10:18:49.023150Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:48.594576Z","title":"Eann: Event adver- sarial neural networks for multi-modal fake news detection","venue":null,"work_id":"fe2efdff-ff9f-4c71-bd35-33e1eff53302","year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.749292Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:ddca2572240e43a22b00565792cd7387e02786a62add27efd803683688f18e58","observation_id":"326a2965-0712-411b-9fe4-12fdafca56b0","resolution":{"observed_at":"2026-08-07T10:18:48.732082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:48.293579Z","title":"Add: Frequency attention and multi- view based knowledge distillation to detect low-quality com- pressed deepfake images","venue":null,"work_id":"77597930-7301-42e9-9ae3-8741f48828f0","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:42.927006Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:85040b8ca6da291e4116d5779ddb61c57d71141bfae85c5339081d7069a8ca49","observation_id":"fc0be62b-035a-444b-8160-7aa51b285d0d","resolution":{"observed_at":"2026-08-07T10:18:48.430658Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.01057","last_updated":"2020-10-02T15:38:03Z","snapshot_observed_at":"2026-08-14T12:27:14.903596Z","submitted_at":"2020-10-02T15:38:03Z","title":"LUKE: Deep Contextualized Entity Representations with Entity-aware Self-attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.01057","snapshot_observed_at":"2026-08-07T10:18:43.063487Z","title":"Luke: Deep contextualized entity representations with entity-aware self-attention.arXiv preprint arXiv:2010.01057, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:43.063487Z"},"links":{"cited_paper":"/paper/2010.01057","citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:306e710134e24d5f19d2eea03f7ee0b2b1a2e0cf92ea3f9bf74b46b0ea01483c","observation_id":"7b741265-4481-47ca-93d6-7de199e97dcc","resolution":{"observed_at":"2026-08-07T10:18:43.063487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:47.954088Z","title":"Unified contrastive learning in image-text-label space","venue":null,"work_id":"49c5f82e-7346-41fd-8d5e-ac70527aea57","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:43.181618Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:d99cbee807b801e56d10c7c428c7cb9196cebb29e51ee0c908e94ac156ce2734","observation_id":"62262f53-224c-49e5-b1c8-5d01a559eab6","resolution":{"observed_at":"2026-08-07T10:18:48.101731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:47.643253Z","title":"Avoid-df: Audio-visual joint learning for detecting deepfake","venue":null,"work_id":"f589de7e-e2f8-4bb0-a448-1e1ba2877f9c","year":2015},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:43.336394Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:f4dffbce6956793fe7033768762acf8f4df30f801bc0496e2bb26a383756cfe6","observation_id":"30f1c272-6a9b-43ae-9a80-36dfb1db2384","resolution":{"observed_at":"2026-08-07T10:18:47.754647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:47.289076Z","title":"Masked relation learning for deepfake detection","venue":null,"work_id":"7f2f7dae-f3b2-4103-a1d7-9c519599f5bc","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:43.472254Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:451cca5dd914a5b7a8e2d534c384d7ae28c0edc2d3b344b87ac2e55ca5c0cdd5","observation_id":"e4680f1e-da14-4390-826b-12ee1fa9e873","resolution":{"observed_at":"2026-08-07T10:18:47.474493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:47.039894Z","title":"Dynamic differ- ence learning with spatio-temporal correlation for deepfake video detection.IEEE Transactions on Information Foren- sics and Security, 2023","venue":null,"work_id":"1cdd3b76-fea5-4f16-8612-89287134cc76","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:43.597185Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:58caf1c1495a0562a2c5d41f71fcaf17a448e92ffe632657697cc2038d5cc107","observation_id":"2aa206df-576c-4f9d-8ae0-71d010a2d957","resolution":{"observed_at":"2026-08-07T10:18:47.152714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:46.681922Z","title":"Bootstrapping multi-view rep- resentations for fake news detection","venue":null,"work_id":"763171cb-d926-4076-95f2-bce321305a95","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:43.725910Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:d4b2071e2789a61d9c6f90ef4e77c07d1036033fe693f7f827c7d736d328ec6a","observation_id":"263db666-2511-498c-8cbb-f5d60b35e574","resolution":{"observed_at":"2026-08-07T10:18:46.855527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:46.363540Z","title":"De- fending against neural fake news.Advances in neural infor- mation processing systems, 32, 2019","venue":null,"work_id":"8f26d800-bed1-4225-884b-01924117a5cb","year":2019},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:43.860003Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:e46efc4a6984460d4bacf2c88a5a4f2455973aa60b13a965b908af8e93aa25c1","observation_id":"8a294af9-a640-4eb0-9148-de8e545a14dc","resolution":{"observed_at":"2026-08-07T10:18:46.516032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:46.130757Z","title":"Multi-attentional deep- fake detection","venue":null,"work_id":"42e4fc54-400a-4c47-a602-4d557386233d","year":2021},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:44.009030Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:91c91e724bc6e5351c99bfe7b5ecc30f2f79f6f406f62a473f8e147712ef9c47","observation_id":"b520f37a-c24b-4b7a-bdda-748ada87a893","resolution":{"observed_at":"2026-08-07T10:18:46.227205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:45.931465Z","title":"Learning self-consistency for deepfake detection","venue":null,"work_id":"f946f13a-c09a-4d77-9cd5-dd090437e281","year":2021},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:44.081791Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:8574da5b48733e45864754e6ff33222543c2ace02f58aaa1d927cc3af214ffc8","observation_id":"22b4bfaf-ef36-44ba-880e-448eacddd51c","resolution":{"observed_at":"2026-08-07T10:18:46.051830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:45.692991Z","title":"A survey of deep facial attribute analysis.International Journal of Computer Vision, 128:2002–2034, 2020","venue":null,"work_id":"ae1208c3-ef3b-4605-afe7-32af6dd8bae6","year":2002},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:44.220404Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:926690caecf92488988b65577c019d270fae1f2cbe8a30168f56f6017b02fb06","observation_id":"23e5772a-3428-4d5f-958a-109a80eb4a4e","resolution":{"observed_at":"2026-08-07T10:18:45.826780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:45.487798Z","title":"Two-stream neural networks for tampered face detection","venue":null,"work_id":"ddf86811-8b94-49a9-add0-005be445f8b6","year":null},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:44.342281Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:ff946fb765406635fe4317af89f08a0292d80eedb69d1f5477fc1c6313bfe052","observation_id":"f908ef1d-926b-402d-bd9a-8e8a61762c57","resolution":{"observed_at":"2026-08-07T10:18:45.577925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:45.247865Z","title":"Multi-modal fake news detec- tion on social media via multi-grained information fusion","venue":null,"work_id":"ccf658cb-364a-4c40-b7b9-aa35cfb83f22","year":2023},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:44.508620Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:2f44f6188df2f75cc45eee49278371ca4edb55836d1a348d597d39ca67d01b55","observation_id":"886156e6-c9fd-4dbd-b033-6973bf827b12","resolution":{"observed_at":"2026-08-07T10:18:45.374466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:44.963663Z","title":"Generalizing to the future: Mitigating entity bias in fake news detection","venue":null,"work_id":"6f76e9ae-33dd-423a-8568-ecdd1ccaefe5","year":2022},"citing_paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:44.636368Z"},"links":{"citing_paper":"/paper/2506.05890"},"observation_digest":"sha256:0e8893be12113f129691ec3c54ec1884a2228c924f8ac0e691bd21d08bbef0b9","observation_id":"b4f8de7f-a2c3-41c9-8e12-b91bd2846310","resolution":{"observed_at":"2026-08-07T10:18:45.085224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.05890","last_updated":"2025-06-06T08:59:07Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T20:25:03.158160Z","submitted_at":"2025-06-06T08:59:07Z","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation"},"reference_resolution":{"displayed":63,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":46},"total_outbound_references":63},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 63 of 63 outbound references and 2 inbound Pith citation observations for arXiv:2506.05890."}