{"as_of":"2026-08-10T03:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f01f984a984b705a82efcedf346bf2ed7233e1f3b9a024ba895219e7ff06f622","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T04:34:23.056458Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.05670/citation-record","integrity":"/paper/2608.05670/integrity","json":"/paper/2608.05670/citation-record.json","paper":"/paper/2608.05670"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.11919","last_updated":"2024-11-28T13:35:56Z","snapshot_observed_at":"2026-07-06T19:52:11.774784Z","submitted_at":"2024-11-18T04:06:04Z","title":"VL-Uncertainty: Detecting Hallucination in Large Vision-Language Model via Uncertainty Estimation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11919","snapshot_observed_at":"2026-08-08T04:34:22.718875Z","title":"Vl-uncertainty: Detecting hallucination in large vision-language model via uncertainty estimation.arXiv preprint arXiv:2411.11919, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.718875Z"},"links":{"cited_paper":"/paper/2411.11919","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:c867cd296c1b0c44472f7ecf45ae95e2bfc113d6aca680cb355290ffd9a0c0fd","observation_id":"d8450b18-d026-4d72-800f-7b14558c59b5","resolution":{"observed_at":"2026-08-08T04:34:22.718875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.724591Z","title":"Questioning the stability of visual question answering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.724591Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:0d3a6eb3c1580e0171f81550987708032c539fb688bf71cebdcd1cc8f6f98d2d","observation_id":"8e9e669a-16d2-4301-b7e3-1bee98167ba1","resolution":{"observed_at":"2026-08-08T04:34:22.724591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.979768Z","title":"Efficient test-time scaling for small vision-language models","venue":null,"work_id":"b518ba6c-eb05-448c-a727-d9348e5128d3","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.729343Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:48151a79dadaba84bbd0a95606bc6fcb0456d30a17125de5a07965ac72c37f7a","observation_id":"291444bc-2dd8-4952-8eb4-8b6fe1c8f41e","resolution":{"observed_at":"2026-08-08T04:34:23.984472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.734303Z","title":"Multi-llm debate: Framework, principals, and interventions.Advances in Neural Information Processing Systems, 37:28938–28964, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.734303Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:259ba127bf0d2e7f61c00b9ba9c1275a966c7d79c977c76cae913248b41161f7","observation_id":"e18ebd78-67fa-49bb-a4d7-30900cb577cc","resolution":{"observed_at":"2026-08-08T04:34:22.734303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-08T04:34:22.739239Z","title":"Self-consistency improves chain of thought reasoning in language models.arXiv preprint arXiv:2203.11171, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.739239Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:86391c925d10f0d345a9faddd1a203dee14d0cdab13d16e0523eb33d0d269317","observation_id":"94f291c1-af88-4a44-9efd-50b0df790b75","resolution":{"observed_at":"2026-08-08T04:34:22.739239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.744339Z","title":"Consistency and uncertainty: Identifying unreliable responses from black- box vision-language models for selective visual question answering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.744339Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:9804d201e6ad16b99fb4a5f10a5f0bc33ccb1aef4ef0ebbd7db0f9ce09e8c57e","observation_id":"cf8a19ea-9e1c-41e4-9244-11906c5e7eff","resolution":{"observed_at":"2026-08-08T04:34:22.744339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.945996Z","title":"Decompose and compare consistency: Measuring vlms’ answer reliability via task-decomposition consistency comparison","venue":null,"work_id":"a0859a76-2671-4782-b33b-ce87044e8bf3","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.749605Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:f7f341e5a6fc53541187b2a7ccf33a69dc83bb52ce7c8c4f01548b2344cc7fa1","observation_id":"e0098460-5a52-42ed-a8d1-22f42d7d02eb","resolution":{"observed_at":"2026-08-08T04:34:23.950491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.09664","last_updated":"2023-04-15T12:55:45Z","snapshot_observed_at":"2026-07-06T14:53:27.667483Z","submitted_at":"2023-02-19T20:10:07Z","title":"Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.09664","snapshot_observed_at":"2026-08-08T04:34:22.754207Z","title":"Semantic uncertainty: Linguistic invariances for uncertainty estimation in natural language generation.arXiv preprint arXiv:2302.09664, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.754207Z"},"links":{"cited_paper":"/paper/2302.09664","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:f5bdc79317130011f9474825b42d26934a97afc75779a2e5e7b3da8a5e38ba0f","observation_id":"0ee27387-dfc1-44c7-b617-2707784ac49e","resolution":{"observed_at":"2026-08-08T04:34:22.754207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.780102Z","title":"Detecting hallucinations in large language models using semantic entropy.Nature, 630(8017):625–630, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.780102Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:77e02e65de2986e975f01347eaa8e784930ac95e37ba10fa2a8ace3052b4d877","observation_id":"4de494b6-d26b-413d-b9b0-2a97935f6137","resolution":{"observed_at":"2026-08-08T04:34:22.780102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15376","last_updated":"2026-04-15T20:47:08Z","snapshot_observed_at":"2026-07-06T23:02:58.361418Z","submitted_at":"2026-04-15T20:47:08Z","title":"Zoom Consistency: A Free Confidence Signal in Multi-Step Visual Grounding Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.15376","snapshot_observed_at":"2026-08-08T04:34:22.806745Z","title":"Zoom consistency: A free confidence signal in multi-step visual grounding pipelines.arXiv preprint arXiv:2604.15376, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.806745Z"},"links":{"cited_paper":"/paper/2604.15376","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:c578364d85f40fce04b52d3eecff30b56e17099c6433b90af65f5d8707fdd7e0","observation_id":"2dfa8160-52a3-4f9d-9fc8-bc8f1c3d96e5","resolution":{"observed_at":"2026-08-08T04:34:22.806745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.922795Z","title":"Vauq: Vision-aware uncertainty quantification for lvlm self-evaluation","venue":null,"work_id":"c509ffb6-c006-4240-b2ee-9d27a16891f7","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.851896Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:ff1021a3d359f1b0fed635a7d9a21825c35602209de9b2f921613e8728c501a7","observation_id":"f0a2656f-c19f-4150-8fb3-5a990c6c7b66","resolution":{"observed_at":"2026-08-08T04:34:23.927081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.907511Z","title":"Vl-calibration: Decoupled confidence calibration for large vision-language models reasoning","venue":null,"work_id":"a029f9d0-731e-4f1d-a9f2-09147a260f15","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.872825Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:563c3a798d97b894b2a68aab40db73946625c7d1b99ac7740ae40d603661aa78","observation_id":"f8d698ac-680a-42c0-9e18-136d2fc439fd","resolution":{"observed_at":"2026-08-08T04:34:23.912793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.892630Z","title":null,"venue":null,"work_id":"8476976b-bae9-437f-b527-1eebdf7812db","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.895077Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:bbb4ecd83850ff35f78548c8429b7a2a56352651846a31415d092cfbb62b2a06","observation_id":"1be52bab-4a6c-498e-9077-525e51081542","resolution":{"observed_at":"2026-08-08T04:34:23.897455Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1710.07300","last_updated":"2018-02-22T22:50:42Z","snapshot_observed_at":"2026-08-04T06:17:25.388464Z","submitted_at":"2017-10-19T18:01:38Z","title":"FigureQA: An Annotated Figure Dataset for Visual Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1710.07300","snapshot_observed_at":"2026-08-08T04:34:22.922646Z","title":"Figureqa: An annotated figure dataset for visual reasoning.arXiv preprint arXiv:1710.07300, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.922646Z"},"links":{"cited_paper":"/paper/1710.07300","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:2658a620a5dda27743dd7af8ad7450f49ea5b230a1143e37202abe7106b843b1","observation_id":"be3b93ca-1489-4db5-b200-8ea065cfb549","resolution":{"observed_at":"2026-08-08T04:34:22.922646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.877136Z","title":"Plotqa: Reasoning over scientific plots","venue":null,"work_id":"c2ef4e71-d2f2-48c0-b98d-0ef7270a8b6e","year":2020},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.944732Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:39660bf92e7535e09412913618492a840d6eea653b3ac7496462cb1b45b3209e","observation_id":"91a9a571-af79-410c-8cca-feaca54835d6","resolution":{"observed_at":"2026-08-08T04:34:23.882252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.959247Z","title":"Chartqa: A benchmark for question answering about charts with visual and logical reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.959247Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:5e74cccb77512475e8f717def67b1a401be3aff130852a8f0feaaad655c010d7","observation_id":"76933f1f-993c-48bf-9550-082e13848bbf","resolution":{"observed_at":"2026-08-08T04:34:22.959247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.850669Z","title":"Chartverse: Scaling chart reasoning via reliable programmatic synthesis from scratch","venue":null,"work_id":"ea4d00d2-5b69-418b-ac55-0680aac28ab5","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.964070Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:4d5ce546ae5f20746982c9299930995acc072e72e7c9aa606043d57ad434444a","observation_id":"ef0ea0cd-0692-46ea-b2df-3cc67458edfe","resolution":{"observed_at":"2026-08-08T04:34:23.856452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.833237Z","title":"Chart-rl: Generalized chart comprehension via reinforcement learning with verifiable rewards","venue":null,"work_id":"a5b93b3b-a103-4abe-84e8-01fd32234f34","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.968703Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:aff8730cbbcc0d08f05e8ae2f5eb7f21b50eb7d587624ef004d7033220d1f5c8","observation_id":"02ba6739-5a20-4ecf-a225-60c48dcbd307","resolution":{"observed_at":"2026-08-08T04:34:23.839203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.973225Z","title":"Chart-rvr: Reinforce- ment learning with verifiable rewards for explainable chart reasoning.arXiv preprint arXiv:2510.10973, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.973225Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:1d255e391a511dff497201578159509c10a02807d9427459c7cc21eda1a7a58f","observation_id":"49a66d6a-5a60-4c1b-a7ba-48dd6f9878e9","resolution":{"observed_at":"2026-08-08T04:34:22.973225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19492","last_updated":"2025-05-31T02:35:38Z","snapshot_observed_at":"2026-08-07T12:04:49.777765Z","submitted_at":"2025-05-31T02:35:38Z","title":"ChartGen: Scaling Chart Understanding Via Code-Guided Synthetic Chart Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.19492","snapshot_observed_at":"2026-08-08T04:34:22.977760Z","title":"Chartgen: Scaling chart understanding via code-guided synthetic chart generation.arXiv preprint arXiv:2507.19492, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.977760Z"},"links":{"cited_paper":"/paper/2507.19492","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:05c2d7a0188da1eeb3a54cee86f4ea6a3da1900f88041913b62eaae9bfb8e435","observation_id":"232e86df-497a-4702-9de9-47195476250b","resolution":{"observed_at":"2026-08-08T04:34:22.977760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.817445Z","title":"Unraveling the truth: Do vlms really understand charts? a deep dive into consistency and robustness","venue":null,"work_id":"927c6b21-8e5d-4cec-ad65-8781211eec34","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.982359Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:362256a21d6a4356336236ff1537e53f9677618a322c93aaefd75e1100f560fd","observation_id":"6f94e84d-a710-4113-aa3e-03287894ce84","resolution":{"observed_at":"2026-08-08T04:34:23.822707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:22.987169Z","title":"Losing the plot: How vlm responses degrade on imperfect charts.arXiv preprint arXiv:2509.18425, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.987169Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:2f3c78207e4132df2dd30d0552a5b0fff3f6eefeb66301aa951478b2dbc0426c","observation_id":"3d84fd80-551f-4539-ba7b-85f5847d8e25","resolution":{"observed_at":"2026-08-08T04:34:22.987169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12506","last_updated":"2026-05-21T07:28:16Z","snapshot_observed_at":"2026-07-06T22:45:44.135338Z","submitted_at":"2026-02-13T01:12:00Z","title":"On Robustness and Chain-of-Thought Consistency of RL-Finetuned VLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.12506","snapshot_observed_at":"2026-08-08T04:34:22.991328Z","title":"On robustness and chain-of-thought consistency of rl-finetuned vlms.arXiv preprint arXiv:2602.12506, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.991328Z"},"links":{"cited_paper":"/paper/2602.12506","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:9ddf51480e63ff0047156e586a448597ccb27ee7f9267d255d11e8fff42dd4bf","observation_id":"7c2d8530-723c-4b78-8c3c-fc111247c42d","resolution":{"observed_at":"2026-08-08T04:34:22.991328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.801678Z","title":"Perception-r1: Advancing multimodal reasoning capabilities of mllms via visual perception reward","venue":null,"work_id":"efe3beaf-0d7b-43f8-8761-ba658f56f2a9","year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:22.996226Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:3e35425ed7f61ac3f3aaad3a06ca3e214e4eff4d33e36e2e8f281002fe54d9d2","observation_id":"434fa5ec-450c-44ad-b825-1bf2cc705593","resolution":{"observed_at":"2026-08-08T04:34:23.806830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.000518Z","title":"V-fat: Benchmarking visual fidelity against text-bias.arXiv preprint arXiv:2601.04897, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.000518Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:3b0a9043fb3754cda22aec744a543949e8548badbd6c3781b1a127687ce2cff8","observation_id":"ba2de16b-0e19-466b-ab7d-dfd83ea8c0aa","resolution":{"observed_at":"2026-08-08T04:34:23.000518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.004811Z","title":"Cdh-bench: A commonsense- driven hallucination benchmark for evaluating visual fidelity in vision-language models.arXiv preprint arXiv:2603.27982, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.004811Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:48a9e1331aa2f5e96081ac92e6fcb0f98e7685b48277cb3f8e2450004f715827","observation_id":"8078357c-3d8d-4bc3-b03a-774b03dd47e8","resolution":{"observed_at":"2026-08-08T04:34:23.004811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.10400","last_updated":"2026-06-09T04:18:38Z","snapshot_observed_at":"2026-08-06T05:17:07.543850Z","submitted_at":"2026-06-09T04:18:38Z","title":"Do Vision-Language Models See or Guess? Measuring and Reducing Textual-Prior Reliance with a Phrasing-Controlled Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.10400","snapshot_observed_at":"2026-08-08T04:34:23.009317Z","title":"Do vision-language models see or guess? measuring and reducing textual-prior reliance with a phrasing-controlled benchmark.arXiv preprint arXiv:2606.10400, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.009317Z"},"links":{"cited_paper":"/paper/2606.10400","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:5bb09eea63ca1ad3b922aa76757c4987ad2020bbf18385b5fa71ab7df3299bfd","observation_id":"9f4b3506-1e79-4111-a926-4c3af9b59b1d","resolution":{"observed_at":"2026-08-08T04:34:23.009317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.014499Z","title":"On the foundations of noise-free selective classification.Journal of Machine Learning Research, 11(5), 2010","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.014499Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:a7781cf07e8db097ccea5b392f2315dbc66276946c68f5d63bd71b29316e1686","observation_id":"01de0153-e369-4b08-9685-0c519f3fb323","resolution":{"observed_at":"2026-08-08T04:34:23.014499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.018756Z","title":"On calibration of modern neural networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.018756Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:f490f40f63aebbfe44aa265de302f3dcc7a5aba6ec6a904e1b5ca720bf30a33e","observation_id":"b7caf36b-99b5-48eb-9345-7f336fbfabea","resolution":{"observed_at":"2026-08-08T04:34:23.018756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.767011Z","title":"selective prediction","venue":null,"work_id":"c614e192-1821-4db7-b881-76e30c9713f4","year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.023142Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:cbf8c3b49d6f33b98f65bd79d47917def2d2d449984989c2315ecc893d524cf7","observation_id":"c00800a2-f918-4bf0-8388-7e0954a91eea","resolution":{"observed_at":"2026-08-08T04:34:23.772362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-08T04:34:23.027631Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters.arXiv preprint arXiv:2408.03314, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.027631Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:8dad930233de632b21f0913ada390cb30465a68c63ef628312e5bf6433cc5440","observation_id":"5a86d081-bba8-4623-b336-48a798995f1e","resolution":{"observed_at":"2026-08-08T04:34:23.027631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-08T04:34:23.032333Z","title":"Qwen2.5-vl technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.032333Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:1a2646c8dd671086c3975877516693141dfa4be04a4e7437d6ac165251dea6b1","observation_id":"dbd5e928-f152-4ede-b039-94f0d677288c","resolution":{"observed_at":"2026-08-08T04:34:23.032333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.037110Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites.Science China Information Sciences, 67(12):220101, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.037110Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:02a3bb665d6a675bf38d86d64027ee84e478c01adcf09c38047d2e7cc024b627","observation_id":"287517cb-49c6-4dde-a2fe-106c27557308","resolution":{"observed_at":"2026-08-08T04:34:23.037110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.742343Z","title":"Equivalence guaranteed","venue":null,"work_id":"773cb8a7-0a1c-4e7d-aa13-df5f1c4d7254","year":2025},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.041257Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:0222ad43d5e6defc4a50c1a1923eabb67aeb6b16d3a7f0108cc1763d5d9d9619","observation_id":"07d488ce-5029-4dbc-8eae-32a09c129c48","resolution":{"observed_at":"2026-08-08T04:34:23.746762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.727149Z","title":"When the answer space has two elements and 𝐾 is odd the first inequality is an equality, and the middle quantity is strictly decreasing along odd𝐾","venue":null,"work_id":"ad3678ac-5104-43b3-a33e-a46b2975e28e","year":null},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.046022Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:797184615b6b49b46169e645a31110ab9e3ddd17e84637b1ac753221f548eacd","observation_id":"82ad8d73-8ef7-4751-ab81-f49ffb356e5b","resolution":{"observed_at":"2026-08-08T04:34:23.732491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.712522Z","title":null,"venue":null,"work_id":"835e901e-d012-4a43-b040-07671125441d","year":null},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.051294Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:7fdff22bf934d33b9e8deedea10e134e382662b3016f9f25ede239f8a9182f04","observation_id":"865fc1df-4c31-44a8-833e-280e49b81d0b","resolution":{"observed_at":"2026-08-08T04:34:23.717004Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:34:23.674895Z","title":"too many arguments","venue":null,"work_id":"975c2900-bf49-419d-8ed0-16ad5017b30c","year":null},"citing_paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-08T04:34:23.056458Z"},"links":{"citing_paper":"/paper/2608.05670"},"observation_digest":"sha256:fdc8dae30c2fcdc90b61b1d2833e68ed5e61d42a956c9e2b551482e2cb97035b","observation_id":"a9b7f48b-c950-40dd-a8a7-6ccfe160cb49","resolution":{"observed_at":"2026-08-08T04:34:23.702478Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.05670","last_updated":"2026-08-06T07:08:05Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T02:27:10.126985Z","submitted_at":"2026-08-06T07:08:05Z","title":"When Does Consensus Mean Correctness? Measuring the Agreement-Accuracy Coupling with Semantics-Preserving Re-Rendering"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":12},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 0 inbound Pith citation observations for arXiv:2608.05670."}