{"as_of":"2026-08-16T05:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8efa093f99724196bbafa4143552e19a6aa5488e4d598cefcc58e224a2900c44","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:51:34.192526Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-28T23:02:46.504962Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-11T21:35:10.148411Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.04363","last_updated":"2024-12-05T17:22:04Z","snapshot_observed_at":"2026-08-12T01:29:04.014466Z","submitted_at":"2024-12-05T17:22:04Z","title":"Challenges in Trustworthy Human Evaluation of Chatbots","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T21:35:10.148411Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2412.04363"},"observation_digest":"sha256:caaecd683817935c25a6da4f998b9f08617a58655144c5bbea8a692f0f74a0ba","observation_id":"5bb5e92d-9583-499e-abeb-39b9c576a79d","resolution":{"observed_at":"2026-08-11T21:35:10.148411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-11T21:10:35.157549Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.04947","last_updated":"2025-05-29T05:29:28Z","snapshot_observed_at":"2026-08-12T01:40:30.259470Z","submitted_at":"2024-12-06T11:07:44Z","title":"C$^2$LEVA: Toward Comprehensive and Contamination-Free Language Model Evaluation","version":3},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T21:10:35.157549Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2412.04947"},"observation_digest":"sha256:f2dd22cff1c0392e2194243b41f3b9d0453c2c2799dd482637782cb33abe8657","observation_id":"b1fd342f-1825-465d-aacd-248e310052f3","resolution":{"observed_at":"2026-08-11T21:10:35.157549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-10T22:51:45.820775Z","title":"CoRR, abs/2406.06565","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.00560","last_updated":"2025-02-11T10:02:55Z","snapshot_observed_at":"2026-08-15T05:53:29.917943Z","submitted_at":"2024-12-31T17:46:51Z","title":"Re-evaluating Automatic LLM System Ranking for Alignment with Human Preference","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T22:51:45.820775Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2501.00560"},"observation_digest":"sha256:9e7fbbbb1d5bb713a93eebc19c4493d3cac252d2f30d08c4fd1552692b9cacb3","observation_id":"082c3b24-4b68-4219-8160-83facebafea4","resolution":{"observed_at":"2026-08-10T22:51:45.820775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-10T15:39:33.154626Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13802","last_updated":"2025-03-09T16:39:06Z","snapshot_observed_at":"2026-08-15T08:30:40.334498Z","submitted_at":"2025-01-23T16:21:15Z","title":"Enhancing LLMs for Governance with Human Oversight: Evaluating and Aligning LLMs on Expert Classification of Climate Misinformation for Detecting False or Misleading Claims about Climate Change","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T15:39:33.154626Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2501.13802"},"observation_digest":"sha256:df4cefd045d946e8e56ae87bde17d60309729a79308f422d1a6a342422a5a77d","observation_id":"be3cb942-fb86-4ae2-9b9c-c642367a9f62","resolution":{"observed_at":"2026-08-10T15:39:33.154626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-10T15:02:43.268780Z","title":"Mixeval: Deriving wisdom of the crowd from llm benchmark mixtures, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.17178","last_updated":"2025-05-27T07:58:46Z","snapshot_observed_at":"2026-08-15T21:47:19.577569Z","submitted_at":"2025-01-24T17:01:14Z","title":"Tuning LLM Judge Design Decisions for 1/1000 of the Cost","version":4},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T15:02:43.268780Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2501.17178"},"observation_digest":"sha256:0d70fe028498fe04c6bfbcf27ae4d0a3e16dc5a2fa4cd6719b4c2a3493bed692","observation_id":"ff803e47-8d9f-4f93-840e-4dd68681c9d5","resolution":{"observed_at":"2026-08-10T15:02:43.268780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-10T00:15:07.772181Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.18251","last_updated":"2025-05-31T09:55:39Z","snapshot_observed_at":"2026-08-14T22:58:24.503561Z","submitted_at":"2025-01-30T10:33:26Z","title":"How to Select Datapoints for Efficient Human Evaluation of NLG Models?","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-10T00:15:07.772181Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2501.18251"},"observation_digest":"sha256:17be58c6501c7b805857fd9863fa21adc25e2057b72c10bc012b1870c9685dcb","observation_id":"6358db70-4ec3-4681-aa24-5c684c816f7e","resolution":{"observed_at":"2026-08-10T00:15:07.772181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-09T23:17:09.188169Z","title":"NexusFlow AI","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.18511","last_updated":"2025-05-23T16:07:21Z","snapshot_observed_at":"2026-08-13T21:37:58.583293Z","submitted_at":"2025-01-30T17:21:44Z","title":"WILDCHAT-50M: A Deep Dive Into the Role of Synthetic Data in Post-Training","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-09T23:17:09.188169Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2501.18511"},"observation_digest":"sha256:f680b7f32bbf2e03dfae5578556f87d4b27359fcbaa127eb3a47803eb293224e","observation_id":"9c174b93-ad88-4921-ba29-320fe7494e34","resolution":{"observed_at":"2026-08-09T23:17:09.188169Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-09T04:08:51.755786Z","title":"Accessed: 2025- 01-16","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.03699","last_updated":"2025-07-23T21:26:26Z","snapshot_observed_at":"2026-08-10T19:03:46.582009Z","submitted_at":"2025-02-06T01:22:06Z","title":"LLM Alignment as Retriever Optimization: An Information Retrieval Perspective","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-09T04:08:51.755786Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2502.03699"},"observation_digest":"sha256:44eeb74c90e9a640414e7b04e35b66d099440c5019b4e36524dc8fb27acce47f","observation_id":"7120ae5c-e864-4839-b264-e5eee69331c8","resolution":{"observed_at":"2026-08-09T04:08:51.755786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-15T20:51:34.192526Z","title":"Mixeval: Deriving wisdom of the crowd from llm benchmark mixtures.arXiv preprint arXiv:2406.06565, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11855","last_updated":"2025-05-17T05:45:16Z","snapshot_observed_at":"2026-08-15T20:44:02.477268Z","submitted_at":"2025-05-17T05:45:16Z","title":"When AI Co-Scientists Fail: SPOT-a Benchmark for Automated Verification of Scientific Research","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T20:51:34.192526Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2505.11855"},"observation_digest":"sha256:9f726582f9d71b0b4b9df59f15b8c28352e026b8f07bc8a0aef186c1b6c5e931","observation_id":"31acb379-afae-4f1f-953f-aa1742758e0c","resolution":{"observed_at":"2026-08-15T20:51:34.192526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-15T20:32:25.231127Z","title":"Mixeval: Deriving wisdom of the crowd from llm benchmark mixtures","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.12808","last_updated":"2025-05-19T07:34:25Z","snapshot_observed_at":"2026-08-15T20:23:52.978911Z","submitted_at":"2025-05-19T07:34:25Z","title":"Decentralized Arena: Towards Democratic and Scalable Automatic Evaluation of Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T20:32:25.231127Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2505.12808"},"observation_digest":"sha256:c9b89096c76478d6ffc41c12cbfc0d16cd565e583330193e727d862bb6ef5441","observation_id":"5c168072-8d54-4881-b6d9-30118242fd5f","resolution":{"observed_at":"2026-08-15T20:32:25.231127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-08-07T10:54:26.336265Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04142","last_updated":"2025-06-04T16:33:44Z","snapshot_observed_at":"2026-08-11T15:46:19.548845Z","submitted_at":"2025-06-04T16:33:44Z","title":"Establishing Trustworthy LLM Evaluation via Shortcut Neuron Analysis","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T10:54:26.336265Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2506.04142"},"observation_digest":"sha256:f5679ed1b36df5078c7e8ab21cf8af8a7925ada584b66c33bad1e326b91fb7f6","observation_id":"77a84dd6-360f-48fe-af96-b2995dd82fd5","resolution":{"observed_at":"2026-08-07T10:54:26.336265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","version":2},"cited_work":{"arxiv_id":"2406.06565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.06565","snapshot_observed_at":"2026-06-28T23:02:46.504962Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","venue":null,"work_id":"86b48ee2-9e1b-494f-8c5b-91454a7e63a7","year":2024},"citing_paper":{"arxiv_id":"2605.31268","last_updated":"2026-05-29T13:01:11Z","snapshot_observed_at":"2026-08-12T23:02:09.228331Z","submitted_at":"2026-05-29T13:01:11Z","title":"Mellum2 Technical Report","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-28T22:58:35.397914Z"},"links":{"cited_paper":"/paper/2406.06565","citing_paper":"/paper/2605.31268"},"observation_digest":"sha256:b858b078f3a9a3c1b05bf4c78de1c16ac69509ef86d684e1b9ea39ca21e86c22","observation_id":"2d8b17e7-3c82-493d-9fb2-3ca2bb6048a2","resolution":{"observed_at":"2026-06-28T23:02:46.506621Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2406.06565/citation-record","integrity":"/paper/2406.06565/integrity","json":"/paper/2406.06565/citation-record.json","paper":"/paper/2406.06565"},"outbound":[],"paper":{"arxiv_id":"2406.06565","last_updated":"2024-10-12T14:13:27Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-12T23:51:54.779130Z","submitted_at":"2024-06-03T05:47:05Z","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2406.06565."}