{"as_of":"2026-08-21T01:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0824cc8ce8257f223a08894a5c2148e2a8c4cf3e06be8e9e49e26f2a97cd6935","coverage":[{"denominator":20,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:50:35.589753Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.09813/citation-record","integrity":"/paper/2506.09813/integrity","json":"/paper/2506.09813/citation-record.json","paper":"/paper/2506.09813"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:37.713801Z","title":"Suppose we have a setKsuch that|K| ≤(α−1)gandKsatisfies positional representation for group sizeg","venue":null,"work_id":"d84cbc48-a6f3-4997-9018-8f47289778c9","year":2015},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.100847Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:1a5b31d67ec83be8a353163396db84fbc72aa99d9c6025f5470e4c4ff2272752","observation_id":"c57aac52-22a9-4036-9b65-d8c1d6e03992","resolution":{"observed_at":"2026-08-07T04:50:37.858463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:38.026971Z","title":"coalition","venue":null,"work_id":"f0d17530-4454-4f9a-b7d6-813af40d86ed","year":2017},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.987732Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:16064d221a4b14f2241846b4403c570a74554502c07ab871041cff2a0f7cebfc","observation_id":"fe0f85e5-9004-42af-ac06-bec14899dcee","resolution":{"observed_at":"2026-08-07T04:50:38.166042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.269088Z","title":"ex- isting subset","venue":null,"work_id":"2cff7daa-b14c-48c0-887c-f36cb64b329b","year":2025},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.589753Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:968a5c9f2f6709831a192175147e999f1628c84c791d9f6e9bf29a43d2516caa","observation_id":"179b307d-bd0b-44c6-ab8b-2a97d59233f7","resolution":{"observed_at":"2026-08-07T04:50:36.338871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:37.158243Z","title":"exact cover by 3 sets","venue":null,"work_id":"ac00432d-fc96-4c19-a66d-9647c45abb63","year":1995},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.294832Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:bb4cd6d2d389d3b26c15fe13e91182c92e3c9cfcc276d557351c6387b4bb77d9","observation_id":"4b06efb5-dea8-41df-aa47-6503a6e0b70a","resolution":{"observed_at":"2026-08-07T04:50:37.345915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.908084Z","title":"Big-G”, “Big-G sparse","venue":null,"work_id":"5827c60f-290f-4734-bcbf-8d364f890a99","year":2022},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.364011Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:cfe3c0b2793c15136e54b3fab6fc888ae72a4d9ed8290f16fdd9cd3777e29808","observation_id":"f49c6c56-08db-4802-bab1-524e6683d252","resolution":{"observed_at":"2026-08-07T04:50:37.045949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.05769","last_updated":"2023-02-12T13:02:27Z","snapshot_observed_at":"2026-08-17T05:35:52.384629Z","submitted_at":"2022-10-11T20:19:11Z","title":"Vote'n'Rank: Revision of Benchmarking with Social Choice Theory","version":3},"cited_work":{"arxiv_id":"2210.05769","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.05769","snapshot_observed_at":"2026-08-07T04:50:35.794918Z","title":"Vote'n'Rank: Revision of Benchmarking with Social Choice Theory","venue":"cs.LG","work_id":"08d9f016-7a06-4b02-b1c5-e67bd02fb391","year":2022},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.653219Z"},"links":{"cited_paper":"/paper/2210.05769","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:b1e900c4a2d5447b90e7643b39e0643598e560d14d554684d2ac6346d4525a47","observation_id":"ae2b18e2-2a99-4110-a9ab-944c2924bd15","resolution":{"observed_at":"2026-08-07T04:50:35.843606Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:37.460322Z","title":"Finally, we can conclude that there mustexist some rankingσ N such that noKwith size|K| ≤ 1 288ϵ2 log(m) satisfiesϵ-positional proportionality","venue":null,"work_id":"350b4b67-0ef4-4c38-aab5-5719871e5a49","year":2001},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.229886Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:973db50c390e20f1cdb7114e84a86dd25c7c31912ab7869b387478f4c1f7b06c","observation_id":"e7dcb8e8-276b-49b0-97d9-265862461d9f","resolution":{"observed_at":"2026-08-07T04:50:37.595715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.719897Z","title":"representative","venue":null,"work_id":"15e8f606-c9ea-4333-9302-8e88305d563b","year":2022},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.456242Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:1234d4f022dc3bff41761909b68bd0f8137260ab3997c932b84c30ca7c2ea6f0","observation_id":"2e825802-02be-4ba3-9f29-68bb5d52dea7","resolution":{"observed_at":"2026-08-07T04:50:36.809215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.419477Z","title":"Core scenarios","venue":null,"work_id":"f870c230-abb7-4c37-aa3e-a56b65b5dc4c","year":2025},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.520700Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:778c9787e725b7c66842ee5f907a49cd2ef0c2d0db3e970b0dddbb933be8ac12","observation_id":"f0529063-da86-4bf5-8efc-b79521ae2bcb","resolution":{"observed_at":"2026-08-07T04:50:36.558221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12015","last_updated":"2025-01-21T10:13:28Z","snapshot_observed_at":"2026-08-14T04:54:37.990391Z","submitted_at":"2025-01-21T10:13:28Z","title":"Full Proportional Justified Representation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12015","snapshot_observed_at":"2026-08-07T04:50:33.898778Z","title":"Full proportional justified representa- tion.arXiv preprint arXiv:2501.12015,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1973,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.898778Z"},"links":{"cited_paper":"/paper/2501.12015","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:6abfc18ad436807b851c9d005a06949b9267c72e29748a906f516e22c6e9b5a7","observation_id":"e038a36f-f85a-4fa4-8f33-cd1fb85fdadf","resolution":{"observed_at":"2026-08-07T04:50:33.898778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01719","last_updated":"2024-05-06T15:09:50Z","snapshot_observed_at":"2026-08-16T13:55:49.076802Z","submitted_at":"2024-05-02T20:28:54Z","title":"Inherent Trade-Offs between Diversity and Stability in Multi-Task Benchmarks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.01719","snapshot_observed_at":"2026-08-07T04:50:34.884507Z","title":"Inherent trade-offs between diversity and stability in multi-task benchmark.arXiv preprint arXiv:2405.01719,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1975,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.884507Z"},"links":{"cited_paper":"/paper/2405.01719","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:afedbfe365bb3fc5f8f72385bc36ed7789dcdd69be7cb4bb99c9815537830e30","observation_id":"dc3a456d-22d5-4e9f-8c31-d09245a1dd4d","resolution":{"observed_at":"2026-08-07T04:50:34.884507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10284","last_updated":"2023-05-17T15:20:31Z","snapshot_observed_at":"2026-08-20T04:53:43.400220Z","submitted_at":"2023-05-17T15:20:31Z","title":"Towards More Robust NLP System Evaluation: Handling Missing Scores in Benchmarks","version":1},"cited_work":{"arxiv_id":"2305.10284","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.10284","snapshot_observed_at":"2026-08-07T04:50:36.132905Z","title":"Towards More Robust NLP System Evaluation: Handling Missing Scores in Benchmarks","venue":"cs.CL","work_id":"4fb01280-ae84-4c8a-ad11-c0f056bb1c60","year":2023},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1979,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.571425Z"},"links":{"cited_paper":"/paper/2305.10284","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:5fed706583cb97b2ec77a64942195dcd97d027c460d6331084cf49c9a8d77f43","observation_id":"c3b10293-3d5f-4ccf-ba46-169950cbb4e1","resolution":{"observed_at":"2026-08-07T04:50:36.186825Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11696","last_updated":"2024-04-01T17:34:34Z","snapshot_observed_at":"2026-08-20T05:38:28.460864Z","submitted_at":"2023-08-22T17:59:30Z","title":"Efficient Benchmarking of Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11696","snapshot_observed_at":"2026-08-07T04:50:34.399951Z","title":"Efficient benchmarking of lan- guage models.arXiv preprint arXiv:2308.11696,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1995,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.399951Z"},"links":{"cited_paper":"/paper/2308.11696","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:7eaa437436ad6e8a533b30fd46400425e84d6996114c3d77c1077d0fafe6f7c7","observation_id":"c8e958b9-687b-4b4d-9dad-93e531d2c948","resolution":{"observed_at":"2026-08-07T04:50:34.399951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:38.330758Z","title":"Proportionally fair clustering revisited","venue":null,"work_id":"f88e80d7-6d35-4e58-93b5-7527d23a028a","year":2020},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2001,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.258254Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:69f98e3fda7b813321c771eec0502fdaa00ee6362d38c91c15d3e8bb0f8305e8","observation_id":"8baec2d6-d329-49a3-943c-d75d13ba6cab","resolution":{"observed_at":"2026-08-07T04:50:38.470912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-18T08:11:45.716032Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-07T04:50:33.759739Z","title":"Swe-bench: Can language models resolve real-world github issues? arXiv preprint arXiv:2310.06770,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2009,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.759739Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:8d151075adf93f2d39fe5d4411f2154eaf4bf05d67f3e39dbf2c480830663b54","observation_id":"6925d52e-4988-4184-b348-c0273ccba638","resolution":{"observed_at":"2026-08-07T04:50:33.759739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-07T04:50:34.762614Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models.arXiv preprint arXiv:2206.04615,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.762614Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:9630523d9034ea25d8fd941de47ebe5b72cd79ac290e6680d855115211047374","observation_id":"6b5578c3-4049-4f60-bfec-3e11530af946","resolution":{"observed_at":"2026-08-07T04:50:34.762614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-17T02:59:38.954198Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-07T04:50:34.511713Z","title":"tinybenchmarks: evaluating llms with fewer examples.arXiv preprint arXiv:2402.14992,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.511713Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:bf4854838b924fd79acdef257f5dec333ad41678bcdfe7ab7bee3fba14ebce48","observation_id":"a856b101-903c-4d8b-b96d-34f9f7d54854","resolution":{"observed_at":"2026-08-07T04:50:34.511713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:38.592258Z","title":"Exact algorithms for set multicover and multiset multicover problems","venue":null,"work_id":"d8b7bd71-7fc7-42b9-9013-1e7cd6c04f18","year":2009},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.655898Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:548dadb7a5dd8d0ae06e8e203af7b760247a2b51d33463519ee9bfbc82757a66","observation_id":"1e67bd58-e739-47c0-9caa-5c6f1e8dd4bd","resolution":{"observed_at":"2026-08-07T04:50:38.685753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-10T23:10:13.900680Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-07T04:50:34.163425Z","title":"Holistic evaluation of language models.arXiv preprint arXiv:2211.09110,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.163425Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:c912d09747c35e6cba51897fc42372fa0eb0645adb3df083f28d3966dcc039f8","observation_id":"45d531a0-0a4f-4002-ba70-99ca07e15346","resolution":{"observed_at":"2026-08-07T04:50:34.163425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05952","last_updated":"2024-10-08T12:08:46Z","snapshot_observed_at":"2026-08-16T13:11:32.301941Z","submitted_at":"2024-10-08T12:08:46Z","title":"Active Evaluation Acquisition for Efficient LLM Benchmarking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05952","snapshot_observed_at":"2026-08-07T04:50:34.063035Z","title":"Active evaluation acquisition for efficient llm benchmarking.arXiv preprint arXiv:2410.05952,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.063035Z"},"links":{"cited_paper":"/paper/2410.05952","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:13e7d1ab0660015c5a8512ee63f29b7c04a4127e12f970b270bbb59d597cc647","observation_id":"5146f728-8393-40fd-9c30-057814cb016a","resolution":{"observed_at":"2026-08-07T04:50:34.063035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T05:35:06.006864Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks"},"reference_resolution":{"displayed":20,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":10},"total_outbound_references":20},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 20 of 20 outbound references and 0 inbound Pith citation observations for arXiv:2506.09813."}