{"as_of":"2026-08-18T22:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4d1c2d892cd885f597b25abef6ba0c712313ba0205058e5256ca5025df1d825b","coverage":[{"denominator":13,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:23:32.774387Z","state":"measured"},{"denominator":16,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":16,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T23:01:35.774091Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T07:04:44.553235Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"cited_work":{"arxiv_id":"2507.11059","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.11059","snapshot_observed_at":"2026-07-08T07:04:44.553235Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","venue":"cs.SE","work_id":"386604a4-ae4c-41f8-866b-ed3d0bc6ec4a","year":2025},"citing_paper":{"arxiv_id":"2607.06411","last_updated":"2026-07-19T18:49:59Z","snapshot_observed_at":"2026-08-17T12:37:53.897561Z","submitted_at":"2026-07-07T15:41:22Z","title":"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-08T06:55:40.830922Z"},"links":{"cited_paper":"/paper/2507.11059","citing_paper":"/paper/2607.06411"},"observation_digest":"sha256:ea7ef381851508c69f00f9ad7639c97a8c6aaaeaae7f27c088821aedd9cce78b","observation_id":"361e4e2a-fec5-4709-b57d-e0f611e9d605","resolution":{"observed_at":"2026-07-08T07:04:44.554643Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.11059","snapshot_observed_at":"2026-08-02T08:21:50.116198Z","title":"Adamenko, M","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.06411","last_updated":"2026-07-19T18:49:59Z","snapshot_observed_at":"2026-08-17T12:37:53.897561Z","submitted_at":"2026-07-07T15:41:22Z","title":"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T08:21:50.116198Z"},"links":{"cited_paper":"/paper/2507.11059","citing_paper":"/paper/2607.06411"},"observation_digest":"sha256:9d133e461f4bde5d160060193702d85a77d9a5fc4e3fbe76875239507d92fe0a","observation_id":"ca14c0ea-849c-4e29-b863-c20cd752f562","resolution":{"observed_at":"2026-08-02T08:21:50.116198Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.11059","snapshot_observed_at":"2026-08-10T23:01:35.774091Z","title":"Zadorozhny, Ivan Lopatin, Dmitry Babayev, Alena Fenogenova, and Valentin Malykh","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.06663","last_updated":"2026-08-07T00:19:48Z","snapshot_observed_at":"2026-08-14T07:59:30.675858Z","submitted_at":"2026-08-07T00:19:48Z","title":"The Horizon Gap: Planning, Memory, Execution, Training, and Evaluation for Long-Horizon LLM Agents","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-10T23:01:35.774091Z"},"links":{"cited_paper":"/paper/2507.11059","citing_paper":"/paper/2608.06663"},"observation_digest":"sha256:25c35f0008efb5e3be1f4ffe755aeb536e9a593bb1c004c56e1d386e7b310d55","observation_id":"1ace0f74-a7fd-4d09-97ad-8c1d4715aed4","resolution":{"observed_at":"2026-08-10T23:01:35.774091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.11059/citation-record","integrity":"/paper/2507.11059/integrity","json":"/paper/2507.11059/citation-record.json","paper":"/paper/2507.11059"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:23:31.554040Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:31.554040Z"},"links":{"citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:ea940b7fa473c70c0af75f15da93235e2837907d8171818f7a485e7dd480ac34","observation_id":"522319be-0f16-4023-a4d8-71ab6f5f9fd5","resolution":{"observed_at":"2026-08-06T17:23:31.554040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:23:31.638050Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:31.638050Z"},"links":{"citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:0927396c70779b5a58cec4c0927de00a369281b4838a023bf40b0a4673fe49ed","observation_id":"ab8db74e-8677-4fcd-aa93-0816f2d8ef0f","resolution":{"observed_at":"2026-08-06T17:23:31.638050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06992","last_updated":"2024-10-10T13:13:09Z","snapshot_observed_at":"2026-08-16T13:11:03.237498Z","submitted_at":"2024-10-09T15:38:53Z","title":"SWE-Bench+: Enhanced Coding Benchmark for LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06992","snapshot_observed_at":"2026-08-06T17:23:31.742334Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:31.742334Z"},"links":{"cited_paper":"/paper/2410.06992","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:20816bda7ac932c3ed8f252a5243f1684916288f7e4d24c6cffdd2f09e0fee9b","observation_id":"d597f088-f028-4633-961e-c630e8acdb65","resolution":{"observed_at":"2026-08-06T17:23:31.742334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T17:23:31.863826Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:31.863826Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:95821d0094d3a49a893b77cfabd50fcaa9bd005850c35a99691c6021d6b5f3ea","observation_id":"05a5ca85-3c73-4f89-91e0-8424e04323e2","resolution":{"observed_at":"2026-08-06T17:23:31.863826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-06T17:23:31.983323Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:31.983323Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:b074d9ded83a3baae8dad64f174536f01c43309c456becfafbd838c80e3a2197","observation_id":"88c76691-db89-4dd7-8a6c-9620c427d861","resolution":{"observed_at":"2026-08-06T17:23:31.983323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-08-16T07:05:57.323612Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07974","snapshot_observed_at":"2026-08-06T17:23:32.089123Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.089123Z"},"links":{"cited_paper":"/paper/2403.07974","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:ea04b112573f1d677ede365f746055fe5827e68cf6b8ec72e0ac03b5080adbe3","observation_id":"b12e7bcb-9458-4b80-b0d9-2145d51f405d","resolution":{"observed_at":"2026-08-06T17:23:32.089123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-18T08:11:45.716032Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-06T17:23:32.208207Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.208207Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:2fb316d33403733ec3a675af39c367a0e989aa0e0423f23697a602e5f8a929f3","observation_id":"2a49d633-9b25-4c0f-827a-bdc085364b3b","resolution":{"observed_at":"2026-08-06T17:23:32.208207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.21139","last_updated":"2025-06-06T07:53:20Z","snapshot_observed_at":"2026-08-16T19:05:11.273764Z","submitted_at":"2024-12-30T18:15:39Z","title":"Training Software Engineering Agents and Verifiers with SWE-Gym","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.21139","snapshot_observed_at":"2026-08-06T17:23:32.338078Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.338078Z"},"links":{"cited_paper":"/paper/2412.21139","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:d97eff0215a02a101c3305bb42b048c3a4ae81870002dc14dfccdc7f63fd27e4","observation_id":"d40b88ce-95d8-41bb-b4e0-dba096edcfa4","resolution":{"observed_at":"2026-08-06T17:23:32.338078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:23:32.463424Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.463424Z"},"links":{"citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:2f81aa3830f167f04324e84516ec673c2243d238c71583a4b4f96b59fc497aec","observation_id":"385ade8c-74be-42a5-93d6-d511c1c86999","resolution":{"observed_at":"2026-08-06T17:23:32.463424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T17:23:32.561248Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.561248Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:697cfac1638b143b4e9fa81340becc88855037e03ee2599f23c3b11899dfbcc1","observation_id":"eaa81217-c354-40cf-bc03-7d6e0cc5a7bc","resolution":{"observed_at":"2026-08-06T17:23:32.561248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T17:23:32.632089Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.632089Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:2aefaeaed0d08c2123c96ad86bd2df719a1919015275ea4873cd9e42d8ddcc6e","observation_id":"b50e8b67-0ba7-4a18-b4b6-31a3cd5258ab","resolution":{"observed_at":"2026-08-06T17:23:32.632089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.21798","last_updated":"2025-05-21T17:21:45Z","snapshot_observed_at":"2026-08-11T02:21:44.551484Z","submitted_at":"2025-04-30T16:56:06Z","title":"SWE-smith: Scaling Data for Software Engineering Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.21798","snapshot_observed_at":"2026-08-06T17:23:32.695270Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.695270Z"},"links":{"cited_paper":"/paper/2504.21798","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:979438926a57794cbc3c268185d36540dbe1733ffa0de5f4c033183ce8e019ec","observation_id":"ba76c1e7-a3b9-4e89-b264-d8b8dfc199b5","resolution":{"observed_at":"2026-08-06T17:23:32.695270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02605","last_updated":"2025-04-03T14:06:17Z","snapshot_observed_at":"2026-08-09T23:51:45.534260Z","submitted_at":"2025-04-03T14:06:17Z","title":"Multi-SWE-bench: A Multilingual Benchmark for Issue Resolving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02605","snapshot_observed_at":"2026-08-06T17:23:32.774387Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T17:23:32.774387Z"},"links":{"cited_paper":"/paper/2504.02605","citing_paper":"/paper/2507.11059"},"observation_digest":"sha256:ae84ba73c9a6ab004fcbae6605544676670047f6971084ee06506f48bb603654","observation_id":"47d99099-fcc2-423c-96d7-957e3e954dc6","resolution":{"observed_at":"2026-08-06T17:23:32.774387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.11059","last_updated":"2026-07-11T11:56:31Z","latest_version":3,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-16T05:42:41.138478Z","submitted_at":"2025-07-15T07:52:33Z","title":"SWE-MERA: A Dynamic Benchmark for Agenticly Evaluating Large Language Models on Software Engineering Tasks"},"reference_resolution":{"displayed":13,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":13},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 13 of 13 outbound references and 3 inbound Pith citation observations for arXiv:2507.11059."}