{"as_of":"2026-08-10T12:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:970f8d0cda8dc72912d662970f6b8a4f9c228b40ddaaa3ecf7400c1edb4e4bb8","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:59:16.541167Z","state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.01470/citation-record","integrity":"/paper/2507.01470/integrity","json":"/paper/2507.01470/citation-record.json","paper":"/paper/2507.01470"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1901.08492","last_updated":"2019-01-24T16:44:16Z","snapshot_observed_at":"2026-07-06T07:28:48.758265Z","submitted_at":"2019-01-24T16:44:16Z","title":"Feudal Multi-Agent Hierarchies for Cooperative Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1901.08492","snapshot_observed_at":"2026-08-06T20:59:13.632344Z","title":"Feudal multi-agent hierarchies for cooperative reinforcement learning, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.632344Z"},"links":{"cited_paper":"/paper/1901.08492","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:ac265427aa2fe1b1346079671471c964e169591b44bb64a751a8e1a9cee59a6d","observation_id":"7d4da9db-1c46-43ae-a885-322a86a7f863","resolution":{"observed_at":"2026-08-06T20:59:13.632344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.06565","last_updated":"2016-07-25T17:23:29Z","snapshot_observed_at":"2026-07-06T05:00:46.434335Z","submitted_at":"2016-06-21T13:37:05Z","title":"Concrete Problems in AI Safety","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.06565","snapshot_observed_at":"2026-08-06T20:59:13.697707Z","title":"Concrete Problems in AI Safety , July 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.697707Z"},"links":{"cited_paper":"/paper/1606.06565","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:f3ebf5667531cd380330bbab02d8aef440a52abd5bfd69b9662c6320f831f01e","observation_id":"99c0bd26-131a-419b-b279-4e8d9470bc95","resolution":{"observed_at":"2026-08-06T20:59:13.697707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.514654Z","title":"Hindsight experience replay","venue":null,"work_id":"f0bdaef6-ca0c-4c45-80e3-19a03b151aff","year":2017},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.773390Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:184457c5ad812803f6109d8ac18f996fa25082c05be68408987ed0bf719e16bf","observation_id":"8836cee5-a8d6-4999-a299-a65dc61cdfee","resolution":{"observed_at":"2026-08-06T20:59:19.581118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.395720Z","title":"Dynamic programming","venue":null,"work_id":"17dc7a8f-efad-4bf5-93d5-f6ddea51f80f","year":1957},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.828930Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:20c7dd28cb26f09d024fc8f0baaf10c4676df8a667c36c14fa108198385f5fab","observation_id":"14ad7b67-bf52-4977-83fa-a00466b041a1","resolution":{"observed_at":"2026-08-06T20:59:19.452897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.12894","last_updated":"2018-10-30T17:44:42Z","snapshot_observed_at":"2026-07-06T07:11:32.319931Z","submitted_at":"2018-10-30T17:44:42Z","title":"Exploration by Random Network Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.12894","snapshot_observed_at":"2026-08-06T20:59:13.902535Z","title":"Exploration by Random Network Distillation , October 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.902535Z"},"links":{"cited_paper":"/paper/1810.12894","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:cf615138cba4f24b1fb0bb836d56483f8d009dc6987664e3db60cd513cd9f2d5","observation_id":"e4fb16a7-b28d-410a-9931-4d125ab8aa59","resolution":{"observed_at":"2026-08-06T20:59:13.902535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:13.972079Z","title":null,"venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.972079Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:ec240d07493897eb0e162aaf173f43ff404507f052bf4125d5b276e183fb98b1","observation_id":"bc377857-9684-4efc-b592-66816d3ead45","resolution":{"observed_at":"2026-08-06T20:59:13.972079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2024.12411","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.496695Z","title":"HiSOMA : A hierarchical multi-agent model integrating self-organizing neural networks with multi-agent deep reinforcement learning","venue":null,"work_id":"6a22b63a-6f71-40d0-a0c4-55ee5cb672bc","year":2024},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.067851Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:5af579498ad59afe733df6b630f12f758537eee1f8d6e78f83fc911b36891422","observation_id":"5bfdeb87-e46b-4126-ac23-e2f7d48435d9","resolution":{"observed_at":"2026-08-06T20:59:17.554700Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.235043Z","title":"MASER : Multi-agent reinforcement learning with subgoals generated from experience replay buffer","venue":null,"work_id":"c6fffe30-35d0-4ec3-8293-646e6a25a360","year":2022},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.151775Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:e8f982e556726b17c65cf213b447762324e4fb9ff2cf0bd84d3d5e4efaec4748","observation_id":"dde65c51-d02b-4875-bbc3-adef23066d2a","resolution":{"observed_at":"2026-08-06T20:59:19.333627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.154172Z","title":"Automatic discovery of subgoals in reinforcement learning using strongly connected components","venue":null,"work_id":"c51f28ab-91d6-4763-ba86-39e81eb529f9","year":2009},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.229955Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:8463d2d81bea116b40723e844079415b7d5fab851fc9ce2157523c8e2a7d8add","observation_id":"d311e481-1434-40ba-9ab9-c61e70c41aed","resolution":{"observed_at":"2026-08-06T20:59:19.194182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:14.308450Z","title":"Exploration in deep reinforcement learning: A survey","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.308450Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:0cba49a3a2d934d3a8c53c9689af8aff43e66abb787d7fc318fe378c0281b488","observation_id":"467cdcee-7649-43a1-8f56-49382aafe968","resolution":{"observed_at":"2026-08-06T20:59:14.308450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:14.397566Z","title":"Lecun, L","venue":null,"work_id":null,"year":1998},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.397566Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:1723eaf6b3024e873be50a10d8bd201532e7e8186daab102e9a2c977b33a05c4","observation_id":"5c6a3b94-79d5-418b-9334-34cc91bef9c4","resolution":{"observed_at":"2026-08-06T20:59:14.397566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.039857Z","title":"Automatic discovery of subgoals in reinforcement learning using diverse density","venue":null,"work_id":"e4b218a1-8947-4e8d-93a0-6d530d4a2ccc","year":2001},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.518805Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:bc5fc76ad898263809d04c23591f600fcac2d4f8f8bb70cbd8ffb09e28a8ee85","observation_id":"d9ce9363-1912-4b43-b219-5e02f2885fd2","resolution":{"observed_at":"2026-08-06T20:59:19.082703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.54097/er0mx710","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Research on Multi -agent Sparse Reward Problem","venue":"Highlights in Science Engineering and Technology","work_id":"ef5e1eb5-d041-4b33-8cba-8caa6133e739","year":2024},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.604076Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:c46942ae51e23579ab26624a310e57bd5eb2605d0c211710dd694c603daaecff","observation_id":"5d82be80-cdee-46ff-818b-67cb0c575fd3","resolution":{"observed_at":"2026-08-06T20:59:17.239155Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.918288Z","title":"Laser learning environment: A new environment for coordination-critical multi-agent tasks","venue":null,"work_id":"12179465-535f-4fd4-a2ba-7f6e30804c13","year":2025},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.679103Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:5d4a9a8c71d1b24c4443e2fdcd969ea73e377efe40fc46e91ff6370d21c4169b","observation_id":"5a082f66-e373-4a5f-a84a-68e4806cc08a","resolution":{"observed_at":"2026-08-06T20:59:18.975027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1613/jair.1.14390","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"An overview of environmental features that impact deep reinforcement learning in sparse-reward domains","venue":"Journal of Artificial Intelligence Research","work_id":"c747ad70-7389-42c7-b36e-795f6b6a5a92","year":2023},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.776492Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:b4cec6195d3ae8fdf4b3d9cb46638d790145e566ba0ce57a0e455a7b594a5fa4","observation_id":"2c60f28c-eda0-4176-a4fa-434b66a51ee5","resolution":{"observed_at":"2026-08-06T20:59:17.060987Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:14.850478Z","title":"Efros, and Trevor Darrell","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.850478Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:9a8291aa587762e144e5989c889c617726579282f99e9766f6d7b4e8ab710a5f","observation_id":"ddab0dbb-b9fd-4f0d-800b-2f05df26db66","resolution":{"observed_at":"2026-08-06T20:59:14.850478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.672546Z","title":"Learning to Drive a Bicycle using Reinforcement Learning and Shaping","venue":null,"work_id":"85148589-e67f-4edc-b20e-67a57988eb17","year":1998},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.972744Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:31a15e6c8d8fc031897ed7986c97eedae22d32747478306a9af456a4c489c4e1","observation_id":"8c7f1459-afb7-4eca-8ca3-3382338a08f2","resolution":{"observed_at":"2026-08-06T20:59:18.795445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.11485","last_updated":"2018-06-06T17:58:09Z","snapshot_observed_at":"2026-08-06T12:03:41.509273Z","submitted_at":"2018-03-30T14:23:39Z","title":"QMIX: Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.11485","snapshot_observed_at":"2026-08-06T20:59:15.058602Z","title":"QMIX : Monotonic Value Function Factorisation for Deep Multi - Agent Reinforcement Learning","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.058602Z"},"links":{"cited_paper":"/paper/1803.11485","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:3089c2140469a09424c6a64d0dc3577f01219dd0b3afb698de8d4b7934a1167f","observation_id":"cbeecba4-df41-4b81-83a4-5ee1fa5b4c80","resolution":{"observed_at":"2026-08-06T20:59:15.058602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1902.04043","last_updated":"2019-12-09T07:26:52Z","snapshot_observed_at":"2026-08-10T01:55:31.555267Z","submitted_at":"2019-02-11T18:43:53Z","title":"The StarCraft Multi-Agent Challenge","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1902.04043","snapshot_observed_at":"2026-08-06T20:59:15.116702Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.116702Z"},"links":{"cited_paper":"/paper/1902.04043","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:80b04203c786c0b1f1a765fb3a5c2ae8c02b3ae4e4fff53ecf3988f7f9416f1c","observation_id":"3724e73d-6be2-4af4-8957-7a144c55dbcc","resolution":{"observed_at":"2026-08-06T20:59:15.116702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.438668Z","title":"Normalized cuts and image segmentation","venue":null,"work_id":"6e70bcff-f908-430c-8d3c-3aed37261f2d","year":2000},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.206690Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:76817a623f9c07111943dd24664478eaddc52c03709c1a51f1927446590a3f96","observation_id":"5a62b7a4-3661-41f7-885b-ab7abe8a9ad7","resolution":{"observed_at":"2026-08-06T20:59:18.586251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:15.274544Z","title":"Wolfe, and Andrew G","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.274544Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:d00afeee20ee32e411303a5c95e572aecb7408724b92a05777220d8168637c87","observation_id":"f0da5f00-7721-4c86-8dcd-a8d99e8b560c","resolution":{"observed_at":"2026-08-06T20:59:15.274544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.177805Z","title":"Leibo, Karl Tuyls, and Thore Graepel","venue":null,"work_id":"67ae067f-cf7f-40c5-9d87-0c4524004db9","year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.321206Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:0909eb7dc31a8b51486a1acf74991e1829c61ff5ec94dce5d4a62eeb84cfdc88","observation_id":"082ccee4-2d28-4ccb-ab72-b3b636db67bd","resolution":{"observed_at":"2026-08-06T20:59:18.304632Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3643852","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Faster MIL -based subgoal identification for reinforcement learning by tuning fewer hyperparameters","venue":"ACM Transactions on Autonomous and Adaptive Systems","work_id":"07f52d78-6d5d-4e2b-8e21-2697cfd817a8","year":2024},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.405171Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:5c6c6d47a301403d0d50fec248e2e42a8edc7a89b55b8310bec352d5910d7e10","observation_id":"6775740b-eb04-4378-a2ee-978b43fb2dca","resolution":{"observed_at":"2026-08-06T20:59:16.854696Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.995952Z","title":"Sutton and Andrew G","venue":null,"work_id":"aaf2cd12-0ed0-4071-984b-e07986803b97","year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.448124Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:6218b04f1b8489610bf72bc236c159e34b8d03bfadc01399920d5e381ecb8af9","observation_id":"25b4aadd-313e-4b70-a586-c290970713a2","resolution":{"observed_at":"2026-08-06T20:59:18.088589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:15.511885Z","title":"Sutton, Doina Precup, and Satinder Singh","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.511885Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:cc2d3089f98ba0db36987a0f906e5d36cd3eaea85971e5936b4a6455f69e694c","observation_id":"308c9bdf-4d87-4580-a124-1fbafbe0520f","resolution":{"observed_at":"2026-08-06T20:59:15.511885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.908255Z","title":"\\#exploration: A study of count-based exploration for deep reinforcement learning","venue":null,"work_id":"3928da3e-5842-4b98-b7ac-d8c4f6edd087","year":2017},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.649124Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:3bc1df70af5168533bae8a2f90fa099aab9924848cafc2400b4c84f1d74f3726","observation_id":"2c91e146-c36f-40b5-a796-6d9299277dbe","resolution":{"observed_at":"2026-08-06T20:59:17.954977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.795026Z","title":"Keeping your distance: Solving sparse reward tasks using self-balancing shaped rewards","venue":null,"work_id":"d961fbfb-ab76-46e1-9d92-1fb12e7c3610","year":2019},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.787024Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:e8052f22bfb34f6a92b6a5b9cf5819917e2feb0ac89486b8145431c42b745964","observation_id":"8a892e58-5ae4-4b7c-bf72-9a0072e396e2","resolution":{"observed_at":"2026-08-06T20:59:17.854308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1509.06461","last_updated":"2015-12-08T21:19:16Z","snapshot_observed_at":"2026-08-07T14:09:19.448496Z","submitted_at":"2015-09-22T04:40:22Z","title":"Deep Reinforcement Learning with Double Q-learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1509.06461","snapshot_observed_at":"2026-08-06T20:59:15.925516Z","title":"Deep reinforcement learning with double Q - Learning","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.925516Z"},"links":{"cited_paper":"/paper/1509.06461","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:6bb5756a1f857745880cd370b139cf7493dd8da6b692b43857f9aab28e5a1c19","observation_id":"26d0f6c0-c6e6-46bb-b059-8b1e0516bb2b","resolution":{"observed_at":"2026-08-06T20:59:15.925516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2008.01062","last_updated":"2021-10-04T01:36:59Z","snapshot_observed_at":"2026-07-06T09:44:11.377445Z","submitted_at":"2020-08-03T17:52:09Z","title":"QPLEX: Duplex Dueling Multi-Agent Q-Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2008.01062","snapshot_observed_at":"2026-08-06T20:59:16.045425Z","title":"QPLEX : Duplex Dueling Multi - Agent Q - Learning , October 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.045425Z"},"links":{"cited_paper":"/paper/2008.01062","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:4e94cbe472fc0cea89baafaee1d1960084ed1e756581fc82e630cd8e146b41b5","observation_id":"4e0992bd-c580-4fd7-870d-383f6305ef15","resolution":{"observed_at":"2026-08-06T20:59:16.045425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.691744Z","title":null,"venue":null,"work_id":"0977ecb7-f1e9-4155-b6bf-0d24354219b0","year":2009},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.145414Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:367eb2959735bb919d9165b34d439f6fbdeb30d15395e736eb0e525af1afb019","observation_id":"2df939bb-a422-41e5-a98d-c16d404ec35e","resolution":{"observed_at":"2026-08-06T20:59:17.742184Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1609/aaai.v37i10.26386","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"HAVEN : Hierarchical cooperative multi-agent reinforcement learning with dual coordination mechanism","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","work_id":"2bff427b-47e1-46c7-a29f-2fe5cb43eb53","year":2023},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.303192Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:9d9fe45cfa49752358e2c718af6456cdc4efe7e339b3ca4e73e182e78bc3f232","observation_id":"5f4be7ec-1b16-4267-b6c5-7c5eba4fd65b","resolution":{"observed_at":"2026-08-06T20:59:16.693495Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.643317Z","title":"Ng, Daishi Harada, and Stuart Russell","venue":null,"work_id":"d7f54022-8bdb-44dd-be86-bf07b1daff46","year":1999},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.392828Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:99ecb932af7638d93be7bda452e2f7f3506a949b9e0e4550b891ba1d499bccf9","observation_id":"23accf97-5d55-4607-9472-1b9d15576a60","resolution":{"observed_at":"2026-08-06T20:59:17.668579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:16.541167Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.541167Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:84f83bdd8efebe2f97b665aed825cf71161a1497bfc7405daca25a9b9bb18250","observation_id":"ff4fd410-ad6b-4afd-aa4c-6643d5dac742","resolution":{"observed_at":"2026-08-06T20:59:16.541167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":15,"verified_exact":4,"verified_fuzzy":13},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 0 inbound Pith citation observations for arXiv:2507.01470."}