{"as_of":"2026-08-10T08:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:63870b2ab2a7a8288fb0587b4bd03e302e0d7ca75bd258a21aa669ea1311a3e3","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:06:32.504261Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.21274/citation-record","integrity":"/paper/2507.21274/integrity","json":"/paper/2507.21274/citation-record.json","paper":"/paper/2507.21274"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.270530Z","title":"Llama 3 model card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.270530Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:0dd5176c76eb81505cd3f67a91a4af664d4ac324f2d0b9068240eb4e76c64b71","observation_id":"9527e20e-14bb-4f47-890d-93febf486a30","resolution":{"observed_at":"2026-08-06T13:06:32.270530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.283632Z","title":"The claude 3 model family: Opus, sonnet, haiku","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.283632Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:6e99119b0663202499abe567ac73f9b6751f0e8909f0b3bff9c31e990abc0946","observation_id":"31fce3d4-00b5-477b-a939-fa271ee711c5","resolution":{"observed_at":"2026-08-06T13:06:32.283632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.219613Z","title":"Tallrec: An effective and efficient tuning framework to align large language model with recommendation","venue":null,"work_id":"88f2a52f-195a-466f-a419-e9f7f3d72d08","year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.289399Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:c371894f0393f81f1c4ba8a3768a8e3e95be0ff42ad75cfdfe93347ac83b4992","observation_id":"6cf12dc9-72f4-42f0-b329-8c7c7fafc310","resolution":{"observed_at":"2026-08-06T13:06:33.224678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.200504Z","title":"Adversarial model for offline reinforcement learning","venue":null,"work_id":"a7f27caf-510d-4f58-98d2-e0f5142818ba","year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.298008Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:69457959869579a3d166ef7b8988c81246cfcc8546546f5a61b30018eb30083c","observation_id":"86ea9218-6a7d-4a89-b1ee-fe433c9d6f35","resolution":{"observed_at":"2026-08-06T13:06:33.206275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.183862Z","title":"Stochastic approximation with two time scales","venue":null,"work_id":"362259f4-66ad-412b-bc7d-27d685cfc634","year":1997},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.304396Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:8dc8a9c0ab1066b9c197eb36a360e9ff2525de3162aad10f8ae2fa18402b52d3","observation_id":"c39dd834-a376-4763-9a45-3242aab78d80","resolution":{"observed_at":"2026-08-06T13:06:33.188871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16376","last_updated":"2023-07-31T02:48:56Z","snapshot_observed_at":"2026-07-06T16:00:29.832559Z","submitted_at":"2023-07-31T02:48:56Z","title":"When Large Language Models Meet Personalization: Perspectives of Challenges and Opportunities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16376","snapshot_observed_at":"2026-08-06T13:06:32.314209Z","title":"When large language models meet personalization: Perspectives of challenges and opportunities.arXiv preprint arXiv:2307.16376, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.314209Z"},"links":{"cited_paper":"/paper/2307.16376","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:2d8345c47670e360e78188aaac13764f0ae124b55d91c2314b5381ca882463ed","observation_id":"a7b8571b-43b6-4aff-b1c8-3a6987727fee","resolution":{"observed_at":"2026-08-06T13:06:32.314209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.165939Z","title":"Adversarially trained actor critic for offline reinforcement learning","venue":null,"work_id":"926c4295-2101-41ca-8243-365f9df6901a","year":2022},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.321400Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:05d781757206e1d80d121b6d1e20830b58619206a2dae69274468561e1030bd9","observation_id":"2fe45c2c-ab35-4408-b387-13fb4afe38b6","resolution":{"observed_at":"2026-08-06T13:06:33.171035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1409.1259","last_updated":"2014-10-07T18:08:30Z","snapshot_observed_at":"2026-07-06T03:53:24.366023Z","submitted_at":"2014-09-03T21:03:41Z","title":"On the Properties of Neural Machine Translation: Encoder-Decoder Approaches","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.1259","snapshot_observed_at":"2026-08-06T13:06:32.326866Z","title":"On the properties of neural machine translation: Encoder-decoder approaches","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.326866Z"},"links":{"cited_paper":"/paper/1409.1259","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:a1270ecdcc279b2fe0227339cca22c9e1132bc2740eefb32171c9f8689cd131b","observation_id":"235ee978-ba42-4412-b5bf-4161837b600c","resolution":{"observed_at":"2026-08-06T13:06:32.326866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.151079Z","title":"A review of modern recommender systems using generative models (gen-recsys)","venue":null,"work_id":"aa380e47-e993-4ce5-82d2-8ffe4f1fd788","year":2024},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.335538Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:4241c86a04fd373cfd360ed2d4894110fb3ffda0f25e6460d2c0e5e330ed348d","observation_id":"631a51b8-7791-4f5e-b835-bf11c20585b5","resolution":{"observed_at":"2026-08-06T13:06:33.155646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-06T13:06:32.341879Z","title":"Bert: Pre- training of deep bidirectional transformers for language understanding","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.341879Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:f5eb91ffaff8ba028052284f8576998c73dc8cd13e9188b8fbd3af71aaf53c91","observation_id":"ead6fcdf-e467-46ea-ae23-9cd1aded9f74","resolution":{"observed_at":"2026-08-06T13:06:32.341879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02046","last_updated":"2024-04-29T09:06:51Z","snapshot_observed_at":"2026-08-07T00:54:27.770577Z","submitted_at":"2023-07-05T06:03:40Z","title":"Recommender Systems in the Era of Large Language Models (LLMs)","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.02046","snapshot_observed_at":"2026-08-06T13:06:32.347874Z","title":"Recommender systems in the era of large language models (llms)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.347874Z"},"links":{"cited_paper":"/paper/2307.02046","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:4a8645fa40094a205cbd4cc27a3811ed3883a51c8842b64e2ae98701432f4d2d","observation_id":"9efe23b4-678d-427b-b898-80ecc37f7feb","resolution":{"observed_at":"2026-08-06T13:06:32.347874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.135263Z","title":"Addressing function approxi- mation error in actor-critic methods","venue":null,"work_id":"b5bf5c5f-c084-400e-a9ca-4b8a4a741fe0","year":2018},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.363650Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:3c4b0f439005dad57515b6bcaaf91a45b5192f7b7a8b9e44a1cc946564673cf2","observation_id":"d4bc672e-c5d1-43b7-bafe-1f786eff552b","resolution":{"observed_at":"2026-08-06T13:06:33.140090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.117178Z","title":"Soft actor- critic: Off-policy maximum entropy deep reinforcement learning with a stochas- tic actor","venue":null,"work_id":"e17b0a2b-a906-4834-942f-04e3f27b613c","year":2018},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.369523Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:888534ffdbbb7d370c869ed0adaaedc0db1b372b13843b49e2f1b32fdd542240","observation_id":"8510f530-5fd6-422b-8d3e-cf9fd20045bd","resolution":{"observed_at":"2026-08-06T13:06:33.122823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.101140Z","title":"Maxwell Harper and Joseph A","venue":null,"work_id":"e7a9270d-3749-462b-a78d-1d076d2eae1b","year":2015},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.376998Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:b218131f2ba1b2b473cb4492dfa7487e1c86b44853a92a294f694ee487411f2f","observation_id":"ab507f5d-99fd-42ba-992a-c16053591d26","resolution":{"observed_at":"2026-08-06T13:06:33.106299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.083217Z","title":"Large lan- guage models as zero-shot conversational recommenders","venue":null,"work_id":"08689a57-1530-4e09-a732-4a5b419be5df","year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.381848Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:780e95fb9d1c21aae09588211a140b79de21e1d1b24d0271f5a46caf5022ccc7","observation_id":"c7d85c18-6e52-439e-b37e-2632fd564f0a","resolution":{"observed_at":"2026-08-06T13:06:33.088042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.06939","last_updated":"2016-03-29T14:52:58Z","snapshot_observed_at":"2026-07-31T19:00:00.136727Z","submitted_at":"2015-11-21T23:42:59Z","title":"Session-based Recommendations with Recurrent Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1511.06939","snapshot_observed_at":"2026-08-06T13:06:32.387219Z","title":"Session-based recommendations with recurrent neural networks","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.387219Z"},"links":{"cited_paper":"/paper/1511.06939","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:2dec99e3ea35408d139f14bdca6e0529cbcea7430217109d5b162cf7d4120174","observation_id":"76edea6e-06dc-44e9-8ad0-8bf3260464d8","resolution":{"observed_at":"2026-08-06T13:06:32.387219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.064415Z","title":"Towards universal sequence representation learning for recommender systems","venue":null,"work_id":"c4054c90-4ad7-4205-92bb-80aa26b01ef4","year":2022},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.394208Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:cedfdd2f45b11328a6b731d59b14793bfa9099b4467c5875d35689ef03cfc850","observation_id":"44a34131-597b-490d-a7c3-db8d9ab19b74","resolution":{"observed_at":"2026-08-06T13:06:33.070053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-07T07:43:16.294957Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-06T13:06:32.399332Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.399332Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:60e79c46e9c5de0b942764e498098f4d1f20181aba237085649e8ec1d5a793fc","observation_id":"2b46a6e7-e9f2-4988-968a-1200f7ca45dc","resolution":{"observed_at":"2026-08-06T13:06:32.399332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.044263Z","title":"Human-centric dialog training via offline reinforcement learning","venue":null,"work_id":"5f51fba0-f24b-42c5-b670-dd16e628d02f","year":2020},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.405283Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:c20be0d90de9c38e2149acbb48c6577e1ff6beffc4a1993d2230af07b64d83a7","observation_id":"b2bf7143-86ce-49bc-ae97-f31aca2c4e00","resolution":{"observed_at":"2026-08-06T13:06:33.049119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.028415Z","title":"Cumulated gain-based evaluation of ir techniques","venue":null,"work_id":"da8d05cf-a16e-4560-b4ff-68b6daedc87d","year":2002},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.413787Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:591641696a50f865903039b77b6c362e7ddc71c7ef6a880adc86629b44bbc2e1","observation_id":"b2791e8f-824f-4d4b-9c64-2545aa38da62","resolution":{"observed_at":"2026-08-06T13:06:33.033601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:33.010856Z","title":"Kingma and Jimmy Ba","venue":null,"work_id":"15a7d0dc-341f-4eb2-b9d8-e120b99777a7","year":2015},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.420117Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:e24c45d8aaf3e3f159440b9bd24704705a8dd1841c41fdbe092b98db4fb1f31b","observation_id":"40441afe-eff5-4cbe-9117-4ad73561fa0f","resolution":{"observed_at":"2026-08-06T13:06:33.016934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05817","last_updated":"2024-07-09T13:17:52Z","snapshot_observed_at":"2026-08-05T03:33:18.005396Z","submitted_at":"2023-06-09T11:31:50Z","title":"How Can Recommender Systems Benefit from Large Language Models: A Survey","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05817","snapshot_observed_at":"2026-08-06T13:06:32.425492Z","title":"How can recommender systems benefit from large language models: A survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.425492Z"},"links":{"cited_paper":"/paper/2306.05817","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:0c6f9d542f1aab7a1e131f6c0bcf9979529cc80fed2739391da1556cdb0fde07","observation_id":"d9bb156e-7077-4a25-b77e-4046b99350b9","resolution":{"observed_at":"2026-08-06T13:06:32.425492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.992577Z","title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing","venue":null,"work_id":"84763ac6-4f7c-448d-90d8-dc2300e0f29e","year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.430643Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:ccd2d2a02a2d4921eda21f6820fe17ea8de530c89b69bb4c7cb6dbfedf9bba95","observation_id":"0aa9d40c-b56d-47cb-b921-156523c8a3a5","resolution":{"observed_at":"2026-08-06T13:06:32.997853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.975574Z","title":"Diversity-promoting deep reinforcement learning for interactive recommendation","venue":null,"work_id":"d895bfb0-5167-4a74-b49a-f7d35468f619","year":2022},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.436766Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:e5f7c72370c20676ff809dd4042476e5e755719adb37fa15b8a25570b17dff04","observation_id":"3c55693a-53a4-4bff-a68f-1f4d78cc7583","resolution":{"observed_at":"2026-08-06T13:06:32.980680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.957170Z","title":"Convergent temporal-difference learning with arbitrary smooth function approximation","venue":null,"work_id":"c09a5f68-1908-46dd-8a1f-632042a75576","year":2009},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.442735Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:154705796f8eb3852932c00feae6da19b5fd9f78b619769534994eb1bc6a28e9","observation_id":"7a19ab21-727d-43e0-912d-dc2aefe06833","resolution":{"observed_at":"2026-08-06T13:06:32.962113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.939041Z","title":"Recent advances in natural language processing via large pre-trained language models: A survey","venue":null,"work_id":"974c2fe3-c16d-4e87-98ae-15c32aae6b27","year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.447719Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:fada00daf2d604ff21b15e162fcc0990dfc646b576a2ea02d4b47efd7d5e42e4","observation_id":"557e0e8c-9c13-45e2-bb4d-54bb07b0ebd7","resolution":{"observed_at":"2026-08-06T13:06:32.944386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.921024Z","title":"Training language models to follow instructions with human feedback.Advances in Neural Information Processing Systems , 35:27730–27744, 2022","venue":null,"work_id":"6bf1e3ee-af48-4f98-96ca-c575cdfbd890","year":2022},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.453217Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:bc0a7f8711f193bfb699256b280f2c430bb28a10b31197cb5e209853ca6e52d4","observation_id":"1254bda8-8018-46c8-a2a2-f9a3e33fc42c","resolution":{"observed_at":"2026-08-06T13:06:32.926015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-06T13:06:32.458374Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.458374Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:6cd526b061a7dd028184d170f727d609778af4d3b1f1132c13f09cbf7e44e8f3","observation_id":"c5381e4b-f6c2-4ae3-bc1a-c2fbe106b12e","resolution":{"observed_at":"2026-08-06T13:06:32.458374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.903816Z","title":"Choosing the best of both worlds: Diverse and novel recom- mendations through multi-objective reinforcement learning","venue":null,"work_id":"a1e2f62b-aa68-40ff-abc1-1ae5c31b358b","year":2022},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.463913Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:5685c36904fdc1f026910187d6c14375ebd854dcb855b05b2d4454d7d87d055a","observation_id":"257ccca1-7c6c-4e00-bdda-02925ad0f94e","resolution":{"observed_at":"2026-08-06T13:06:32.909052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.886743Z","title":"Learning to summarize with human feedback","venue":null,"work_id":"63080c93-baf1-43dd-8ed3-3cd502e91077","year":2020},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.468816Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:86f9bacc6c9f1335e3c043b025789efb37df621c513ad2a53eee0e59b69e0589","observation_id":"ba59590b-e618-4bec-baca-a6ae722e81ac","resolution":{"observed_at":"2026-08-06T13:06:32.891898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-06T13:06:32.473790Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.473790Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:231a33710796bcd2780861292032467676a5afcecc794c92ca1431a7106d872e","observation_id":"1cfc025b-b6c0-4950-a669-f6b8aeeb1eac","resolution":{"observed_at":"2026-08-06T13:06:32.473790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.869356Z","title":null,"venue":null,"work_id":"083e2a2e-a830-4f9c-9505-230991abe8a8","year":2024},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.479004Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:7c9aa5f9c3254110e284945e2dfe1d8c911cd0e23cac1aab5082804e2066d439","observation_id":"f60b4162-d92c-456d-a661-93715900ebbe","resolution":{"observed_at":"2026-08-06T13:06:32.874037Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2206.06190","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.674389Z","title":"Transrec: Learning transferable recommendation from mixture-of-modality feedback","venue":null,"work_id":"3abff8b4-8bf9-4ca2-8d55-eee78ce54678","year":2022},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.486609Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:1a0abebebde0cfaa250b6fe0fd8440ea8d9b2837b229a1a10a1bc3fe422035ef","observation_id":"b5e0e357-3cea-4b17-9434-9ac56afaef47","resolution":{"observed_at":"2026-08-06T13:06:32.684421Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1911.11361","last_updated":"2019-11-26T06:11:34Z","snapshot_observed_at":"2026-07-06T08:39:58.361914Z","submitted_at":"2019-11-26T06:11:34Z","title":"Behavior Regularized Offline Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.11361","snapshot_observed_at":"2026-08-06T13:06:32.492596Z","title":"Behavior regularized offline rein- forcement learning","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.492596Z"},"links":{"cited_paper":"/paper/1911.11361","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:706c912560f29bd66cb99112aa0973ef1516d13fc7413030c0cbb981bb7fdedf","observation_id":"d51c539b-01bf-43af-8f05-39fdd9744ee8","resolution":{"observed_at":"2026-08-06T13:06:32.492596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-06T13:06:32.497705Z","title":"A survey of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.497705Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:22feb6b17e13764a89155f1aa52bb77d55cb4e14735c05ca63f4d9b35eb9215c","observation_id":"4095c89b-7e48-4600-bee7-e7563e198836","resolution":{"observed_at":"2026-08-06T13:06:32.497705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:06:32.852647Z","title":"Drn: A deep reinforcement learning framework for news recommendation","venue":null,"work_id":"e6f2dcfc-e3e4-4f60-b4c2-47df14e425f5","year":2018},"citing_paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T13:06:32.504261Z"},"links":{"citing_paper":"/paper/2507.21274"},"observation_digest":"sha256:28890341eeb86b2478c62322bea10c57453021cc92d94108846a5690b76700db","observation_id":"3bd8393f-bddc-4a8e-b59c-d6c18f435ca0","resolution":{"observed_at":"2026-08-06T13:06:32.857292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21274","last_updated":"2025-07-28T19:00:40Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T01:21:22.249248Z","submitted_at":"2025-07-28T19:00:40Z","title":"Large Language Model-Enhanced Reinforcement Learning for Diverse and Novel Recommendations"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":1,"verified_fuzzy":21},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 0 inbound Pith citation observations for arXiv:2507.21274."}