{"as_of":"2026-08-11T10:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:acff9af90edae8be1085c4b683acc941f6db37db511e64344045ad5d6029c5f0","coverage":[{"denominator":17,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:14:16.727994Z","state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.21131/citation-record","integrity":"/paper/2507.21131/integrity","json":"/paper/2507.21131/citation-record.json","paper":"/paper/2507.21131"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1606.06565","last_updated":"2016-07-25T17:23:29Z","snapshot_observed_at":"2026-07-06T05:00:46.434335Z","submitted_at":"2016-06-21T13:37:05Z","title":"Concrete Problems in AI Safety","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.06565","snapshot_observed_at":"2026-08-06T15:14:16.642328Z","title":"Concrete problems in ai safety","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.642328Z"},"links":{"cited_paper":"/paper/1606.06565","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:a9209ea56161ce8dd0f1b08dd20f5c43c05551d8f71dd748621774741a1627e5","observation_id":"98fea892-9983-4f3c-bf58-55ae314d04c5","resolution":{"observed_at":"2026-08-06T15:14:16.642328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1805.00899","last_updated":"2018-10-22T17:36:07Z","snapshot_observed_at":"2026-08-02T15:33:17.783178Z","submitted_at":"2018-05-02T16:27:32Z","title":"AI safety via debate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.00899","snapshot_observed_at":"2026-08-06T15:14:16.663769Z","title":"Ai safety via debate.arXiv preprint arXiv:1805.00899,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.663769Z"},"links":{"cited_paper":"/paper/1805.00899","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:382c81c52657cd872afec616b15852e6d49d120e63c78358ff5332e3ae41d866","observation_id":"e5adc283-597b-4b00-82c4-e31932f82b47","resolution":{"observed_at":"2026-08-06T15:14:16.663769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.05221","last_updated":"2022-11-21T16:38:35Z","snapshot_observed_at":"2026-08-06T08:34:11.887259Z","submitted_at":"2022-07-11T22:59:39Z","title":"Language Models (Mostly) Know What They Know","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.05221","snapshot_observed_at":"2026-08-06T15:14:16.669363Z","title":"Language models struggle to generalize alignment from training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.669363Z"},"links":{"cited_paper":"/paper/2207.05221","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:e469ac951b5fbe878f83456f3cde998c9e18f28fa5af46d23a6d7f5685aedd4c","observation_id":"211cf569-56ac-4ca5-b6fb-5609d1d9d85e","resolution":{"observed_at":"2026-08-06T15:14:16.669363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.066532Z","title":"Algorithms for inverse reinforcement learn- ing","venue":null,"work_id":"a0f1304b-df4d-482a-85ec-32c2d975fc2b","year":2000},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.684460Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:21338c237d597a949ccda93954b9d22b81c67da5bab7624b2169ca0f4c1bdcbc","observation_id":"19245a6b-857e-4a22-b0fa-cae434bd318c","resolution":{"observed_at":"2026-08-06T15:14:17.071285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18290","last_updated":"2024-07-29T22:26:36Z","snapshot_observed_at":"2026-08-01T16:34:38.795326Z","submitted_at":"2023-05-29T17:57:46Z","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18290","snapshot_observed_at":"2026-08-06T15:14:16.693748Z","title":"Direct preference opti- mization: Your language model is secretly a reward model.arXiv preprint arXiv:2305.18290,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.693748Z"},"links":{"cited_paper":"/paper/2305.18290","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:52345d9df71e185684f91f4144cbeeaf4e10d485b86b8e8c9674e970d08c0d48","observation_id":"436d7b01-70a3-4706-abef-90a5af555369","resolution":{"observed_at":"2026-08-06T15:14:16.693748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1812.03030","last_updated":"2019-08-27T04:28:47Z","snapshot_observed_at":"2026-08-10T17:55:30.759009Z","submitted_at":"2018-11-30T08:28:38Z","title":"A new system-wide diversity measure for recommendations with efficient algorithms","version":2},"cited_work":{"arxiv_id":"1812.03030","doi":null,"metadata_source":"pith","pith_arxiv_id":"1812.03030","snapshot_observed_at":"2026-08-06T15:14:16.823705Z","title":"A new system-wide diversity measure for recommendations with efficient algorithms","venue":"cs.IR","work_id":"f8bc13d6-80f7-420e-9bf9-f0f67fed91ae","year":2018},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.703719Z"},"links":{"cited_paper":"/paper/1812.03030","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:9514a52754a0bfbfe17a2f35fbff2c10ee124593dcf7d1bf3870476867c6bc19","observation_id":"06b5a3b5-ecfd-448d-91dd-f902350515fc","resolution":{"observed_at":"2026-08-06T15:14:16.828662Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.06083","last_updated":"2020-08-19T18:07:40Z","snapshot_observed_at":"2026-08-10T17:55:43.816782Z","submitted_at":"2020-04-13T17:32:05Z","title":"The Fates of Merging Supermassive Black Holes and a Proposal for a New Class of X-Ray Sources","version":3},"cited_work":{"arxiv_id":"2004.06083","doi":null,"metadata_source":"pith","pith_arxiv_id":"2004.06083","snapshot_observed_at":"2026-08-06T15:14:16.795897Z","title":"The Fates of Merging Supermassive Black Holes and a Proposal for a New Class of X-Ray Sources","venue":"astro-ph.GA","work_id":"42ddaad8-0ba6-4b52-810c-0169cf415c9d","year":2020},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.708646Z"},"links":{"cited_paper":"/paper/2004.06083","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:cc71e07ae2bea2ebd9f85bb22b7846266b1db517fe56209a4c84b33f81ef4683","observation_id":"0fe0c28a-d246-4698-861e-37cba15abb66","resolution":{"observed_at":"2026-08-06T15:14:16.803484Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.028977Z","title":"Red Button","venue":null,"work_id":"09fc904c-09de-4f0c-89e5-14b17d4d2ba3","year":2016},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.727994Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:6813fd6db8ad7fd2fbad1caf6c436d57338e46c78db77830dc162438448a5990","observation_id":"02197dd9-d428-456a-b78c-87ad2378f729","resolution":{"observed_at":"2026-08-06T15:14:17.034115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.03827","last_updated":"2024-03-02T21:33:53Z","snapshot_observed_at":"2026-08-10T11:37:23.229129Z","submitted_at":"2022-12-07T18:17:56Z","title":"Discovering Latent Knowledge in Language Models Without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.03827","snapshot_observed_at":"2026-08-06T15:14:16.689102Z","title":"Discovering latent knowledge in language models without supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2000,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.689102Z"},"links":{"cited_paper":"/paper/2212.03827","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:22120d25440a42162886c9fd7035611761e979e24ef7c1e0bc1fba6155f76db6","observation_id":"8a1a899e-d56f-4157-9daa-0fef7608adf1","resolution":{"observed_at":"2026-08-06T15:14:16.689102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-06T15:14:16.648067Z","title":"Training a helpful and harmless assistant with rlhf","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.648067Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:c075b500dd45ca3848ebb178e58186faa9baa5501d3acc9e39613d1c05f4b42d","observation_id":"dd39c946-5aac-4e79-aa74-8da336e6e031","resolution":{"observed_at":"2026-08-06T15:14:16.648067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.08575","last_updated":"2018-10-19T16:30:48Z","snapshot_observed_at":"2026-08-04T21:33:05.702276Z","submitted_at":"2018-10-19T16:30:48Z","title":"Supervising strong learners by amplifying weak experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.08575","snapshot_observed_at":"2026-08-06T15:14:16.652974Z","title":"Supervising strong learners by amplifying weak experts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.652974Z"},"links":{"cited_paper":"/paper/1810.08575","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:c752dc274c2d8ac595e042665491f18348a7672f57010d053f31ff5721d36f82","observation_id":"ad369e9c-2031-4c53-8021-f4544014d216","resolution":{"observed_at":"2026-08-06T15:14:16.652974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.14375","last_updated":"2022-09-28T19:04:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-28T19:04:43Z","title":"Improving alignment of dialogue agents via targeted human judgements","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.14375","snapshot_observed_at":"2026-08-06T15:14:16.657949Z","title":"Improving align- ment of dialogue agents via targeted human judgements","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.657949Z"},"links":{"cited_paper":"/paper/2209.14375","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:0f7b421aaaa7daa9fcd5c6fc83498c9a3bd7fcff9943a0d75db111f83c0ccb4f","observation_id":"a1494bd8-4baf-4426-8e33-f0d8c6450a55","resolution":{"observed_at":"2026-08-06T15:14:16.657949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.05802","last_updated":"2022-06-14T01:16:24Z","snapshot_observed_at":"2026-07-06T13:19:58.934755Z","submitted_at":"2022-06-12T17:40:53Z","title":"Self-critiquing models for assisting human evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.05802","snapshot_observed_at":"2026-08-06T15:14:16.698792Z","title":"Self-critiquing models for assisting human evaluators.arXiv preprint arXiv:2206.05802,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.698792Z"},"links":{"cited_paper":"/paper/2206.05802","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:04f05451d1935dcb8acc28a9a7eff75cd694f340347d314f3378ad3149856ab7","observation_id":"4cbba20d-fc0e-45f2-833b-801d7b7fbe1b","resolution":{"observed_at":"2026-08-06T15:14:16.698792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02231","last_updated":"2023-10-03T17:41:46Z","snapshot_observed_at":"2026-08-10T17:54:58.730456Z","submitted_at":"2023-10-03T17:41:46Z","title":"Spin-Spin Coupling at Small $x$: Worm-Gear and Pretzelosity TMDs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02231","snapshot_observed_at":"2026-08-06T15:14:16.713331Z","title":"Alignment of language agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.713331Z"},"links":{"cited_paper":"/paper/2310.02231","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:d4f7c7ede808365ad942bfa79020a4226d7d138de9452b3d3bd09e04bb76d27f","observation_id":"858244ef-3017-4256-819a-15683edcdaf7","resolution":{"observed_at":"2026-08-06T15:14:16.713331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11147","last_updated":"2022-03-21T17:26:29Z","snapshot_observed_at":"2026-08-01T19:23:25.711617Z","submitted_at":"2022-03-21T17:26:29Z","title":"Teaching language models to support answers with verified quotes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11147","snapshot_observed_at":"2026-08-06T15:14:16.679333Z","title":"Teaching language models to support answers with verified quotes.arXiv preprint arXiv:2203.11147,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.679333Z"},"links":{"cited_paper":"/paper/2203.11147","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:5c4df11d9fa6a1e3adc17ee857b5393fa1761d8f576da1fc41b48141fe7684a8","observation_id":"d9c4c7f2-95ab-46a5-925c-e1e001df020f","resolution":{"observed_at":"2026-08-06T15:14:16.679333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.082720Z","title":"Pebble: Feedback- efficient interactive reinforcement learning via relabeling experience and un- supervised pre-training","venue":null,"work_id":"eb1a4119-7b0e-4c87-957d-e2d1f43ad201","year":2021},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.674730Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:97fb86282cee0fa9b2e6416920e58a3ca69b4c1d678e152d5f382fc0c385235a","observation_id":"0719c205-ff88-44cc-a248-71620dc3f487","resolution":{"observed_at":"2026-08-06T15:14:17.087493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.046056Z","title":"Red button,","venue":null,"work_id":"2837bb63-1dc5-47fc-95c0-e84b6c7be0f6","year":2016},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.720775Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:4ca4763b38794cc7b0f9f2f4dd19a6d9ee849760cd7fc2ca5e158e3fe1822382","observation_id":"8029b33b-822d-4a61-b8c2-f69458cf70f9","resolution":{"observed_at":"2026-08-06T15:14:17.052660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-10T17:55:43.321594Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback"},"reference_resolution":{"displayed":17,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":4},"total_outbound_references":17},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 17 of 17 outbound references and 0 inbound Pith citation observations for arXiv:2507.21131."}