{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:YZXRZNOBSFJX4E4WWP4ERH5YVA","short_pith_number":"pith:YZXRZNOB","canonical_record":{"source":{"id":"2306.00398","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-01T07:00:07Z","cross_cats_sorted":[],"title_canon_sha256":"c8254db0374ed555ebbbc298a6bb56d209cd87a1543a44bc829080cee93e4623","abstract_canon_sha256":"4e3734f58f234a310c7db8e2bedffb471ea8e083ccc59be85d0682ad0aaa4780"},"schema_version":"1.0"},"canonical_sha256":"c66f1cb5c191537e1396b3f8489fb8a818c54deacb52ffce4ca1825d939154eb","source":{"kind":"arxiv","id":"2306.00398","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.00398","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"arxiv_version","alias_value":"2306.00398v3","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00398","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"pith_short_12","alias_value":"YZXRZNOBSFJX","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"pith_short_16","alias_value":"YZXRZNOBSFJX4E4W","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"pith_short_8","alias_value":"YZXRZNOB","created_at":"2026-07-05T09:58:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:YZXRZNOBSFJX4E4WWP4ERH5YVA","target":"record","payload":{"canonical_record":{"source":{"id":"2306.00398","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-01T07:00:07Z","cross_cats_sorted":[],"title_canon_sha256":"c8254db0374ed555ebbbc298a6bb56d209cd87a1543a44bc829080cee93e4623","abstract_canon_sha256":"4e3734f58f234a310c7db8e2bedffb471ea8e083ccc59be85d0682ad0aaa4780"},"schema_version":"1.0"},"canonical_sha256":"c66f1cb5c191537e1396b3f8489fb8a818c54deacb52ffce4ca1825d939154eb","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:16.066433Z","signature_b64":"/8cvTb4F4xMVn5Ta2rs+hfswnuwUq2b5q3Z72gve8lDIYEG/sgtgklCOdJllQkZd76FM89eh3U0wFv3Qrj1xCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c66f1cb5c191537e1396b3f8489fb8a818c54deacb52ffce4ca1825d939154eb","last_reissued_at":"2026-07-05T09:58:16.066045Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:16.066045Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2306.00398","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:58:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CAb9Y9veZ7in/4Puq/ZMU8QNnBP231u2yFNUYqULN2IPxCm73pjoC04TTc2SS2gaMV1JP4yUyAcddWnjMtEwAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T06:14:40.632118Z"},"content_sha256":"ef104baa09ad8bbdc53228b59b7df4e165d630f90b4c58ef37bfda6ce1827163","schema_version":"1.0","event_id":"sha256:ef104baa09ad8bbdc53228b59b7df4e165d630f90b4c58ef37bfda6ce1827163"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:YZXRZNOBSFJX4E4WWP4ERH5YVA","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Preference-grounded Token-level Guidance for Language Model Fine-tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Caiming Xiong, Congying Xia, Mingyuan Zhou, Shentao Yang, Shujian Zhang, Yihao Feng","submitted_at":"2023-06-01T07:00:07Z","abstract_excerpt":"Aligning language models (LMs) with preferences is an important problem in natural language generation. A key challenge is that preferences are typically provided at the sequence level while LM training and generation both occur at the token level. There is, therefore, a granularity mismatch between the preference and the LM training losses, which may complicate the learning problem. In this paper, we address this issue by developing an alternate training process, where we iterate between grounding the sequence-level preference into token-level training guidance, and improving the LM with the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00398","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.00398/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:58:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iXYWCfj/S6bzb/yH9urbSz3mp2YP7ny8gdZBXGSFoGH8CCtK9poESI6PgBzRqhaPbB9nO58BFrpmZ1vE96G5BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T06:14:40.632726Z"},"content_sha256":"500ff466359c697437ac69fe860788acdf908b47c74a96f1436530ac67263df4","schema_version":"1.0","event_id":"sha256:500ff466359c697437ac69fe860788acdf908b47c74a96f1436530ac67263df4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YZXRZNOBSFJX4E4WWP4ERH5YVA/bundle.json","state_url":"https://pith.science/pith/YZXRZNOBSFJX4E4WWP4ERH5YVA/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YZXRZNOBSFJX4E4WWP4ERH5YVA/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-22T06:14:40Z","links":{"resolver":"https://pith.science/pith/YZXRZNOBSFJX4E4WWP4ERH5YVA","bundle":"https://pith.science/pith/YZXRZNOBSFJX4E4WWP4ERH5YVA/bundle.json","state":"https://pith.science/pith/YZXRZNOBSFJX4E4WWP4ERH5YVA/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YZXRZNOBSFJX4E4WWP4ERH5YVA/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:YZXRZNOBSFJX4E4WWP4ERH5YVA","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4e3734f58f234a310c7db8e2bedffb471ea8e083ccc59be85d0682ad0aaa4780","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-01T07:00:07Z","title_canon_sha256":"c8254db0374ed555ebbbc298a6bb56d209cd87a1543a44bc829080cee93e4623"},"schema_version":"1.0","source":{"id":"2306.00398","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.00398","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"arxiv_version","alias_value":"2306.00398v3","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00398","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"pith_short_12","alias_value":"YZXRZNOBSFJX","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"pith_short_16","alias_value":"YZXRZNOBSFJX4E4W","created_at":"2026-07-05T09:58:16Z"},{"alias_kind":"pith_short_8","alias_value":"YZXRZNOB","created_at":"2026-07-05T09:58:16Z"}],"graph_snapshots":[{"event_id":"sha256:500ff466359c697437ac69fe860788acdf908b47c74a96f1436530ac67263df4","target":"graph","created_at":"2026-07-05T09:58:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2306.00398/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Aligning language models (LMs) with preferences is an important problem in natural language generation. A key challenge is that preferences are typically provided at the sequence level while LM training and generation both occur at the token level. There is, therefore, a granularity mismatch between the preference and the LM training losses, which may complicate the learning problem. In this paper, we address this issue by developing an alternate training process, where we iterate between grounding the sequence-level preference into token-level training guidance, and improving the LM with the ","authors_text":"Caiming Xiong, Congying Xia, Mingyuan Zhou, Shentao Yang, Shujian Zhang, Yihao Feng","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-01T07:00:07Z","title":"Preference-grounded Token-level Guidance for Language Model Fine-tuning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00398","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ef104baa09ad8bbdc53228b59b7df4e165d630f90b4c58ef37bfda6ce1827163","target":"record","created_at":"2026-07-05T09:58:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4e3734f58f234a310c7db8e2bedffb471ea8e083ccc59be85d0682ad0aaa4780","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-01T07:00:07Z","title_canon_sha256":"c8254db0374ed555ebbbc298a6bb56d209cd87a1543a44bc829080cee93e4623"},"schema_version":"1.0","source":{"id":"2306.00398","kind":"arxiv","version":3}},"canonical_sha256":"c66f1cb5c191537e1396b3f8489fb8a818c54deacb52ffce4ca1825d939154eb","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c66f1cb5c191537e1396b3f8489fb8a818c54deacb52ffce4ca1825d939154eb","first_computed_at":"2026-07-05T09:58:16.066045Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:58:16.066045Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"/8cvTb4F4xMVn5Ta2rs+hfswnuwUq2b5q3Z72gve8lDIYEG/sgtgklCOdJllQkZd76FM89eh3U0wFv3Qrj1xCA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:58:16.066433Z","signed_message":"canonical_sha256_bytes"},"source_id":"2306.00398","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ef104baa09ad8bbdc53228b59b7df4e165d630f90b4c58ef37bfda6ce1827163","sha256:500ff466359c697437ac69fe860788acdf908b47c74a96f1436530ac67263df4"],"state_sha256":"3604c1081c804c750261963b042fc132874cb518004ff6445d5a1f8a18e2d38c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qA4zT0SH2bgeuXhkrGCOZGMwsV1ULXotE0faJtsTsbixwLna8r6o3G1X09sbl1kSR6SU0QczPM5PtGxR+catAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-22T06:14:40.637780Z","bundle_sha256":"cdc5aaff096df255a64c1797946d4713bfabfb2804b9251dd23e18312e4c0797"}}