{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:II3TKI3IQBDJT2P55QUPHOLTUJ","short_pith_number":"pith:II3TKI3I","canonical_record":{"source":{"id":"2505.08735","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-13T16:47:00Z","cross_cats_sorted":[],"title_canon_sha256":"5052dd7c8a9868fe2f8bca6cf5bd2528f9d2184cbcbf42f038f89313b5215870","abstract_canon_sha256":"56b71fd3e5afc1a098f18ad99f6fe36e4debb228eef249967136e275efae834f"},"schema_version":"1.0"},"canonical_sha256":"4237352368804699e9fdec28f3b973a265edd200b7f807222ee51241f3f3a9e0","source":{"kind":"arxiv","id":"2505.08735","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.08735","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"arxiv_version","alias_value":"2505.08735v1","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08735","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"pith_short_12","alias_value":"II3TKI3IQBDJ","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"pith_short_16","alias_value":"II3TKI3IQBDJT2P5","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"pith_short_8","alias_value":"II3TKI3I","created_at":"2026-07-05T11:02:35Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:II3TKI3IQBDJT2P55QUPHOLTUJ","target":"record","payload":{"canonical_record":{"source":{"id":"2505.08735","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-13T16:47:00Z","cross_cats_sorted":[],"title_canon_sha256":"5052dd7c8a9868fe2f8bca6cf5bd2528f9d2184cbcbf42f038f89313b5215870","abstract_canon_sha256":"56b71fd3e5afc1a098f18ad99f6fe36e4debb228eef249967136e275efae834f"},"schema_version":"1.0"},"canonical_sha256":"4237352368804699e9fdec28f3b973a265edd200b7f807222ee51241f3f3a9e0","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:35.556561Z","signature_b64":"jZSSdzQw8p36brU2HF1Id1QnbgTaBRmXLF/42yLT2LeHleueSCJhSUeAc9JRXKZpeIdfpi6BuS1mG8iVhkctBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4237352368804699e9fdec28f3b973a265edd200b7f807222ee51241f3f3a9e0","last_reissued_at":"2026-07-05T11:02:35.556068Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:35.556068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.08735","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:02:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YQegsMlKC+BGCWSO90eck5CUyNQOoa9lOZYAwHX2k6MtDfsVVwXGPNBsred/8sW64NMcvrJgxp0LfSfxig8jDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T17:53:53.562209Z"},"content_sha256":"2d85e7eff0fcd8da1ede76ce5deb31f033855fb19ea138fe81f832d0b7f72c46","schema_version":"1.0","event_id":"sha256:2d85e7eff0fcd8da1ede76ce5deb31f033855fb19ea138fe81f832d0b7f72c46"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:II3TKI3IQBDJT2P55QUPHOLTUJ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Preference Optimization for Combinatorial Optimization Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bin Zhu, Chun Yuan, Guanquan Lin, Lijun Sun, Mingjun Pan, You-Wei Luo, Zhien Dai","submitted_at":"2025-05-13T16:47:00Z","abstract_excerpt":"Reinforcement Learning (RL) has emerged as a powerful tool for neural combinatorial optimization, enabling models to learn heuristics that solve complex problems without requiring expert knowledge. Despite significant progress, existing RL approaches face challenges such as diminishing reward signals and inefficient exploration in vast combinatorial action spaces, leading to inefficiency. In this paper, we propose Preference Optimization, a novel method that transforms quantitative reward signals into qualitative preference signals via statistical comparison modeling, emphasizing the superiori"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08735","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08735/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:02:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"D7SV1iEkzHdgiJsk5m9IAAMUUDFlc/RUfVrQUQ2T1sXx4fPchxYPF5YPe1ZX+TB4PrDLP9CRwDEHHa/dX9RtBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T17:53:53.562783Z"},"content_sha256":"13f70e9376ac59a0394abddce680098de50c7e8d7ef90de1aed59faf36096e43","schema_version":"1.0","event_id":"sha256:13f70e9376ac59a0394abddce680098de50c7e8d7ef90de1aed59faf36096e43"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/II3TKI3IQBDJT2P55QUPHOLTUJ/bundle.json","state_url":"https://pith.science/pith/II3TKI3IQBDJT2P55QUPHOLTUJ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/II3TKI3IQBDJT2P55QUPHOLTUJ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T17:53:53Z","links":{"resolver":"https://pith.science/pith/II3TKI3IQBDJT2P55QUPHOLTUJ","bundle":"https://pith.science/pith/II3TKI3IQBDJT2P55QUPHOLTUJ/bundle.json","state":"https://pith.science/pith/II3TKI3IQBDJT2P55QUPHOLTUJ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/II3TKI3IQBDJT2P55QUPHOLTUJ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:II3TKI3IQBDJT2P55QUPHOLTUJ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"56b71fd3e5afc1a098f18ad99f6fe36e4debb228eef249967136e275efae834f","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-13T16:47:00Z","title_canon_sha256":"5052dd7c8a9868fe2f8bca6cf5bd2528f9d2184cbcbf42f038f89313b5215870"},"schema_version":"1.0","source":{"id":"2505.08735","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.08735","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"arxiv_version","alias_value":"2505.08735v1","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08735","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"pith_short_12","alias_value":"II3TKI3IQBDJ","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"pith_short_16","alias_value":"II3TKI3IQBDJT2P5","created_at":"2026-07-05T11:02:35Z"},{"alias_kind":"pith_short_8","alias_value":"II3TKI3I","created_at":"2026-07-05T11:02:35Z"}],"graph_snapshots":[{"event_id":"sha256:13f70e9376ac59a0394abddce680098de50c7e8d7ef90de1aed59faf36096e43","target":"graph","created_at":"2026-07-05T11:02:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.08735/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning (RL) has emerged as a powerful tool for neural combinatorial optimization, enabling models to learn heuristics that solve complex problems without requiring expert knowledge. Despite significant progress, existing RL approaches face challenges such as diminishing reward signals and inefficient exploration in vast combinatorial action spaces, leading to inefficiency. In this paper, we propose Preference Optimization, a novel method that transforms quantitative reward signals into qualitative preference signals via statistical comparison modeling, emphasizing the superiori","authors_text":"Bin Zhu, Chun Yuan, Guanquan Lin, Lijun Sun, Mingjun Pan, You-Wei Luo, Zhien Dai","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-13T16:47:00Z","title":"Preference Optimization for Combinatorial Optimization Problems"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08735","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2d85e7eff0fcd8da1ede76ce5deb31f033855fb19ea138fe81f832d0b7f72c46","target":"record","created_at":"2026-07-05T11:02:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"56b71fd3e5afc1a098f18ad99f6fe36e4debb228eef249967136e275efae834f","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-13T16:47:00Z","title_canon_sha256":"5052dd7c8a9868fe2f8bca6cf5bd2528f9d2184cbcbf42f038f89313b5215870"},"schema_version":"1.0","source":{"id":"2505.08735","kind":"arxiv","version":1}},"canonical_sha256":"4237352368804699e9fdec28f3b973a265edd200b7f807222ee51241f3f3a9e0","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4237352368804699e9fdec28f3b973a265edd200b7f807222ee51241f3f3a9e0","first_computed_at":"2026-07-05T11:02:35.556068Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:02:35.556068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jZSSdzQw8p36brU2HF1Id1QnbgTaBRmXLF/42yLT2LeHleueSCJhSUeAc9JRXKZpeIdfpi6BuS1mG8iVhkctBw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:02:35.556561Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.08735","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2d85e7eff0fcd8da1ede76ce5deb31f033855fb19ea138fe81f832d0b7f72c46","sha256:13f70e9376ac59a0394abddce680098de50c7e8d7ef90de1aed59faf36096e43"],"state_sha256":"9fc3389b0da381af693a5a693fe1b476425a6b83ff8aeea78eb0c2c26803ae3a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hwR1YCn0Mdv6tCTb5KKbxegCklKpD+9z30B3riALj6/QbVBo2W6ib4XgIvgYU/4Ct+BjMJClUmiWKfUAc163Cg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T17:53:53.581029Z","bundle_sha256":"a7c0c0a05f7e77a94e513ff844d154af552cdf11b5b2532bb8c08fcd9fb7514e"}}