{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:EGGUEICS6QSJG3FYBNQQRVPDMS","short_pith_number":"pith:EGGUEICS","schema_version":"1.0","canonical_sha256":"218d422052f424936cb80b6108d5e36495fe63ab6bd28ece7b3686563994b117","source":{"kind":"arxiv","id":"2004.09731","version":1},"attestation_state":"computed","paper":{"title":"Learning Goal-oriented Dialogue Policy with Opposite Agent Awareness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lizi Liao, Minlie Huang, Tat-Seng Chua, Xiaoyan Zhu, Yan Huang, Zheng Zhang, Zitao Liu","submitted_at":"2020-04-21T03:13:44Z","abstract_excerpt":"Most existing approaches for goal-oriented dialogue policy learning used reinforcement learning, which focuses on the target agent policy and simply treat the opposite agent policy as part of the environment. While in real-world scenarios, the behavior of an opposite agent often exhibits certain patterns or underlies hidden policies, which can be inferred and utilized by the target agent to facilitate its own decision making. This strategy is common in human mental simulation by first imaging a specific action and the probable results before really acting it. We therefore propose an opposite b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.09731","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-21T03:13:44Z","cross_cats_sorted":[],"title_canon_sha256":"7f487c4604ae0e14c6d40ab7b440d92ddff2c67d3e44d14fe76b5865f551ca80","abstract_canon_sha256":"a2cc5ce7650f3f64a084eb6325308829780a93558a29c6c656833c226ede62ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:56:53.101253Z","signature_b64":"HrS0kTEfMAd4yjypRMuqh6R3n17hlsjtoMlvs0sQh3ZbfanibqTMjgM0IdyIB1Viya/pzZQmiXmytR3Q+ZdFBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"218d422052f424936cb80b6108d5e36495fe63ab6bd28ece7b3686563994b117","last_reissued_at":"2026-07-05T00:56:53.100862Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:56:53.100862Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Goal-oriented Dialogue Policy with Opposite Agent Awareness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lizi Liao, Minlie Huang, Tat-Seng Chua, Xiaoyan Zhu, Yan Huang, Zheng Zhang, Zitao Liu","submitted_at":"2020-04-21T03:13:44Z","abstract_excerpt":"Most existing approaches for goal-oriented dialogue policy learning used reinforcement learning, which focuses on the target agent policy and simply treat the opposite agent policy as part of the environment. While in real-world scenarios, the behavior of an opposite agent often exhibits certain patterns or underlies hidden policies, which can be inferred and utilized by the target agent to facilitate its own decision making. This strategy is common in human mental simulation by first imaging a specific action and the probable results before really acting it. We therefore propose an opposite b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.09731","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.09731/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.09731","created_at":"2026-07-05T00:56:53.100922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.09731v1","created_at":"2026-07-05T00:56:53.100922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.09731","created_at":"2026-07-05T00:56:53.100922+00:00"},{"alias_kind":"pith_short_12","alias_value":"EGGUEICS6QSJ","created_at":"2026-07-05T00:56:53.100922+00:00"},{"alias_kind":"pith_short_16","alias_value":"EGGUEICS6QSJG3FY","created_at":"2026-07-05T00:56:53.100922+00:00"},{"alias_kind":"pith_short_8","alias_value":"EGGUEICS","created_at":"2026-07-05T00:56:53.100922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.19652","citing_title":"Tailored Conversations beyond LLMs: A RL-Based Dialogue Manager","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS","json":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS.json","graph_json":"https://pith.science/api/pith-number/EGGUEICS6QSJG3FYBNQQRVPDMS/graph.json","events_json":"https://pith.science/api/pith-number/EGGUEICS6QSJG3FYBNQQRVPDMS/events.json","paper":"https://pith.science/paper/EGGUEICS"},"agent_actions":{"view_html":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS","download_json":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS.json","view_paper":"https://pith.science/paper/EGGUEICS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.09731&json=true","fetch_graph":"https://pith.science/api/pith-number/EGGUEICS6QSJG3FYBNQQRVPDMS/graph.json","fetch_events":"https://pith.science/api/pith-number/EGGUEICS6QSJG3FYBNQQRVPDMS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS/action/storage_attestation","attest_author":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS/action/author_attestation","sign_citation":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS/action/citation_signature","submit_replication":"https://pith.science/pith/EGGUEICS6QSJG3FYBNQQRVPDMS/action/replication_record"}},"created_at":"2026-07-05T00:56:53.100922+00:00","updated_at":"2026-07-05T00:56:53.100922+00:00"}