{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:EV3B2VJDFSY7Y7HUSVQCEZKPDP","short_pith_number":"pith:EV3B2VJD","schema_version":"1.0","canonical_sha256":"25761d55232cb1fc7cf4956022654f1bcfbfe325f88f08c9cb8d0acf7af25451","source":{"kind":"arxiv","id":"2107.07201","version":3},"attestation_state":"computed","paper":{"title":"Neighbor-view Enhanced Model for Vision and Language Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dong An, Liang Wang, Qi Wu, Tieniu Tan, Yan Huang, Yuankai Qi","submitted_at":"2021-07-15T09:11:02Z","abstract_excerpt":"Vision and Language Navigation (VLN) requires an agent to navigate to a target location by following natural language instructions. Most of existing works represent a navigation candidate by the feature of the corresponding single view where the candidate lies in. However, an instruction may mention landmarks out of the single view as references, which might lead to failures of textual-visual matching of existing methods. In this work, we propose a multi-module Neighbor-View Enhanced Model (NvEM) to adaptively incorporate visual contexts from neighbor views for better textual-visual matching. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.07201","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-07-15T09:11:02Z","cross_cats_sorted":[],"title_canon_sha256":"ff90c6082874afec94138af9d0b0c69c7daf974f89627919743a5ce4e8a5e8a9","abstract_canon_sha256":"d11df8917668dad757556cec7f09341771fcbf0af093a93686b18ce46a3444db"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:00:18.805397Z","signature_b64":"aEJ1jHdbrRCaU42CNyNms4A1PWxWCMiKXuKMywLNCnTyZTDkdRgm/gf6X88A63wvKDx6s/jMaCgUuEGuaZv+CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25761d55232cb1fc7cf4956022654f1bcfbfe325f88f08c9cb8d0acf7af25451","last_reissued_at":"2026-07-05T03:00:18.804898Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:00:18.804898Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neighbor-view Enhanced Model for Vision and Language Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dong An, Liang Wang, Qi Wu, Tieniu Tan, Yan Huang, Yuankai Qi","submitted_at":"2021-07-15T09:11:02Z","abstract_excerpt":"Vision and Language Navigation (VLN) requires an agent to navigate to a target location by following natural language instructions. Most of existing works represent a navigation candidate by the feature of the corresponding single view where the candidate lies in. However, an instruction may mention landmarks out of the single view as references, which might lead to failures of textual-visual matching of existing methods. In this work, we propose a multi-module Neighbor-View Enhanced Model (NvEM) to adaptively incorporate visual contexts from neighbor views for better textual-visual matching. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.07201","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.07201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.07201","created_at":"2026-07-05T03:00:18.804955+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.07201v3","created_at":"2026-07-05T03:00:18.804955+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.07201","created_at":"2026-07-05T03:00:18.804955+00:00"},{"alias_kind":"pith_short_12","alias_value":"EV3B2VJDFSY7","created_at":"2026-07-05T03:00:18.804955+00:00"},{"alias_kind":"pith_short_16","alias_value":"EV3B2VJDFSY7Y7HU","created_at":"2026-07-05T03:00:18.804955+00:00"},{"alias_kind":"pith_short_8","alias_value":"EV3B2VJD","created_at":"2026-07-05T03:00:18.804955+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.16516","citing_title":"Think Hierarchically, Act Dynamically: Hierarchical Multi-modal Fusion and Reasoning for Vision-and-Language Navigation","ref_index":2021,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP","json":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP.json","graph_json":"https://pith.science/api/pith-number/EV3B2VJDFSY7Y7HUSVQCEZKPDP/graph.json","events_json":"https://pith.science/api/pith-number/EV3B2VJDFSY7Y7HUSVQCEZKPDP/events.json","paper":"https://pith.science/paper/EV3B2VJD"},"agent_actions":{"view_html":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP","download_json":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP.json","view_paper":"https://pith.science/paper/EV3B2VJD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.07201&json=true","fetch_graph":"https://pith.science/api/pith-number/EV3B2VJDFSY7Y7HUSVQCEZKPDP/graph.json","fetch_events":"https://pith.science/api/pith-number/EV3B2VJDFSY7Y7HUSVQCEZKPDP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP/action/storage_attestation","attest_author":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP/action/author_attestation","sign_citation":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP/action/citation_signature","submit_replication":"https://pith.science/pith/EV3B2VJDFSY7Y7HUSVQCEZKPDP/action/replication_record"}},"created_at":"2026-07-05T03:00:18.804955+00:00","updated_at":"2026-07-05T03:00:18.804955+00:00"}