{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DHHI2GTCXKW4AJGTECUGD57GY7","short_pith_number":"pith:DHHI2GTC","schema_version":"1.0","canonical_sha256":"19ce8d1a62baadc024d320a861f7e6c7daa8e71f39a06eaa01998758f3883e32","source":{"kind":"arxiv","id":"2406.10165","version":1},"attestation_state":"computed","paper":{"title":"CarLLaVA: Vision language models for camera-only closed-loop driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Alice Karnsund, Ana-Maria Marcu, Benoit Hanotte, Elahe Arani, Jamie Shotton, Jan H\\\"unermann, Katrin Renz, Long Chen, Oleg Sinavski","submitted_at":"2024-06-14T16:35:47Z","abstract_excerpt":"In this technical report, we present CarLLaVA, a Vision Language Model (VLM) for autonomous driving, developed for the CARLA Autonomous Driving Challenge 2.0. CarLLaVA uses the vision encoder of the LLaVA VLM and the LLaMA architecture as backbone, achieving state-of-the-art closed-loop driving performance with only camera input and without the need for complex or expensive labels. Additionally, we show preliminary results on predicting language commentary alongside the driving output. CarLLaVA uses a semi-disentangled output representation of both path predictions and waypoints, getting the a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10165","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-14T16:35:47Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"2004ab7caade1527d78d72d5957cec67ee0252b6e1cd61a746398c88a322c397","abstract_canon_sha256":"b31f3019b246ca084991f13ed207c8194d27ed5ef76864488f7c830af288b998"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:06.351416Z","signature_b64":"c2nlb+PKXsdk6RRBl5iO9JMtBNGtShMo4/qpwM4vdPFLv9KXF0k7jjyS40Iv2EverjrQy8Ynuz+PpgoKEQSVDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19ce8d1a62baadc024d320a861f7e6c7daa8e71f39a06eaa01998758f3883e32","last_reissued_at":"2026-07-05T08:32:06.350780Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:06.350780Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CarLLaVA: Vision language models for camera-only closed-loop driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Alice Karnsund, Ana-Maria Marcu, Benoit Hanotte, Elahe Arani, Jamie Shotton, Jan H\\\"unermann, Katrin Renz, Long Chen, Oleg Sinavski","submitted_at":"2024-06-14T16:35:47Z","abstract_excerpt":"In this technical report, we present CarLLaVA, a Vision Language Model (VLM) for autonomous driving, developed for the CARLA Autonomous Driving Challenge 2.0. CarLLaVA uses the vision encoder of the LLaVA VLM and the LLaMA architecture as backbone, achieving state-of-the-art closed-loop driving performance with only camera input and without the need for complex or expensive labels. Additionally, we show preliminary results on predicting language commentary alongside the driving output. CarLLaVA uses a semi-disentangled output representation of both path predictions and waypoints, getting the a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10165","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10165/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10165","created_at":"2026-07-05T08:32:06.350848+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10165v1","created_at":"2026-07-05T08:32:06.350848+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10165","created_at":"2026-07-05T08:32:06.350848+00:00"},{"alias_kind":"pith_short_12","alias_value":"DHHI2GTCXKW4","created_at":"2026-07-05T08:32:06.350848+00:00"},{"alias_kind":"pith_short_16","alias_value":"DHHI2GTCXKW4AJGT","created_at":"2026-07-05T08:32:06.350848+00:00"},{"alias_kind":"pith_short_8","alias_value":"DHHI2GTC","created_at":"2026-07-05T08:32:06.350848+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01658","citing_title":"Teaching Vision-Language-Action Models What to See and Where to Look","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12396","citing_title":"VLGA: Vision-Language-Geometry-Action Models for Autonomous Driving","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22089","citing_title":"LVDrive: Latent Visual Representation Enhanced Vision-Language-Action Autonomous Driving Model","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2509.00789","citing_title":"CogDriver: Integrating Cognitive Inertia for Temporally Coherent Planning in Autonomous Driving","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19755","citing_title":"ORION: A Holistic End-to-End Autonomous Driving Framework by Vision-Language Instructed Action Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12796","citing_title":"DriveVLA-W0: World Models Amplify Data Scaling Law in Autonomous Driving","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01762","citing_title":"AlignDrive: Aligned Lateral-Longitudinal Planning for End-to-End Autonomous Driving","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13757","citing_title":"AutoVLA: A Vision-Language-Action Model for End-to-End Autonomous Driving with Adaptive Reasoning and Reinforcement Fine-Tuning","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00813","citing_title":"DVGT-2: Vision-Geometry-Action Model for Autonomous Driving at Scale","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10564","citing_title":"DeepSight: Long-Horizon World Modeling via Latent States Prediction for End-to-End Autonomous Driving","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7","json":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7.json","graph_json":"https://pith.science/api/pith-number/DHHI2GTCXKW4AJGTECUGD57GY7/graph.json","events_json":"https://pith.science/api/pith-number/DHHI2GTCXKW4AJGTECUGD57GY7/events.json","paper":"https://pith.science/paper/DHHI2GTC"},"agent_actions":{"view_html":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7","download_json":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7.json","view_paper":"https://pith.science/paper/DHHI2GTC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10165&json=true","fetch_graph":"https://pith.science/api/pith-number/DHHI2GTCXKW4AJGTECUGD57GY7/graph.json","fetch_events":"https://pith.science/api/pith-number/DHHI2GTCXKW4AJGTECUGD57GY7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7/action/storage_attestation","attest_author":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7/action/author_attestation","sign_citation":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7/action/citation_signature","submit_replication":"https://pith.science/pith/DHHI2GTCXKW4AJGTECUGD57GY7/action/replication_record"}},"created_at":"2026-07-05T08:32:06.350848+00:00","updated_at":"2026-07-05T08:32:06.350848+00:00"}