{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GATOGMQKNFSCZAGVOBMPXLEHCM","short_pith_number":"pith:GATOGMQK","schema_version":"1.0","canonical_sha256":"3026e3320a69642c80d57058fbac87131afecc1a87834c2e3abc943aea0c9d79","source":{"kind":"arxiv","id":"2402.14289","version":1},"attestation_state":"computed","paper":{"title":"TinyLLaVA: A Framework of Small-scale Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Baichuan Zhou, Jie Luo, Ji Wu, Junlong Jia, Lei Huang, Xien Liu, Xi Weng, Ying Hu","submitted_at":"2024-02-22T05:05:30Z","abstract_excerpt":"We present the TinyLLaVA framework that provides a unified perspective in designing and analyzing the small-scale Large Multimodal Models (LMMs). We empirically study the effects of different vision encoders, connection modules, language models, training data and training recipes. Our extensive experiments showed that better quality of data combined with better training recipes, smaller LMMs can consistently achieve on-par performances compared to bigger LMMs. Under our framework, we train a family of small-scale LMMs. Our best model, TinyLLaVA-3.1B, achieves better overall performance against"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14289","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-22T05:05:30Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8906ab8e8ce88494ccff872266b194903373db01685eb52857b025554a6ca8d8","abstract_canon_sha256":"196406d52f3c9727523dc4b135842f54b33ff281f9aa35c81b4f6ecfd6131ad2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:08.125353Z","signature_b64":"1CQ5kyqFNJjOFvlVAYw81NsFQAZQ575asD4T6VumfCgrqfhYfDKE+o1bcDUPGW5CsZljoy16Yyb8gXc2NjiJDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3026e3320a69642c80d57058fbac87131afecc1a87834c2e3abc943aea0c9d79","last_reissued_at":"2026-07-05T07:48:08.124881Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:08.124881Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TinyLLaVA: A Framework of Small-scale Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Baichuan Zhou, Jie Luo, Ji Wu, Junlong Jia, Lei Huang, Xien Liu, Xi Weng, Ying Hu","submitted_at":"2024-02-22T05:05:30Z","abstract_excerpt":"We present the TinyLLaVA framework that provides a unified perspective in designing and analyzing the small-scale Large Multimodal Models (LMMs). We empirically study the effects of different vision encoders, connection modules, language models, training data and training recipes. Our extensive experiments showed that better quality of data combined with better training recipes, smaller LMMs can consistently achieve on-par performances compared to bigger LMMs. Under our framework, we train a family of small-scale LMMs. Our best model, TinyLLaVA-3.1B, achieves better overall performance against"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14289","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14289/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14289","created_at":"2026-07-05T07:48:08.124941+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14289v1","created_at":"2026-07-05T07:48:08.124941+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14289","created_at":"2026-07-05T07:48:08.124941+00:00"},{"alias_kind":"pith_short_12","alias_value":"GATOGMQKNFSC","created_at":"2026-07-05T07:48:08.124941+00:00"},{"alias_kind":"pith_short_16","alias_value":"GATOGMQKNFSCZAGV","created_at":"2026-07-05T07:48:08.124941+00:00"},{"alias_kind":"pith_short_8","alias_value":"GATOGMQK","created_at":"2026-07-05T07:48:08.124941+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08029","citing_title":"Rethinking Small VLM Quantization: From Component-Wise Analysis to Hardware-Aware Edge Deployment","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24156","citing_title":"Accelerating Multimodal Large Language Models with Prior-Corrected Token Reduction","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":281,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28604","citing_title":"Mining Multi-Modality Spatio-Temporal Cues for Video Important Person Identification","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00573","citing_title":"LASER: Loss-Aware Singular-value Decomposition and Rank Allocation for Efficient Low-Precision Vision-Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00535","citing_title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2412.08110","citing_title":"The ART of Composition: Attention-Regularized Training for Compositional Visual Grounding","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2511.23253","citing_title":"AgroCoT: A Chain-of-Thought Benchmark for Evaluating Reasoning in Vision-Language Models for Agriculture","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18117","citing_title":"Online In-Context Distillation for Low-Resource Vision Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2403.20330","citing_title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10641","citing_title":"LLaVA-CKD: Bottom-Up Cascaded Knowledge Distillation for Vision-Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01844","citing_title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04372","citing_title":"Graph-to-Frame RAG: Visual-Space Knowledge Fusion for Training-Free and Auditable Video Reasoning","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14629","citing_title":"Switch-KD: Visual-Switch Knowledge Distillation for Vision-Language Models","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21952","citing_title":"Focus Session: Hardware and Software Techniques for Accelerating Multimodal Foundation Models","ref_index":70,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM","json":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM.json","graph_json":"https://pith.science/api/pith-number/GATOGMQKNFSCZAGVOBMPXLEHCM/graph.json","events_json":"https://pith.science/api/pith-number/GATOGMQKNFSCZAGVOBMPXLEHCM/events.json","paper":"https://pith.science/paper/GATOGMQK"},"agent_actions":{"view_html":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM","download_json":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM.json","view_paper":"https://pith.science/paper/GATOGMQK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14289&json=true","fetch_graph":"https://pith.science/api/pith-number/GATOGMQKNFSCZAGVOBMPXLEHCM/graph.json","fetch_events":"https://pith.science/api/pith-number/GATOGMQKNFSCZAGVOBMPXLEHCM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM/action/storage_attestation","attest_author":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM/action/author_attestation","sign_citation":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM/action/citation_signature","submit_replication":"https://pith.science/pith/GATOGMQKNFSCZAGVOBMPXLEHCM/action/replication_record"}},"created_at":"2026-07-05T07:48:08.124941+00:00","updated_at":"2026-07-05T07:48:08.124941+00:00"}