{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4R6EEAZVNFCOCXY2C7BO4SOUKQ","short_pith_number":"pith:4R6EEAZV","schema_version":"1.0","canonical_sha256":"e47c4203356944e15f1a17c2ee49d4543bfd3c870c0e57f0fc5c71adac241cd9","source":{"kind":"arxiv","id":"2412.08378","version":3},"attestation_state":"computed","paper":{"title":"FILA: Fine-Grained Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Zheng, Jun Song, Shiding Zhu, Wenhui Dong, Yanan Guo, Yingbo Wang","submitted_at":"2024-12-11T13:41:21Z","abstract_excerpt":"Recently, there has been growing interest in the capability of multimodal large language models (MLLMs) to process high-resolution images. A common approach currently involves dynamically cropping the original high-resolution image into smaller sub-images, which are then fed into a vision encoder that was pre-trained on lower-resolution images. However, this cropping approach often truncates objects and connected areas in the original image, causing semantic breaks. To address this limitation, we introduce HyViLM, designed to process images of any resolution while retaining the overall context"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08378","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-11T13:41:21Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5643bd289d860017b1cd6b3d96d232854fec4e1db61b5b8826cd43573637ca17","abstract_canon_sha256":"69ec308adee73eeae2fc6a78a813c445759f12084974dd336cac8ed5cc6afdce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:13.125731Z","signature_b64":"cmOH+oFiZRKCR+UdaZGRWxbwVv50v0rmsvJAI4Z6ABOfEh0/gshhY4guFkdHnJclkXYEZkaSHOdwAFXaiLqUBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e47c4203356944e15f1a17c2ee49d4543bfd3c870c0e57f0fc5c71adac241cd9","last_reissued_at":"2026-07-05T10:56:13.125267Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:13.125267Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FILA: Fine-Grained Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Zheng, Jun Song, Shiding Zhu, Wenhui Dong, Yanan Guo, Yingbo Wang","submitted_at":"2024-12-11T13:41:21Z","abstract_excerpt":"Recently, there has been growing interest in the capability of multimodal large language models (MLLMs) to process high-resolution images. A common approach currently involves dynamically cropping the original high-resolution image into smaller sub-images, which are then fed into a vision encoder that was pre-trained on lower-resolution images. However, this cropping approach often truncates objects and connected areas in the original image, causing semantic breaks. To address this limitation, we introduce HyViLM, designed to process images of any resolution while retaining the overall context"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08378","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08378","created_at":"2026-07-05T10:56:13.125326+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08378v3","created_at":"2026-07-05T10:56:13.125326+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08378","created_at":"2026-07-05T10:56:13.125326+00:00"},{"alias_kind":"pith_short_12","alias_value":"4R6EEAZVNFCO","created_at":"2026-07-05T10:56:13.125326+00:00"},{"alias_kind":"pith_short_16","alias_value":"4R6EEAZVNFCOCXY2","created_at":"2026-07-05T10:56:13.125326+00:00"},{"alias_kind":"pith_short_8","alias_value":"4R6EEAZV","created_at":"2026-07-05T10:56:13.125326+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16909","citing_title":"TOBench: A Task-Oriented Omni-Modal Benchmark for Real-World Tool-Using Agents","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ","json":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ.json","graph_json":"https://pith.science/api/pith-number/4R6EEAZVNFCOCXY2C7BO4SOUKQ/graph.json","events_json":"https://pith.science/api/pith-number/4R6EEAZVNFCOCXY2C7BO4SOUKQ/events.json","paper":"https://pith.science/paper/4R6EEAZV"},"agent_actions":{"view_html":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ","download_json":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ.json","view_paper":"https://pith.science/paper/4R6EEAZV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08378&json=true","fetch_graph":"https://pith.science/api/pith-number/4R6EEAZVNFCOCXY2C7BO4SOUKQ/graph.json","fetch_events":"https://pith.science/api/pith-number/4R6EEAZVNFCOCXY2C7BO4SOUKQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ/action/storage_attestation","attest_author":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ/action/author_attestation","sign_citation":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ/action/citation_signature","submit_replication":"https://pith.science/pith/4R6EEAZVNFCOCXY2C7BO4SOUKQ/action/replication_record"}},"created_at":"2026-07-05T10:56:13.125326+00:00","updated_at":"2026-07-05T10:56:13.125326+00:00"}