{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FUM3P6KK3GIJOCHWP6LOFKHCWT","short_pith_number":"pith:FUM3P6KK","schema_version":"1.0","canonical_sha256":"2d19b7f94ad9909708f67f96e2a8e2b4e6d905a59c09debe7d444d9add1df203","source":{"kind":"arxiv","id":"2505.12194","version":1},"attestation_state":"computed","paper":{"title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Cecilia Mauceri, Christoffer Heckman, Doncey Albin, Dusty Woods, Xuefei Sun","submitted_at":"2025-05-18T02:07:55Z","abstract_excerpt":"Multimodal large language models (MLLMs) have demonstrated remarkable abilities in comprehending visual input alongside text input. Typically, these models are trained on extensive data sourced from the internet, which are sufficient for general tasks such as scene understanding and question answering. However, they often underperform on specialized tasks where online data is scarce, such as determining spatial relationships between objects or localizing unique target objects within a group of objects sharing similar features. In response to this challenge, we introduce the SUN-Spot v2.0 datas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12194","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-05-18T02:07:55Z","cross_cats_sorted":[],"title_canon_sha256":"0d1c53833eeedac34f4c288b37278a5c7db5357d5ecda0b36f4bb5e3caa024f6","abstract_canon_sha256":"d812b369c7706ff917b3b83722f3141319ee6a740af897848e41a325bc019902"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:59.948999Z","signature_b64":"u+ZM1TkrEEWvl03IjmPGBAxCHNgV1nB6Fd/MG1VTIUn8XLDCwgtFoBpgkr1W7m3u825WDSl9wyE8lRMQ6JvtCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d19b7f94ad9909708f67f96e2a8e2b4e6d905a59c09debe7d444d9add1df203","last_reissued_at":"2026-07-05T11:04:59.948391Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:59.948391Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Cecilia Mauceri, Christoffer Heckman, Doncey Albin, Dusty Woods, Xuefei Sun","submitted_at":"2025-05-18T02:07:55Z","abstract_excerpt":"Multimodal large language models (MLLMs) have demonstrated remarkable abilities in comprehending visual input alongside text input. Typically, these models are trained on extensive data sourced from the internet, which are sufficient for general tasks such as scene understanding and question answering. However, they often underperform on specialized tasks where online data is scarce, such as determining spatial relationships between objects or localizing unique target objects within a group of objects sharing similar features. In response to this challenge, we introduce the SUN-Spot v2.0 datas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12194","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12194/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12194","created_at":"2026-07-05T11:04:59.948460+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12194v1","created_at":"2026-07-05T11:04:59.948460+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12194","created_at":"2026-07-05T11:04:59.948460+00:00"},{"alias_kind":"pith_short_12","alias_value":"FUM3P6KK3GIJ","created_at":"2026-07-05T11:04:59.948460+00:00"},{"alias_kind":"pith_short_16","alias_value":"FUM3P6KK3GIJOCHW","created_at":"2026-07-05T11:04:59.948460+00:00"},{"alias_kind":"pith_short_8","alias_value":"FUM3P6KK","created_at":"2026-07-05T11:04:59.948460+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT","json":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT.json","graph_json":"https://pith.science/api/pith-number/FUM3P6KK3GIJOCHWP6LOFKHCWT/graph.json","events_json":"https://pith.science/api/pith-number/FUM3P6KK3GIJOCHWP6LOFKHCWT/events.json","paper":"https://pith.science/paper/FUM3P6KK"},"agent_actions":{"view_html":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT","download_json":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT.json","view_paper":"https://pith.science/paper/FUM3P6KK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12194&json=true","fetch_graph":"https://pith.science/api/pith-number/FUM3P6KK3GIJOCHWP6LOFKHCWT/graph.json","fetch_events":"https://pith.science/api/pith-number/FUM3P6KK3GIJOCHWP6LOFKHCWT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT/action/storage_attestation","attest_author":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT/action/author_attestation","sign_citation":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT/action/citation_signature","submit_replication":"https://pith.science/pith/FUM3P6KK3GIJOCHWP6LOFKHCWT/action/replication_record"}},"created_at":"2026-07-05T11:04:59.948460+00:00","updated_at":"2026-07-05T11:04:59.948460+00:00"}