{"as_of":"2026-08-16T02:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:46167a83dc388c41b1b4271a76ce55d412c37f15bccfd25c3f1bc28c96880866","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:57:43.814957Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.01409/citation-record","integrity":"/paper/2507.01409/integrity","json":"/paper/2507.01409/citation-record.json","paper":"/paper/2507.01409"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.373580Z","title":"nocaps: novel object caption- ing at scale","venue":null,"work_id":"418dec77-a98f-45e2-99ac-532ed8fb5adf","year":2019},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:40.863410Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:d8fad84320175d7a8eba01758d91c842b3b4612c03229b636557ade15ea0f1ed","observation_id":"23c193f0-fcbf-409e-9afb-1c3f6380cf0f","resolution":{"observed_at":"2026-08-06T20:57:50.377461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-06T20:57:41.010599Z","title":"Qwen-vl: A versatile vision-language model for un- derstanding, localization, text reading, and beyond","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.010599Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:ed533da4e6fad5c363aa767ecb62a44c8fb9a7ef71f447c97b3b14723e047e23","observation_id":"0c028d1e-0a60-42c1-92f7-fe72baa042e9","resolution":{"observed_at":"2026-08-06T20:57:41.010599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:41.063951Z","title":"Meteor: An automatic metric for mt evaluation with improved correlation with hu- man judgments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.063951Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:95e25e9da87bb6dad38ab4d82702135139e06d043c5a9168500afa48fe36664e","observation_id":"98a3d402-8f0e-484a-b3e2-1e931e31acf5","resolution":{"observed_at":"2026-08-06T20:57:41.063951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12971","last_updated":"2023-10-19T17:59:01Z","snapshot_observed_at":"2026-08-13T21:11:20.544454Z","submitted_at":"2023-10-19T17:59:01Z","title":"CLAIR: Evaluating Image Captions with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12971","snapshot_observed_at":"2026-08-06T20:57:41.148599Z","title":"Clair: Evaluating image captions with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.148599Z"},"links":{"cited_paper":"/paper/2310.12971","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:dce8a84d54fb5bfb7b3613e5802516951ff9db3cae45dac4480e9af5c3751d99","observation_id":"4bbe6334-ef0b-4e2e-bd90-19b83e20b05b","resolution":{"observed_at":"2026-08-06T20:57:41.148599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.327666Z","title":"Sharegpt4v: Improving large multi-modal models with better captions","venue":null,"work_id":"f0a00e50-7673-4d3f-80d6-8c5cdab8ed54","year":2025},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.200346Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:6a05fbb61cd4d516f962d7951d84bfd1403126613c4a7969d74793308d82cfe1","observation_id":"4bc130e2-e184-44c5-9f86-021db4f19c83","resolution":{"observed_at":"2026-08-06T20:57:50.361544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.148317Z","title":"Learning distinct and representative modes for image captioning","venue":null,"work_id":"68f5c3c7-9357-43e7-8728-df80540b8221","year":2022},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.272463Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:ef6e75394eee3e72cc1693f9818787ec8756c323c9b2116ff01ca2d6fbbf6f99","observation_id":"c3e50c8b-5a6d-4037-8224-16f6d5b40117","resolution":{"observed_at":"2026-08-06T20:57:50.256312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.068430Z","title":"Say as you wish: Fine-grained control of image caption generation with abstract scene graphs","venue":null,"work_id":"f01dec0a-fbc1-4a78-bfc7-10eba2fdbfa0","year":2020},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.323774Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:5cb94b899229bcb4f7cba505c9397751b5e39b839776373d26e0eceaaa0857dd","observation_id":"039945e0-ed40-4bbc-81e1-adb3b38bcef7","resolution":{"observed_at":"2026-08-06T20:57:50.130384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:49.799247Z","title":"Length- controllable image captioning","venue":null,"work_id":"bc0b5756-c790-40d8-b42b-53afe28d1fac","year":2020},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.429597Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:1234152c6973a64662c9cdd15dda782d957f366fa9ae681cede69da2d9dc5ca4","observation_id":"d22aa95c-0deb-4224-b9d0-4c60011d7d24","resolution":{"observed_at":"2026-08-06T20:57:49.947292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:49.584049Z","title":"Flexcap: Describe anything in images in controllable detail","venue":null,"work_id":"04052fe0-e825-47bb-9d99-2c1a68a74c09","year":2024},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.519290Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:d310d03b395422ddd863ee8066e283c2cb0738c98801deb1eaa0863b46b86919","observation_id":"6cf09b56-11aa-473c-8825-2501cac2aac5","resolution":{"observed_at":"2026-08-06T20:57:49.711358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:49.330757Z","title":"Captioning images taken by people who are blind","venue":null,"work_id":"c7573e9c-b8df-4025-b7fa-a70e8f6825f5","year":2020},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.604022Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:74f8f4856e239ef62cd91fdd6ecc37cbdf0b518f29fd0e99ee9af91446417fcd","observation_id":"9e47b379-494d-45b2-b272-ed203ffb7691","resolution":{"observed_at":"2026-08-06T20:57:49.458330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08718","last_updated":"2022-03-23T19:47:21Z","snapshot_observed_at":"2026-07-06T11:01:02.207193Z","submitted_at":"2021-04-18T05:00:29Z","title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08718","snapshot_observed_at":"2026-08-06T20:57:41.666447Z","title":"Clipscore: A reference-free evaluation met- ric for image captioning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.666447Z"},"links":{"cited_paper":"/paper/2104.08718","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:b72cf3435e5092dbd3741a172354738da9b34505c1bde9c5a22c2fe60b84179a","observation_id":"808936fa-ad98-4e25-ae44-08d21972e8a3","resolution":{"observed_at":"2026-08-06T20:57:41.666447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13912","last_updated":"2024-06-20T01:03:13Z","snapshot_observed_at":"2026-08-12T23:38:48.840611Z","submitted_at":"2024-06-20T01:03:13Z","title":"From Descriptive Richness to Bias: Unveiling the Dark Side of Generative Image Caption Enrichment","version":1},"cited_work":{"arxiv_id":"2406.13912","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.13912","snapshot_observed_at":"2026-08-06T20:57:44.198219Z","title":"From Descriptive Richness to Bias: Unveiling the Dark Side of Generative Image Caption Enrichment","venue":"cs.CV","work_id":"a6b8007b-2e30-40c0-b4e0-13d358a1e881","year":2024},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.739561Z"},"links":{"cited_paper":"/paper/2406.13912","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:13d286f09e7aa1752d2a8b08d2e5dad26af98273d955f6c48ecb869112e4228b","observation_id":"a578ebce-6207-4a05-81cf-045317bd1fcb","resolution":{"observed_at":"2026-08-06T20:57:44.319196Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:49.287426Z","title":"spaCy 2: Natural lan- guage understanding with Bloom embeddings, convolutional neural networks and incremental parsing","venue":null,"work_id":"d8e3858e-91ee-4897-bfdd-e80d8c2387a8","year":2017},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.792838Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:f2d7bcd9a4695b4d2551acc494c3376b8003355e8c1febf4837e6572794ad8f2","observation_id":"704ae124-6352-48a4-b68a-a73072c524a5","resolution":{"observed_at":"2026-08-06T20:57:49.291660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:49.105804Z","title":"Scaling up vision-language pre-training for image captioning","venue":null,"work_id":"86ffa726-d81c-4550-959e-36e33a3558af","year":2022},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.867976Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:b2bba3cb903aff4cd5c27cf5b31fa1d2d32e77b7438101883e9291e8d1c2ea1a","observation_id":"240ab681-5ec4-44e4-8537-a8f1de854a0d","resolution":{"observed_at":"2026-08-06T20:57:49.186220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.957667Z","title":"Noise-aware learning from web-crawled image-text data for image captioning","venue":null,"work_id":"45549866-dfb1-4c6a-815a-e8a226787d00","year":2023},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:41.936237Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:1f4c291a1fb9d259dc9d993791ee2d729c8a787f3e1b56b7651da5da2f26df8f","observation_id":"734de0ac-2c70-48d9-af20-4d474fbdc132","resolution":{"observed_at":"2026-08-06T20:57:49.030536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.805433Z","title":"Imageability-and length-controllable image caption- ing","venue":null,"work_id":"467e3c90-97e2-4c8a-b399-a1540955735b","year":2021},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.044551Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:0fd131beaa51a803bcb27897a73eb2a11c46c070e2e70c25af538415872d42ea","observation_id":"78d3db17-31d7-492a-843b-7380dd714b6f","resolution":{"observed_at":"2026-08-06T20:57:48.874016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.637255Z","title":"Novel dataset for fine-grained image categorization: Stanford dogs","venue":null,"work_id":"f6269712-10ef-4f0d-b09a-1d41e43ab600","year":2011},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.109552Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:39a22c1f2173a25c02bcdad61a1ac55191e7988e362fa445a48eacdb28722d87","observation_id":"71e3f1bb-f1b5-49e5-9a59-50eb91f8f388","resolution":{"observed_at":"2026-08-06T20:57:48.723129Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.491798Z","title":"3d object representations for fine-grained categoriza- tion","venue":null,"work_id":"d35db2b0-2f33-452e-9137-b940e0686500","year":2013},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.170472Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:186334e79fb12462da6cff9d0bef25aab6df54044508749677210fcc0b00e80b","observation_id":"e99e17d9-f1f5-4a99-8950-db1539daf23d","resolution":{"observed_at":"2026-08-06T20:57:48.556463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.359073Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"bc590272-eab6-4693-a747-8e109b414e20","year":2023},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.257722Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:de9082e144f4c84a00351a71a0f6ff80f4ac8955c07e28500190dc0384e11739","observation_id":"542a7d66-22b8-443b-ab42-d781021d9d9c","resolution":{"observed_at":"2026-08-06T20:57:48.420916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.207360Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"9a2de24a-2a4f-4c6f-bef1-0cfda4e61774","year":null},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.330718Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:e691b6294885a4147e1940c6479644ff20644a1d8ff43c4dc4efa0c78f3cb2d7","observation_id":"9f90f9c3-c286-4526-a596-672d18b9fefb","resolution":{"observed_at":"2026-08-06T20:57:48.285100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:42.442960Z","title":"Rouge: A package for automatic evaluation of summaries","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.442960Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:c0966488d8583cc9f696659d18d2fe2c68ef9b04374c90cca50e1b01feb2aa0f","observation_id":"aa863290-8ad4-4580-9f3a-a8b08ebbd7ab","resolution":{"observed_at":"2026-08-06T20:57:42.442960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.062295Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"735f2141-8bc1-42f5-9c54-0ddbcc497f1e","year":2014},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.488770Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:a281bbfd40cbc8fe842053f4992eb86e0e7237688a8fcbd5ca0ddc4166a4f7de","observation_id":"7b9ddc2a-5de5-4bc2-8679-2a349ffbc6c7","resolution":{"observed_at":"2026-08-06T20:57:48.124268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.925764Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"dbc91e58-f6d2-4211-a3e5-f72a9c97f476","year":2024},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.544533Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:f9ad38df8f03346e73e6a9e646c444b004ca7c0ebc195a967eff0960e781f1f0","observation_id":"1eaa68be-ff7f-4a48-a2a1-d1c81cf2cf1b","resolution":{"observed_at":"2026-08-06T20:57:47.990037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.732498Z","title":"Visual instruction tuning","venue":null,"work_id":"2b98ac6d-bf36-4917-94e0-d4cfcd65230a","year":2024},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.619570Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:efc5e76996b23ee9e27d025caf54933759dacca7d31969be9250df0ba854d08a","observation_id":"d468e8f9-ca17-4a9e-af92-4dc22603aff2","resolution":{"observed_at":"2026-08-06T20:57:47.830214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.559940Z","title":"Quark: Controllable text generation with reinforced unlearn- ing","venue":null,"work_id":"80c2d9e8-847f-4d2f-a39f-c2d210145a84","year":2022},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.665570Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:80d09dd0104d02e26b69cbbca6455db29cd0953bfecd2426d87520ab5eec4ba2","observation_id":"97266667-6336-4c80-b398-c6c3d72c1260","resolution":{"observed_at":"2026-08-06T20:57:47.636736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.11848","last_updated":"2020-02-27T00:09:25Z","snapshot_observed_at":"2026-08-10T07:46:14.293421Z","submitted_at":"2020-02-27T00:09:25Z","title":"Analysis of diversity-accuracy tradeoff in image captioning","version":1},"cited_work":{"arxiv_id":"2002.11848","doi":null,"metadata_source":"pith","pith_arxiv_id":"2002.11848","snapshot_observed_at":"2026-08-06T20:57:43.992533Z","title":"Analysis of diversity-accuracy tradeoff in image captioning","venue":"cs.CL","work_id":"27c283a7-10db-4127-955d-e4cb73404ac8","year":2020},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.727730Z"},"links":{"cited_paper":"/paper/2002.11848","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:2fae8851cc80193012e3a91190776870b301e6c1ac53f93e80ac607d2424706a","observation_id":"4b44490d-1b22-4409-9e77-2e74025b1cb4","resolution":{"observed_at":"2026-08-06T20:57:44.081239Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.393210Z","title":"Wordnet: a lexical database for english","venue":null,"work_id":"b37533d9-17f5-450d-9f35-bc88070ccaa2","year":1995},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.793847Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:ffdf91be81cb0bf60b00571e2213dc487e0f213b1367a23b3e03631608fed641","observation_id":"8136eee2-7055-4fe8-9a3a-7ec44d51c14f","resolution":{"observed_at":"2026-08-06T20:57:47.493190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.224685Z","title":"Docci: De- scriptions of connected and contrasting images","venue":null,"work_id":"d37bcaeb-c435-4fd4-8c61-11d26aaca98c","year":2025},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.839594Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:b34dde57cab1bbd18bbd7572efebdb762084aab4f19739cf9ed14434e65841e6","observation_id":"0016f794-ffe0-4ba3-8610-bad21f19fe48","resolution":{"observed_at":"2026-08-06T20:57:47.295276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:42.880815Z","title":"Bleu: a method for automatic evaluation of machine translation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.880815Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:8aa194bf24da51e654485a2f9488ed26029894cdb2bc245ecdfaf569492f8490","observation_id":"fb2e6639-b104-4cf7-bd81-bd65ec352477","resolution":{"observed_at":"2026-08-06T20:57:42.880815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.00247","last_updated":"2021-01-26T12:54:33Z","snapshot_observed_at":"2026-08-11T03:47:32.551562Z","submitted_at":"2020-05-01T07:03:42Z","title":"AdapterFusion: Non-Destructive Task Composition for Transfer Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.00247","snapshot_observed_at":"2026-08-06T20:57:42.933340Z","title":"Adapterfusion: Non- destructive task composition for transfer learning","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.933340Z"},"links":{"cited_paper":"/paper/2005.00247","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:b224fb3e11370c43f04094065540baad347a49819a82abc7c6ca3adbfe6ab1a1","observation_id":"eed8de9d-100e-448d-a8d2-3ae512c021eb","resolution":{"observed_at":"2026-08-06T20:57:42.933340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.074308Z","title":"Connecting vision and lan- guage with localized narratives","venue":null,"work_id":"5e6ec99c-0008-4155-96f1-c4c25b199c08","year":2020},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:42.998228Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:198ebfed57cc1e1e719be9d65d11d94c42a4cff490d67e03a3d1de8d1dc0fd02","observation_id":"5316de52-961b-40c0-b7db-852e7a0fc40a","resolution":{"observed_at":"2026-08-06T20:57:47.149194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:46.774181Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"4bdf7bb1-6417-495b-a374-1ff0b7467549","year":2021},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.052984Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:831cf7b28a56026636d11e1dbbdad874f5605a53c55ec3ea5f1e5995675f5b5a","observation_id":"5618f520-0610-4196-a40a-fb661c7de7a5","resolution":{"observed_at":"2026-08-06T20:57:46.905067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:46.469935Z","title":"Laion coco: 600m syn- thetic captions from laion2b-en, 2022","venue":null,"work_id":"957ea7fd-97ec-4dec-abf6-4fb8613c0346","year":2022},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.118604Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:3d00f261ad3ce6bcf250db9f7e6c22027c8236d0f252ee2e630d3e8a5315b36b","observation_id":"88721885-833b-4233-befa-73387514739f","resolution":{"observed_at":"2026-08-06T20:57:46.655352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T20:57:43.172375Z","title":"Llama 2: Open foundation and fine-tuned chat models.arXiv preprint arXiv:2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.172375Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:7b1821ac3dc9728b504fb8c34c1c3856603b1f7d0aa4ae84a7f0da61aed53ca9","observation_id":"4de8f8cd-bf19-4567-b6f5-90febde4c629","resolution":{"observed_at":"2026-08-06T20:57:43.172375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:46.163186Z","title":"Cider: Consensus-based image description evalua- tion","venue":null,"work_id":"33c1f24c-b878-46bf-85e6-74a643c934ce","year":2015},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.233214Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:d57a7f128439aa98f90865958bd4876f6dd05abd1743e1e729ff2d0528e7195a","observation_id":"93576c59-dea6-48f1-82c6-a92beb03628d","resolution":{"observed_at":"2026-08-06T20:57:46.295126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.07904","last_updated":"2022-03-16T18:52:36Z","snapshot_observed_at":"2026-08-13T17:52:41.506368Z","submitted_at":"2021-10-15T07:35:58Z","title":"SPoT: Better Frozen Model Adaptation through Soft Prompt Transfer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.07904","snapshot_observed_at":"2026-08-06T20:57:43.319076Z","title":"Spot: Better frozen model adaptation through soft prompt transfer.arXiv preprint arXiv:2110.07904, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.319076Z"},"links":{"cited_paper":"/paper/2110.07904","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:b84c969e1a706808b42c4ab22e2e6bbed4013c35e34f9960f43fcb7537e4b3cb","observation_id":"d4112cec-91a4-4d86-b55a-f3fb3cce6801","resolution":{"observed_at":"2026-08-06T20:57:43.319076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:45.885937Z","title":null,"venue":null,"work_id":"90039d61-d178-421c-bec9-49a3cbe3687c","year":2011},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.370694Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:c9389c2de7efbb7e74c979cccb49f0c2bacdc9eba4de1ebd28fee34d87931d75","observation_id":"371328cd-aa87-47ea-abd7-b37cfe608627","resolution":{"observed_at":"2026-08-06T20:57:46.011132Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:45.618894Z","title":"Controllable image captioning via prompting","venue":null,"work_id":"cd2a2f39-53b5-4bba-90a0-98dc9bbb1487","year":2023},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.418905Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:b0a4c584622b8f2f7dcdf1b5448dfe42cf9e7ecd5c08334cf4ac1499e16b6e8d","observation_id":"bd45be43-a0ea-4c92-bf7e-bf57b20515ce","resolution":{"observed_at":"2026-08-06T20:57:45.742872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-06T20:57:43.481086Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.481086Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:6dab37c535c5ea06f23d75b1fc6a7c79735b9692a3cc354c32c315b1654b04b9","observation_id":"cb589c28-6c8f-4190-9d92-a3ea9585cc13","resolution":{"observed_at":"2026-08-06T20:57:43.481086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:45.360354Z","title":"On diversity in image captioning: Metrics and methods","venue":null,"work_id":"e65fa2ae-b0bf-4804-acf5-bc0b22adf19e","year":2020},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.527334Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:fd608c4b3264c1c48aa2d74d428121c842bbb11b61209a403ff1412362a0c2a7","observation_id":"1f7af472-dad7-43a5-98e0-e99d3b1234bc","resolution":{"observed_at":"2026-08-06T20:57:45.467822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:45.086824Z","title":"Model soups: averaging weights of multiple fine-tuned models improves accuracy without increasing in- ference time","venue":null,"work_id":"9a133257-ad03-47ae-a0ef-51882245d478","year":2022},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.582119Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:9743f74e7c73a31f2656f0d55535e6fcecbb882ce2809d64f6d9d2bf9685499e","observation_id":"c0d0ba97-855c-44fb-a359-3db2b8f9ff63","resolution":{"observed_at":"2026-08-06T20:57:45.173175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:43.643606Z","title":"xgen-mm (blip-3): A family of open large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.643606Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:1929ed21c76557b83e57bea16d95bdb67e7b69ca930d611dc6e4dabdf86294e9","observation_id":"a3a58286-6d6d-4192-9d4d-cba0c6cb14ff","resolution":{"observed_at":"2026-08-06T20:57:43.643606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-06T20:57:43.704897Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.704897Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:fa0f2a48b0d7311ed60ce1b98f2aed538f43559395aeb2d2e7c8595e8969ad64","observation_id":"f27bd746-0aee-4691-8b65-77d1e0726134","resolution":{"observed_at":"2026-08-06T20:57:43.704897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:44.812842Z","title":"Regularization and variable se- lection via the elastic net","venue":null,"work_id":"359f6f7c-302d-41f9-9fa0-5a124615e829","year":null},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.754324Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:08f69a755a200bc427817d9da9cdcee5a10b2ef0620825cc7090587027f99dec","observation_id":"54e0c28a-1b44-4e36-bf23-211828c73dde","resolution":{"observed_at":"2026-08-06T20:57:44.936216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:44.463129Z","title":"image”, “side","venue":null,"work_id":"249338d9-65e9-46a0-b678-2f9c1feb7522","year":2000},"citing_paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning","version":1},"reference_index":2005,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:43.814957Z"},"links":{"citing_paper":"/paper/2507.01409"},"observation_digest":"sha256:5654205bd21553c59e95dddbf2e99de23be7181ddb14da7199ed592172b2da1d","observation_id":"db19b62a-88e5-44e8-955f-06ab7ca7ac7c","resolution":{"observed_at":"2026-08-06T20:57:44.637445Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.01409","last_updated":"2025-07-02T07:02:45Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T01:50:59.924887Z","submitted_at":"2025-07-02T07:02:45Z","title":"CaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":2,"verified_fuzzy":29},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 0 inbound Pith citation observations for arXiv:2507.01409."}