{"as_of":"2026-08-18T22:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:adf6a924e6b3871199545b965a31f4dd655eb98d4a6ad828c667ba2c808ec01b","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:36:43.741522Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.19027/citation-record","integrity":"/paper/2607.19027/integrity","json":"/paper/2607.19027/citation-record.json","paper":"/paper/2607.19027"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T15:36:43.520012Z","title":"arXiv preprint arXiv:2303.08774 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.520012Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:131fb8824900558583386b88ec1f57a10fb98935cbd79dda99706cbeaae22cfb","observation_id":"4b0d7f98-8450-448d-8bc1-d41791981ea2","resolution":{"observed_at":"2026-08-15T15:36:43.520012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.538679Z","title":"IEEE transactions on pattern analysis and machine intelligence38(10), 2069–2081 (2015)","venue":null,"work_id":"7f07595b-b283-4ad7-b960-ee47b88d19b3","year":2015},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.525234Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:9fcc77d8c826f3b264ef82d71d0ed0bb1cc03ecc803006735f1764c267564226","observation_id":"81d764f2-fc39-496a-9437-9e93a34ea719","resolution":{"observed_at":"2026-08-15T15:36:44.543613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.529664Z","title":"In: Proceedings of the ieee conference on computer vision and pattern recognition","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.529664Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:c5ff56a63de2e54f4a257b8e66be95c126be037f4a5a6a8b8617eadc8bd463f4","observation_id":"1165ce58-a597-4f99-8d30-4b62dd0c23b1","resolution":{"observed_at":"2026-08-15T15:36:43.529664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.515853Z","title":"In: Proceedings of the IEEE/CVF International Conference on Computer Vision","venue":null,"work_id":"2f8f4ff4-6dc0-421f-bd77-e8a72bc18d4d","year":2025},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.534180Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:e9e0749783fa0eb7defd3331165faa32bac0dd30ce595507dac42e642e37ea5e","observation_id":"fb53f648-76a4-4195-ae7d-2e50191e7823","resolution":{"observed_at":"2026-08-15T15:36:44.520261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09478","last_updated":"2023-11-07T18:25:48Z","snapshot_observed_at":"2026-08-17T13:04:04.064087Z","submitted_at":"2023-10-14T03:22:07Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.09478","snapshot_observed_at":"2026-08-15T15:36:43.538839Z","title":"arXiv preprint arXiv:2310.09478 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.538839Z"},"links":{"cited_paper":"/paper/2310.09478","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:be40790cf68e3bfbad79ed436e14e76ee344a0dd49667f6ded90debc57663105","observation_id":"17eaca40-aa40-4c5c-a70b-b4cd4a5171bd","resolution":{"observed_at":"2026-08-15T15:36:43.538839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21439","last_updated":"2024-09-25T06:14:03Z","snapshot_observed_at":"2026-08-17T00:27:44.606608Z","submitted_at":"2024-07-31T08:43:17Z","title":"MLLM Is a Strong Reranker: Advancing Multimodal Retrieval-augmented Generation via Knowledge-enhanced Reranking and Noise-injected Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21439","snapshot_observed_at":"2026-08-15T15:36:43.543626Z","title":"arXiv preprint arXiv:2407.21439 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.543626Z"},"links":{"cited_paper":"/paper/2407.21439","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:fac2d68c43cee0ed54a493fdb3d742fe1377acd8dfb48f983f7e68397f5324ad","observation_id":"01479933-8141-4996-a352-b434b846aa2e","resolution":{"observed_at":"2026-08-15T15:36:43.543626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.501296Z","title":"In: Transfer Learning for Natural Language Processing Workshop","venue":null,"work_id":"b42e09e6-ee2f-4b30-a4b1-ec5a0453b275","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.548627Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:2db2cf727b8226c745e7eb7af5159231704a6ad323d080f87dffc1c2ba986608","observation_id":"06583ffb-0d9d-46c3-b6ca-6fc17ed7c199","resolution":{"observed_at":"2026-08-15T15:36:44.505789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.552914Z","title":"In: Proceedings of the IEEE international conference on computer vision","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.552914Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:5d8613a859b0d11cb73ab5ab1b58d55ced82ea898806972c109bac6f1f8bbc51","observation_id":"7c20b64c-2e53-4795-8888-782c48fb8ef1","resolution":{"observed_at":"2026-08-15T15:36:43.552914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T15:36:43.557069Z","title":"arXiv preprint arXiv:2407.21783 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.557069Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:1cc8dcdde7e7e020f8f5f1e6cc7d792adbcadfdf431bf8af931123c0945125cf","observation_id":"4ea5e12b-9a68-412a-a4d5-fa8852ae87d6","resolution":{"observed_at":"2026-08-15T15:36:43.557069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.561208Z","title":"In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.561208Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:542c370ed885b5820d72a4822d839afbd3d758e329595d715612fb3ec5812ad9","observation_id":"6d9bfb05-b27a-4576-80e1-12285a40d541","resolution":{"observed_at":"2026-08-15T15:36:43.561208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.469152Z","title":"In: Companion Proceedings of the ACM on Web Conference 2025","venue":null,"work_id":"822f6983-3271-4612-9638-0a9cb7c26b9d","year":2025},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.565415Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:a05ac65a05d40c8cef5849c00dc2b8ad2c179b47a594d5c58d7a3c156ec31cea","observation_id":"8909095e-4af8-442b-b03b-3997bf2ecfd6","resolution":{"observed_at":"2026-08-15T15:36:44.473518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.455115Z","title":"In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":"c98a2312-be30-4957-af23-578dfbdb6e8f","year":2022},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.569869Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:02d4d5d73ae3b4c99d987d896c108eaf695af9a91eb2fb72e006990150097e71","observation_id":"48fae83c-08db-4daa-a523-e5237b1a5391","resolution":{"observed_at":"2026-08-15T15:36:44.459558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.440787Z","title":"In: Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision","venue":null,"work_id":"f74ac423-714f-46e4-ad91-bfa5d25f1cd6","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.573962Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:0cd1a84dbc0f35ce00f8053a0b3cd817f61dbc8d601e1cd0ba5d896a936e1519","observation_id":"2c6dac03-e002-4d76-ab60-2f5cf91eb4c3","resolution":{"observed_at":"2026-08-15T15:36:44.445734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.426864Z","title":"In: Proceedings of the 31st ACM International Conference on Multimedia","venue":null,"work_id":"c0d0456f-3fae-4da5-832d-f818ff6039d1","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.578073Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:44cf96f5fefea80263ebac153c3e0643a2876544feed9ea70f993eeb7209f65e","observation_id":"ff9366fa-c590-43f4-834c-ce9051dfbd9f","resolution":{"observed_at":"2026-08-15T15:36:44.431462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.413010Z","title":"In: Proceedings of the IEEE international conference on computer vision","venue":null,"work_id":"0f13722a-1f9a-40bb-b61a-7eb0762c6021","year":2017},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.582240Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:15ef6566a5bdcc485ed3a847a9841eb02219657f150fa2f2932f19f1b68f90d6","observation_id":"ccd48965-8dc0-41c5-bb6e-1418f31100f6","resolution":{"observed_at":"2026-08-15T15:36:44.417487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17428","last_updated":"2025-02-25T00:35:18Z","snapshot_observed_at":"2026-08-14T08:11:36.232487Z","submitted_at":"2024-05-27T17:59:45Z","title":"NV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.17428","snapshot_observed_at":"2026-08-15T15:36:43.586285Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.586285Z"},"links":{"cited_paper":"/paper/2405.17428","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:e23294114b1eb3a7877e298883b8aa0214d5bd3c8bd9d76140751e8e74ece3e7","observation_id":"38bc2419-bad6-47e1-8742-c7bf21e14dd1","resolution":{"observed_at":"2026-08-15T15:36:43.586285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.590519Z","title":"Advances in Neural Information Processing Systems34, 11846–11858 (2021)","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.590519Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:46c30aa85d6222ef84646a8c586188cac542a75507d4c7e0b03bde89140d757a","observation_id":"c1961ef8-7a70-4129-a569-b51d1ca93455","resolution":{"observed_at":"2026-08-15T15:36:43.590519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.390286Z","title":"In: Computer Vision–ECCV 2020: 16th European Con- ference, Glasgow, UK, August 23–28, 2020, Proceedings, Part XXI 16","venue":null,"work_id":"53949750-77cb-49dd-a399-5fb379be0a3d","year":2020},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.594794Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:a0d664239bcabeb585ed30eef45eccee4cc5d89183ae2e68c87430c518ba5c51","observation_id":"917bc589-0ea3-4707-8b4f-caada90ec9ff","resolution":{"observed_at":"2026-08-15T15:36:44.394828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.599067Z","title":"In: International conference on machine learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.599067Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:f83f529ec7ab0d8ee161625bd2d1d961c4ef97bd0f962e41c24f56f875ab7452","observation_id":"52157333-27b5-4100-be62-70d72b316826","resolution":{"observed_at":"2026-08-15T15:36:43.599067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.603476Z","title":"In: International confer- ence on machine learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.603476Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:fa11750e3b3f10cdf2fb20d55d4f39c611f39cfdff3172955a2c05c10a8c9a29","observation_id":"491fceb6-8aaf-4055-b35d-df1c1065415e","resolution":{"observed_at":"2026-08-15T15:36:43.603476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02571","last_updated":"2025-02-22T05:33:26Z","snapshot_observed_at":"2026-08-16T13:03:00.866176Z","submitted_at":"2024-11-04T20:06:34Z","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02571","snapshot_observed_at":"2026-08-15T15:36:43.607636Z","title":"arXiv preprint arXiv:2411.02571 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.607636Z"},"links":{"cited_paper":"/paper/2411.02571","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:74b5069c929ce2d36da3ef9a50547da97767f99eb6641615fafde8f4430c3835","observation_id":"bec05775-74d0-4105-bb14-853c3bbb394c","resolution":{"observed_at":"2026-08-15T15:36:43.607636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.357205Z","title":null,"venue":null,"work_id":"f33779a1-5bf3-45c2-9d16-69031d872384","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.612074Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:18b9093ab754247bcf3d127b0b0103f335bd88bab5d328902bd091c77843aeba","observation_id":"ba80e7f2-3d93-43b6-8b94-6f54381d4e28","resolution":{"observed_at":"2026-08-15T15:36:44.361394Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.343127Z","title":"In: Proceed- ings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"70cdce02-08d1-4c7a-ba30-c084fc2b804f","year":2022},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.616116Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:57f8c10efe6b6f25f97eb8c9cb0c214611a2287c4ba70148b15607588c9f35d1","observation_id":"c0883dcb-bc3c-4531-9527-f9c764c976b7","resolution":{"observed_at":"2026-08-15T15:36:44.347859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.329283Z","title":"In: Proceedings of the AAAI conference on artificial intelligence","venue":null,"work_id":"6e34d64f-70f1-40a3-bf61-6e1911794263","year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.620223Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:fd1b52f20af97da67158296078a6ab9259c6fb3e40d5d5e26fec9684babe2128","observation_id":"88b171f6-72a1-433b-8110-f4c40231fccf","resolution":{"observed_at":"2026-08-15T15:36:44.333718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.315486Z","title":"In: Proceedings of the IEEE/CVF winter con- ference on applications of computer vision","venue":null,"work_id":"29ca9fea-ca1c-41ff-a8db-03eb0101b3c3","year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.624351Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:ce10a87d69263e50bd4e7266e888af20340e6da8e8130d32d95b25b6e968b90e","observation_id":"5f6eed93-67b6-4468-8950-b9d26f0881a7","resolution":{"observed_at":"2026-08-15T15:36:44.319867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.301068Z","title":"In: Proceedings of the IEEE/CVF International Conference on Computer Vision","venue":null,"work_id":"411b2e55-e867-47d7-8a14-b86f7504fab4","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.628620Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:d800ad49f74483a1c6fcd8cb1f89060a64adc3d5133b1df7033e9127979c021b","observation_id":"9dce1a8f-0d99-460e-a489-8804b779e0c6","resolution":{"observed_at":"2026-08-15T15:36:44.305513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.632812Z","title":"In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.632812Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:dc1e76bda66eaafede5f330fe6f62b93f4ef7b6a9dc87e7ecf27dbae450498f1","observation_id":"e7f2cc77-82d9-4959-8be6-98fcefb04bad","resolution":{"observed_at":"2026-08-15T15:36:43.632812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-15T15:36:43.637093Z","title":"arXiv preprint arXiv:2306.05424 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.637093Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:83f07f0da898b116beb8ef68e5211f2614f6a4e06677cdf15930645ecb3ca5b8","observation_id":"985bc98a-ec0f-4a5f-8329-40dbe8391d95","resolution":{"observed_at":"2026-08-15T15:36:43.637093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.277720Z","title":"In: Proceedings of the IEEE/CVF international conference on computer vision","venue":null,"work_id":"741a20b2-ccb6-4ee7-b453-6e8d21504fe6","year":2021},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.641675Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:2e14531978533ba7f1aa394a3340a377d582a6dd58d24fed1fe9dcf79b4f5bb2","observation_id":"817683a9-2c3c-4ec2-af6a-c7b381425e6f","resolution":{"observed_at":"2026-08-15T15:36:44.282770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-17T13:03:40.359628Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-15T15:36:43.645788Z","title":"arXiv preprint arXiv:2304.07193 (2023) Mitigating Modality and Language-Style Gaps for ZMR 17","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.645788Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:ad30458a94b3d169da6c6e8ae61d975a1e2ed8c413ef2a2a560ef80abc0ce1b7","observation_id":"1b1bc65b-2b77-422d-ab35-f184b960cc8c","resolution":{"observed_at":"2026-08-15T15:36:43.645788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-15T15:36:43.650352Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.650352Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:18478bcb8d31481b30275bfbcfe90c69ad3f3b9888dca0e1df5c618afb722553","observation_id":"964a9f15-e6f8-40b0-a61b-fd35d164912d","resolution":{"observed_at":"2026-08-15T15:36:43.650352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.654562Z","title":"In: International conference on machine learning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.654562Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:337952b4c9ff274e1867b2ab7d59b5fe55c3dba92898758702d8d908f6e297a2","observation_id":"ee81fede-8b39-4421-959c-a910c1ad24e9","resolution":{"observed_at":"2026-08-15T15:36:43.654562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.10084","last_updated":"2019-08-27T08:50:17Z","snapshot_observed_at":"2026-08-14T05:02:11.716316Z","submitted_at":"2019-08-27T08:50:17Z","title":"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.10084","snapshot_observed_at":"2026-08-15T15:36:43.658646Z","title":"arXiv preprint arXiv:1908.10084 (2019)","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.658646Z"},"links":{"cited_paper":"/paper/1908.10084","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:61409d554d221b404bcea74c0082cf5be236161c97e7c944eb39cadab1f6a086","observation_id":"4070d86d-ac9d-4bbe-98b5-df4f2961ad72","resolution":{"observed_at":"2026-08-15T15:36:43.658646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.663075Z","title":"In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.663075Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:c4f2abd38d5f37850b2a07c61fd238246c65182bfc254511d68273f901416895","observation_id":"3885910a-9c35-4e85-ba94-1a58afb362de","resolution":{"observed_at":"2026-08-15T15:36:43.663075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.245587Z","title":"In: 2007 IEEE conference on computer vision and pattern recognition","venue":null,"work_id":"e5acd231-2b70-495d-8266-ca88e5c3c54d","year":2007},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.667296Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:5b16ec7044bf6fc5fc1f6eb7e8152066b669497f8ea36a18bda2f6c96daba471","observation_id":"fe63f298-62a3-4347-b642-84d9989c3092","resolution":{"observed_at":"2026-08-15T15:36:44.249828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09542","last_updated":"2024-12-28T06:20:54Z","snapshot_observed_at":"2026-08-16T15:39:19.922047Z","submitted_at":"2023-04-19T10:16:03Z","title":"Is ChatGPT Good at Search? Investigating Large Language Models as Re-Ranking Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09542","snapshot_observed_at":"2026-08-15T15:36:43.671742Z","title":"arXiv preprint arXiv:2304.09542 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.671742Z"},"links":{"cited_paper":"/paper/2304.09542","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:40fbb28f96b41a8756c19ac3708c80ecf08fdc200337c34818d2ef99c5bced74","observation_id":"aeef787e-e90c-4f17-ba08-f84ee41798f6","resolution":{"observed_at":"2026-08-15T15:36:43.671742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.230948Z","title":"IEEE Signal Process- ing Letters31, 521–525 (2023)","venue":null,"work_id":"b4c70205-a2ca-4702-8aeb-95c2cfa03dd3","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.676102Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:7c04134a9b1f7338c386986be76937c6c1de8eb795f2bac1aef8d6db6fcc6078","observation_id":"e7a378b7-d7a4-46da-8100-79540e1d8f22","resolution":{"observed_at":"2026-08-15T15:36:44.235972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-15T15:36:43.680326Z","title":"arXiv preprint arXiv:2302.13971 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.680326Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:a13a503db2cc2e4f43ab8140147abef8d8e282e710d3a0fe1a525aa019fc9383","observation_id":"ba3447de-773c-4e87-a323-48b9c103cb43","resolution":{"observed_at":"2026-08-15T15:36:43.680326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-15T15:36:43.684741Z","title":"arXiv preprint arXiv:2307.09288 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.684741Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:0ccd60e681ebb78f571314987332b97b27365a3fedcb4439d3758916351505a0","observation_id":"fc148b43-b398-4672-b504-3d49e315d160","resolution":{"observed_at":"2026-08-15T15:36:43.684741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.216778Z","title":"In: Proceedings of the 30th ACM international conference on multimedia","venue":null,"work_id":"77eb1f2b-5447-4b79-a8cc-e183cb8cacc6","year":2022},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.689108Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:f2ed5c620c854326bc84a14fc52dd19edb8719d3ae69b4b49db5c68d7113a1cc","observation_id":"dde679b1-a34e-43b6-9aab-ef76ab99cb14","resolution":{"observed_at":"2026-08-15T15:36:44.221162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.202866Z","title":null,"venue":null,"work_id":"02c64229-22ba-4722-90bb-1c8ddb6b834f","year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.693394Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:3fc9adbc16f021861809939cde1bedd742fd01b29b16e4d8dc88b33a5adf78c3","observation_id":"a7e7a4fc-8a83-44fa-9ed2-321d659e88c4","resolution":{"observed_at":"2026-08-15T15:36:44.207118Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.189078Z","title":null,"venue":null,"work_id":"2e20240e-c8cc-4d63-b540-e6c4c52285d6","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.697813Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:586550a4123b8e3b92e566bffbe8f60198136c0b7928d1ceaaaa98a48c7ed508","observation_id":"8c9e8033-36f6-40d0-bfc4-b6ebeef793ed","resolution":{"observed_at":"2026-08-15T15:36:44.193397Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.12499","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.879634Z","title":"arXiv preprint arXiv:2505.12499 (2025)","venue":null,"work_id":"a226a92a-6ec6-4cea-985e-5fad5e56732a","year":2025},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.702089Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:dc2d37bf24ca2d9ef5fc7254e52a5f054e5a896521d87a94f26282da60154bdb","observation_id":"18060491-b26b-40be-a32d-3d3f151e0036","resolution":{"observed_at":"2026-08-15T15:36:43.889388Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3390/app14051894","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:43.773410Z","title":"Applied Sciences14(5), 1894 (Feb 2024).https: //doi.org/10.3390/app14051894,http://dx.doi.org/10.3390/app14051894","venue":null,"work_id":"ea03b896-1483-48b0-934c-e4bddea4c806","year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.706580Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:18a08ec948339e2d5f8d5beba87f23198aacac19700255d170165a63b43a9bd2","observation_id":"21f00db2-dcf1-4262-a5cb-35ce818a5835","resolution":{"observed_at":"2026-08-15T15:36:43.779138Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.174845Z","title":"In: 2024 International Joint Conference on Neural Networks (IJCNN)","venue":null,"work_id":"3b3d07d2-e15a-4178-afec-f1df62fe8449","year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.711058Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:1291afa310976e0559257084d476c40c823f66585584e55893bd97a32a6e6e4d","observation_id":"e75cd50a-f651-475c-a2e1-5107e5c52926","resolution":{"observed_at":"2026-08-15T15:36:44.179586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.159746Z","title":"In: Proceedings of the AAAI Conference on Artificial Intelligence","venue":null,"work_id":"ea660982-2f7a-4925-8d1e-522223dfb199","year":2025},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.715227Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:a4e79962e2f2f7a5b5e51f104df3cf486d1141dcf783ad96011edb251a79c3d5","observation_id":"2da1ffdf-739f-4b8e-a8aa-7027768a9e1e","resolution":{"observed_at":"2026-08-15T15:36:44.164682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.144514Z","title":"In: 2024 IEEE International Conference on Multimedia and Expo (ICME)","venue":null,"work_id":"0189d790-9e2e-486b-82b4-22aa3429b508","year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.719475Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:63c69ad9117fc484e405356f5b1c655b6286a0726463fd5ad79d416bbbe34554","observation_id":"3ef94fb4-f7ed-4b78-8323-5f9c59d4fde2","resolution":{"observed_at":"2026-08-15T15:36:44.149098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.130300Z","title":null,"venue":null,"work_id":"af7ae2b1-4487-488f-8d7b-ed43f6262246","year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.723769Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:333e19da11165dbe354be1587c58bad794c191aedce7ceb42c0b1b6f307a2ffd","observation_id":"c16f0a09-0bf4-4e39-a3be-70e64d8afda2","resolution":{"observed_at":"2026-08-15T15:36:44.134680Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-08-13T15:50:38.254753Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-15T15:36:43.728051Z","title":"arXiv preprint arXiv:2306.02858 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.728051Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:965abe76a453d80ead1fd8b2d7efdd63b32a24f1b155c1dfc4dc678e2e5e1a14","observation_id":"186125b7-6443-4fcd-a5f3-ea2755705896","resolution":{"observed_at":"2026-08-15T15:36:43.728051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.115510Z","title":"In: European Conference on Com- puter Vision","venue":null,"work_id":"cded7b19-2f1f-40a6-a42d-b57443fd535d","year":2024},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.732809Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:b9d1acde23a171b61ccfd68b928f125ea5de8eb2820850c513d599174ed2ebeb","observation_id":"898e4125-309c-427e-b042-bca895939769","resolution":{"observed_at":"2026-08-15T15:36:44.120202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.101160Z","title":"In: Proceedings of the AAAI Conference on Artificial Intelligence","venue":null,"work_id":"54a379c8-3a27-44f8-aca5-bc922fa16dfc","year":2022},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.737013Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:cd17d185fa08e1b2e72dfafe48fd81a1b783eefcd546951d7d7acc1e0730fe04","observation_id":"64fdb272-ffe7-4fd1-a822-88bc0b59c2d4","resolution":{"observed_at":"2026-08-15T15:36:44.105783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:36:44.085932Z","title":"Yes” and “No","venue":null,"work_id":"ebf0f156-7f8a-4fda-9ca8-146ab09eb7fb","year":2022},"citing_paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T15:36:43.741522Z"},"links":{"citing_paper":"/paper/2607.19027"},"observation_digest":"sha256:49f6f2ec4c17480ff94cc6f6af8b99dea6797b537dc0dfd721642a24e9bb834a","observation_id":"4ad58414-8e77-4558-b72f-573a31fd3bfa","resolution":{"observed_at":"2026-08-15T15:36:44.090595Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.19027","last_updated":"2026-07-21T12:16:29Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T03:25:11.484281Z","submitted_at":"2026-07-21T12:16:29Z","title":"Mitigating Modality and Language-Style Gaps for Zero-Shot Video Moment Retrieval"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":27,"verified_exact":2,"verified_fuzzy":22},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 0 inbound Pith citation observations for arXiv:2607.19027."}