{"as_of":"2026-08-15T09:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:35e2c529415a871c2130c90604b21511c71d8a1b7f0e2d5c564cf0997fb3a433","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:59:13.889276Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.00891/citation-record","integrity":"/paper/2506.00891/integrity","json":"/paper/2506.00891/citation-record.json","paper":"/paper/2506.00891"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:17.228549Z","title":"Dual alignment unsupervised domain adaptation for video-text retrieval,","venue":null,"work_id":"4dc1af16-61be-44f3-ad69-a5361c35ab96","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.514757Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:f9a37bdd41fe72a0077b820bdc0ea01656d9f056e6aa4efec4ad59b6ffd8ae17","observation_id":"33c92fad-9261-4d56-bf0d-d082fe73b9ae","resolution":{"observed_at":"2026-08-07T11:59:17.273131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:17.094410Z","title":"Uncertainty-aware alignment network for cross-domain video-text retrieval,","venue":null,"work_id":"1d49577e-c1d3-4339-a995-279c126ec908","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.647809Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:4b7f7e14759975114a383c0b35266cec651269d151be3814dffc656299481842","observation_id":"d5f6b0bd-f6da-4e68-bdfd-2a5e15af7dee","resolution":{"observed_at":"2026-08-07T11:59:17.140821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.947162Z","title":"Uatvr: Uncertainty-adaptive text-video retrieval,","venue":null,"work_id":"047000c4-d9c2-44bc-b272-87e9e2f96715","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.791831Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:e037c4d14eb62e4f00a8fb127236af48fa42b406eea0b92023670b9a985e258d","observation_id":"c0d04294-2bec-40df-8d99-81f1f7d8065a","resolution":{"observed_at":"2026-08-07T11:59:17.003651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.824756Z","title":"Unified coarse-to-fine alignment for video-text retrieval,","venue":null,"work_id":"8a517748-6fde-498e-8931-9188dcb79174","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:11.897250Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:8597fcd4fe05c08c2d5401eabd5ae9392aa54a96340aaabdd99a16171c3eb857","observation_id":"064c0b14-3e82-4499-95d4-2c61e04c105b","resolution":{"observed_at":"2026-08-07T11:59:16.880777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.694867Z","title":"Text is mass: Modeling as stochastic embedding for text-video retrieval,","venue":null,"work_id":"2779c44e-4f73-4853-a031-d4f52e710111","year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.017241Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:55faed56d7d3d1c940028f2c084be30442a90e4ad488d8ad5791e46f633a5894","observation_id":"faf0d322-6db3-4479-aed4-332607f4d5e0","resolution":{"observed_at":"2026-08-07T11:59:16.741366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.535933Z","title":"Dgl: Dynamic global-local prompt tuning for text-video retrieval,","venue":null,"work_id":"b2d4a8a7-edd6-40ca-a93b-3034cf2e9794","year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.141183Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:e4cd1cece995ae8c5364c922c14dcb5cbf91d2c6199528e83356bce6223f6f15","observation_id":"05f0067b-d255-4df0-a344-77864ccc298c","resolution":{"observed_at":"2026-08-07T11:59:16.630962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12218","last_updated":"2023-05-20T15:48:47Z","snapshot_observed_at":"2026-08-13T11:38:11.686344Z","submitted_at":"2023-05-20T15:48:47Z","title":"Text-Video Retrieval with Disentangled Conceptualization and Set-to-Set Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12218","snapshot_observed_at":"2026-08-07T11:59:12.242279Z","title":"Text-video retrieval with disentangled conceptualization and set-to-set alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.242279Z"},"links":{"cited_paper":"/paper/2305.12218","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:0e232afb250d47e6efe7a7f65d4d76e460a810061fed461ae33a31155e91dc4b","observation_id":"e309d534-57c2-4926-abbc-243c2eda77c7","resolution":{"observed_at":"2026-08-07T11:59:12.242279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.386619Z","title":"Multi-feature graph attention network for cross-modal video-text retrieval,","venue":null,"work_id":"10e25cac-2b91-4658-bb62-63109f3fee83","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.298946Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:a57545bd6407599518f59ce067969cf8a3953ca0329e0ea83308150a1d3ab3e6","observation_id":"eebc9fa1-ac43-411b-bd52-8801a0cc2a57","resolution":{"observed_at":"2026-08-07T11:59:16.458283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.258588Z","title":"What matters: Attentive and relational feature aggregation network for video-text retrieval,","venue":null,"work_id":"f0e1cecd-93ce-40f3-87b4-93b296af548d","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.378562Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:04b0d3be029ab9194665eca9939c2fcdc5aab8c2ea54cb4cad862e2d4d632ee6","observation_id":"b75f6d79-b864-4502-9375-3ff3917229ab","resolution":{"observed_at":"2026-08-07T11:59:16.305160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:16.038081Z","title":"Partially relevant video retrieval,","venue":null,"work_id":"152ebeca-424a-4039-98db-d76645aa8089","year":2022},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.433909Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:0f28b8c7de0f8744d111db07cf040adb43630f46855d396aab5d9d1a25f3d46d","observation_id":"151fd767-ce02-453d-87bd-f3f9fcb4e770","resolution":{"observed_at":"2026-08-07T11:59:16.151729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.13824","last_updated":"2024-05-22T16:55:31Z","snapshot_observed_at":"2026-08-13T00:01:59.342730Z","submitted_at":"2024-05-22T16:55:31Z","title":"GMMFormer v2: An Uncertainty-aware Framework for Partially Relevant Video Retrieval","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.13824","snapshot_observed_at":"2026-08-07T11:59:12.507772Z","title":"Gmmformer v2: An uncertainty-aware framework for partially relevant video retrieval,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.507772Z"},"links":{"cited_paper":"/paper/2405.13824","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:425ad8d147cc0b72da9f0a74cd6a1e06df2cd52bd276398005a5468fd94c31db","observation_id":"7ee8147d-027e-4773-884b-a4288312a559","resolution":{"observed_at":"2026-08-07T11:59:12.507772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.939203Z","title":"Gmmformer: Gaussian-mixture- model based transformer for efficient partially relevant video retrieval,","venue":null,"work_id":"815f9c96-25b6-48e1-8cd2-eed3e8ba24e9","year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.633308Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:10d0c3144e7e38923c7ab887532fa2ed1dc9dd7746c23024a29b6cf30c5ba256","observation_id":"1a1dad76-254a-4336-9483-a13d861ed2a3","resolution":{"observed_at":"2026-08-07T11:59:15.982314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.767329Z","title":"Dual learning with dynamic knowledge distillation for partially relevant video retrieval,","venue":null,"work_id":"aa79eb66-f520-44c6-8d80-81d51960a944","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.766279Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:4524fb49e02a91e503952c211e39cc59c1132d3c49f3656ed6675def50f9268a","observation_id":"c9668524-38d6-4679-b120-fbb726d98092","resolution":{"observed_at":"2026-08-07T11:59:15.835647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02483","last_updated":"2025-02-12T09:39:06Z","snapshot_observed_at":"2026-08-12T22:51:46.101681Z","submitted_at":"2024-09-04T07:20:01Z","title":"TASAR: Transfer-based Attack on Skeletal Action Recognition","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02483","snapshot_observed_at":"2026-08-07T11:59:12.905371Z","title":"Tasar: Transfer-based attack on skeletal action recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.905371Z"},"links":{"cited_paper":"/paper/2409.02483","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:b82c8508ace520f153422cbfea2451acec76fde9d69e953a00ad89ffc10e51a9","observation_id":"bfc91068-4c81-4b87-a076-1cf354b49220","resolution":{"observed_at":"2026-08-07T11:59:12.905371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.616027Z","title":"Progressive event alignment network for partial relevant video retrieval,","venue":null,"work_id":"01641c12-3b71-46a3-b8cf-a5c6e0db6d50","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:12.970005Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:db1c90d7565f108af4f2527086a6e9662ed6e25b13f97f52a98589696d521b4f","observation_id":"401ed5bd-d328-46e0-afbc-448bc0f3596e","resolution":{"observed_at":"2026-08-07T11:59:15.707201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-15T03:44:13.782919Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-07T11:59:13.027251Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.027251Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:6bc8aba16b914cbc6e5ebf50518d64b944b3d1431e2b4c34fc8d0cbb7e5c08af","observation_id":"2fc4f736-9708-4578-9ce3-60917c7ecfe6","resolution":{"observed_at":"2026-08-07T11:59:13.027251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.05612","last_updated":"2018-07-29T19:11:57Z","snapshot_observed_at":"2026-08-14T20:46:56.185352Z","submitted_at":"2017-07-18T13:51:32Z","title":"VSE++: Improving Visual-Semantic Embeddings with Hard Negatives","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.05612","snapshot_observed_at":"2026-08-07T11:59:13.112215Z","title":"Improv- ing visual-semantic embeddings with hard negatives,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.112215Z"},"links":{"cited_paper":"/paper/1707.05612","citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:0eb72a7577bcfd2fe645e415b536ceaa1ca52e498c85651f6be05602a18b2e0a","observation_id":"f5f306d7-3da1-4bbe-a49b-0fc97179c124","resolution":{"observed_at":"2026-08-07T11:59:13.112215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.466142Z","title":"Video corpus moment retrieval with contrastive learning,","venue":null,"work_id":"86fb2afe-9fb4-46b6-9b89-f416db0a46cc","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.169725Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:a0c0461621059bfbbee5937303bcc604d75c05b0059c7ce22ed99665edc626f4","observation_id":"72a3d49d-6119-4727-a2b8-d4b1a56d04ea","resolution":{"observed_at":"2026-08-07T11:59:15.555960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.342838Z","title":"Dense-captioning events in videos,","venue":null,"work_id":"a67fc9ba-f03b-4430-85fc-410809a5e6b4","year":2017},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.216066Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:2f09306ca54c5cc0dc3933d08a67fc22e6548a7c3822f9620f989adb5b2154e7","observation_id":"55195957-2c3f-411c-9139-f85e6e67f6db","resolution":{"observed_at":"2026-08-07T11:59:15.390041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.211257Z","title":"Tvr: A large-scale dataset for video-subtitle moment retrieval,","venue":null,"work_id":"a1a88a2f-a495-439f-81b3-8b694f63bd8a","year":2020},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.312726Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:9237bedf7087969ba6cad221b141428c675fbc3cb51cdad504af17351705898e","observation_id":"98525be6-3e51-46cd-8b86-ef843f729faf","resolution":{"observed_at":"2026-08-07T11:59:15.258407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:15.074840Z","title":"Dual encoding for zero- example video retrieval,","venue":null,"work_id":"bf573752-2306-495b-9374-77500f3f04ca","year":2019},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.355804Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:615fe170cd12ea29995a7595ea093b3014274816d964dd4fb64ff8a786276754","observation_id":"54ebfa3a-0346-40ef-a76d-231b310ec87f","resolution":{"observed_at":"2026-08-07T11:59:15.143266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.889100Z","title":"W2vv++ fully deep learning for ad-hoc video search,","venue":null,"work_id":"89de93e9-70c0-43ed-bcd1-e93c24392a05","year":2019},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.415062Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:7b126e5ce9a4f9f83f5809c0b7272998fe265f25cae7130532421c2e4817dce1","observation_id":"b8e5ff83-9720-4658-933b-ef60c1bd3f0c","resolution":{"observed_at":"2026-08-07T11:59:14.976715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.717600Z","title":"Cap4video: What can auxiliary captions do for text-video retrieval?,","venue":null,"work_id":"bcd79012-e29d-4bb4-b8af-2310badeef44","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.535395Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:0becd7dda95f73bef45391d072af8765cd569e0abeb00346e689153285b565fb","observation_id":"27bab37a-96cb-461d-9c3d-794850efaa66","resolution":{"observed_at":"2026-08-07T11:59:14.818994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.617763Z","title":"Conquer: Contextual query-aware ranking for video corpus moment retrieval,","venue":null,"work_id":"aa54d8fc-a46e-4110-95dc-574f985342b3","year":2021},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.629463Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:d5458a5d500094638972f2922bcaadf340532505662909413f1a5b7fa64152ee","observation_id":"2a27c75e-479d-4716-bbde-d8f7efee4bf6","resolution":{"observed_at":"2026-08-07T11:59:14.663358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.451201Z","title":"Listen and look: Multi-modal aggregation and co-attention network for video-audio retrieval,","venue":null,"work_id":"923f4f9e-5db0-4027-997a-26f1c27de76f","year":2022},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.723130Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:c532690b6ed7bc8bebeea3bd5227f469c7b6a24106ef9c8d66adb4100d4fc8a8","observation_id":"43946145-345f-4406-ae3d-f924195ddd79","resolution":{"observed_at":"2026-08-07T11:59:14.491209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.265480Z","title":"Mixgen: A new multi-modal data augmentation,","venue":null,"work_id":"77f0dbc8-a798-4735-9537-a11a177b7906","year":2023},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.813900Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:ad39187016c2b6f4830c5e74a8ed86f71e8aa92a71fedce66c26d8c8788f7bf5","observation_id":"6a79fd9c-a0ac-4762-9197-85ad64bf90ed","resolution":{"observed_at":"2026-08-07T11:59:14.336523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:59:14.074927Z","title":"Mapfusion: A novel bev feature fusion network for multi-modal map construction,","venue":null,"work_id":"31db617c-d572-4ddb-b833-6bcb596b68a4","year":2025},"citing_paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:13.889276Z"},"links":{"citing_paper":"/paper/2506.00891"},"observation_digest":"sha256:a6361261231ed5e3cb0eb647bcc1c85fce197aa1b21eedb43407d84a44da3714","observation_id":"b62c90c7-fdcd-42c9-ad9a-9cb15e335668","resolution":{"observed_at":"2026-08-07T11:59:14.141393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.00891","last_updated":"2025-06-03T03:11:35Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T13:10:12.278997Z","submitted_at":"2025-06-01T08:21:45Z","title":"Uneven Event Modeling for Partially Relevant Video Retrieval"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":0,"verified_fuzzy":22},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2506.00891."}