{"as_of":"2026-08-10T01:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c39ed35f9312a9e668b3c8204ef6344b8cc8fe4461e8a9f676d6fd41646f1ea8","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":53,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T20:36:25.525906Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T13:19:51.020711Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-07T12:45:05.096694Z","title":"Bansal, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23656","last_updated":"2025-05-29T17:06:44Z","snapshot_observed_at":"2026-08-07T12:38:23.686012Z","submitted_at":"2025-05-29T17:06:44Z","title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:05.096694Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2505.23656"},"observation_digest":"sha256:046097f69e3ba9a966bbdbb77f51889260cb7bed8c3671b815fa01d47b2e74fe","observation_id":"101b1ba2-fbfd-4999-b886-9c911a6af147","resolution":{"observed_at":"2026-08-07T12:45:05.096694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-06T16:30:50.339759Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.13428","last_updated":"2026-05-26T08:34:24Z","snapshot_observed_at":"2026-08-06T16:22:35.916708Z","submitted_at":"2025-07-17T17:54:09Z","title":"\"PhyWorldBench\": A Comprehensive Evaluation of Physical Realism in Text-to-Video Models","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T16:30:50.339759Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2507.13428"},"observation_digest":"sha256:7f86c28ce431c26b862cf808f864c3c0beb81f23d4fa5c8e0a04224dd8c230e0","observation_id":"5952afe4-b42d-4238-8458-97351e4b5921","resolution":{"observed_at":"2026-08-06T16:30:50.339759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-06T15:27:00.441134Z","title":"Bansal, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15824","last_updated":"2025-07-21T17:30:46Z","snapshot_observed_at":"2026-08-07T10:31:41.147971Z","submitted_at":"2025-07-21T17:30:46Z","title":"Can Your Model Separate Yolks with a Water Bottle? Benchmarking Physical Commonsense Understanding in Video Generation Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T15:27:00.441134Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2507.15824"},"observation_digest":"sha256:d4e8aa4d4ae84876f85e12a84b4c0235f7938ba198301b8780e984b52a1e5a5c","observation_id":"5408ffb6-a523-4df4-b901-5475e57ec79a","resolution":{"observed_at":"2026-08-06T15:27:00.441134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-05T10:59:25.198610Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evalua- tion in Video Generation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.03385","last_updated":"2025-09-03T15:02:40Z","snapshot_observed_at":"2026-08-08T21:04:36.444975Z","submitted_at":"2025-09-03T15:02:40Z","title":"Human Preference-Aligned Concept Customization Benchmark via Decomposed Evaluation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T10:59:25.198610Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2509.03385"},"observation_digest":"sha256:68912ec9d3c4e76a5c158995339a361765263060b465b59854c9b6ee6fff391c","observation_id":"8814fc66-f840-41c2-83fd-f5297fd85d97","resolution":{"observed_at":"2026-08-05T10:59:25.198610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2509.24702","last_updated":"2026-04-06T10:12:03Z","snapshot_observed_at":"2026-08-02T11:34:44.120596Z","submitted_at":"2025-09-29T12:32:54Z","title":"Enhancing Physical Plausibility in Video Generation by Reasoning the Implausibility","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-18T12:55:42.679016Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2509.24702"},"observation_digest":"sha256:3eb9e1e7b3b2ebf3fe388bfb328ba5ce73da7ea55510ea7b09c40f1f57907c42","observation_id":"3512bfe1-4b8d-4c9d-99a2-9424d273ac60","resolution":{"observed_at":"2026-05-18T12:56:24.343732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2511.00062","last_updated":"2026-02-24T21:52:50Z","snapshot_observed_at":"2026-07-06T22:34:38.619949Z","submitted_at":"2025-10-28T22:44:13Z","title":"World Simulation with Video Foundation Models for Physical AI","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T23:01:13.546110Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2511.00062"},"observation_digest":"sha256:8536355e0ea2c3124780133593ff3c67c882ada67e77bcbf07c7046b8e33de26","observation_id":"feea1a15-9db2-4ab4-8253-1d0e571742b9","resolution":{"observed_at":"2026-05-12T23:01:13.918325Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2511.18373","last_updated":"2026-04-11T05:44:20Z","snapshot_observed_at":"2026-08-07T02:29:17.852730Z","submitted_at":"2025-11-23T09:43:44Z","title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T05:55:11.495430Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2511.18373"},"observation_digest":"sha256:cb5e4903e3d2a649699e03fae4794c879de314cf1383ffffe09a20090f2bedb9","observation_id":"500a90b3-6903-4d27-8f9b-9b60434e040c","resolution":{"observed_at":"2026-05-17T05:59:08.743022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-03T19:11:47.098371Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.01803","last_updated":"2026-07-09T15:41:25Z","snapshot_observed_at":"2026-08-08T13:02:50.552789Z","submitted_at":"2025-12-01T15:36:33Z","title":"Generative Action Tell-Tales: Assessing Human Motion in Synthesized Videos","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T19:11:47.098371Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.01803"},"observation_digest":"sha256:a6219aab990e61f58551eb488773e1877d8f2979bcae7cb0a1e3404b86964e27","observation_id":"2eaf1f6a-91c2-4d27-b10a-c018831d97ce","resolution":{"observed_at":"2026-08-03T19:11:47.098371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2512.01843","last_updated":"2026-05-18T11:10:13Z","snapshot_observed_at":"2026-08-08T17:35:04.569091Z","submitted_at":"2025-12-01T16:28:13Z","title":"PhyDetEx: Detecting and Explaining the Physical Plausibility of T2V Models","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T17:57:57.263574Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.01843"},"observation_digest":"sha256:576aee61d13846dd3d1b57e5154ba898bb71f268e50a9ee20155ad1350e941ec","observation_id":"7661578d-9147-4d74-b3f5-cc93263bf7bd","resolution":{"observed_at":"2026-05-21T18:00:27.237634Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2512.05564","last_updated":"2026-04-12T02:52:57Z","snapshot_observed_at":"2026-07-06T22:37:50.431948Z","submitted_at":"2025-12-05T09:39:26Z","title":"ProPhy: Progressive Physical Alignment for Dynamic World Simulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T01:05:08.087136Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.05564"},"observation_digest":"sha256:0813984eb701b6d68043056faca1c70653a8fc3625c9dd7aa27afe73592cabe0","observation_id":"8a6a0aeb-f9d3-42e4-80ff-be1b46426c19","resolution":{"observed_at":"2026-05-17T01:08:47.998118Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2512.23292","last_updated":"2026-05-20T15:48:38Z","snapshot_observed_at":"2026-08-01T23:47:11.303976Z","submitted_at":"2025-12-29T08:26:27Z","title":"Agentic Physical AI toward a Domain-Specific Foundation Model for Nuclear Reactor Control","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T16:57:19.490074Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2512.23292"},"observation_digest":"sha256:6cbecf2e2b17b6c029c2a597815a7259f153a2edd7ca850a31398f30cad28ae3","observation_id":"98814dd8-c6bb-49dc-8692-50fc179782af","resolution":{"observed_at":"2026-05-21T17:00:23.976691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2601.18577","last_updated":"2026-05-20T05:19:37Z","snapshot_observed_at":"2026-07-06T22:43:02.453697Z","submitted_at":"2026-01-26T15:22:27Z","title":"Self-Refining Video Sampling","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T14:37:57.167882Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2601.18577"},"observation_digest":"sha256:7bdc8e4a1ff15d0f6b635c944dffd1dae04cc17d721987b1325e7e93106bdeab","observation_id":"4ed3be5e-8e4f-4170-a9ab-4d69a8c8e3c6","resolution":{"observed_at":"2026-05-21T14:40:14.478534Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.06339","last_updated":"2026-04-07T18:17:05Z","snapshot_observed_at":"2026-07-06T22:54:53.308309Z","submitted_at":"2026-04-07T18:17:05Z","title":"Evolution of Video Generative Foundations","version":1},"reference_index":170,"source":"pdf_text","source_observed_at":"2026-05-10T18:41:38.616611Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.06339"},"observation_digest":"sha256:7c08b0a12e5d2cd9bea68c216264b19e0937ee5f38e347c4180c4eab6f1fcf9a","observation_id":"c657b1b5-45b9-42b6-b311-c1d46f519cb2","resolution":{"observed_at":"2026-05-11T00:05:51.471254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.07348","last_updated":"2026-04-08T17:59:22Z","snapshot_observed_at":"2026-07-06T22:55:37.787732Z","submitted_at":"2026-04-08T17:59:22Z","title":"MoRight: Motion Control Done Right","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T17:38:03.776766Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.07348"},"observation_digest":"sha256:fa01e00f77e6592edfcd54a7e8036996ea9fe87b6149626ac019905c44600c4f","observation_id":"ddbbf14e-72a5-420a-abb7-34edde3c7d7f","resolution":{"observed_at":"2026-05-11T06:26:01.050690Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.09415","last_updated":"2026-04-10T15:27:27Z","snapshot_observed_at":"2026-07-06T22:58:17.167367Z","submitted_at":"2026-04-10T15:27:27Z","title":"PhysInOne: Visual Physics Learning and Reasoning in One Suite","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T16:39:48.066744Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.09415"},"observation_digest":"sha256:af8db8da15f9a801e27c5f01fcd3d6fab0105f56e35b0f197b0131a093266738","observation_id":"6f66a35d-6b39-4dba-991d-3817f187d81b","resolution":{"observed_at":"2026-05-11T08:25:59.537119Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2604.19193","last_updated":"2026-04-21T08:04:02Z","snapshot_observed_at":"2026-08-02T22:05:18.077263Z","submitted_at":"2026-04-21T08:04:02Z","title":"How Far Are Video Models from True Multimodal Reasoning?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T02:44:52.920816Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2604.19193"},"observation_digest":"sha256:f6230e21a19b7b16464ff34e8eab3400af634b353f836d2a930b7849f7afffeb","observation_id":"9864ebcd-f30e-4cc2-9c58-0e545b7f4650","resolution":{"observed_at":"2026-05-11T12:51:03.643358Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.00873","last_updated":"2026-04-24T21:34:52Z","snapshot_observed_at":"2026-07-06T23:14:11.211164Z","submitted_at":"2026-04-24T21:34:52Z","title":"BRITE: A Benchmark for Reliable and Interpretable T2V Evaluation on Implausible Scenarios","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-09T21:12:33.209353Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.00873"},"observation_digest":"sha256:ebf88f63d691894bc03556e11dd625920ddad481cb3994466c38b418c1059fa2","observation_id":"199032ec-d394-47bc-b93f-e3e2d2bee3c2","resolution":{"observed_at":"2026-05-11T14:41:33.649224Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07061","last_updated":"2026-05-29T22:08:19Z","snapshot_observed_at":"2026-08-03T12:53:30.755817Z","submitted_at":"2026-05-08T00:14:07Z","title":"Do Joint Audio-Video Generation Models Understand Physics?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T02:12:04.230076Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07061"},"observation_digest":"sha256:258aa438669b948fe9e7048bcaf053c3fafa6e771e8f02a95b8d4faf0fbeb22b","observation_id":"236bef27-0804-4a57-a7eb-11e53c67c88e","resolution":{"observed_at":"2026-05-11T03:50:56.767419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07061","last_updated":"2026-05-29T22:08:19Z","snapshot_observed_at":"2026-08-03T12:53:30.755817Z","submitted_at":"2026-05-08T00:14:07Z","title":"Do Joint Audio-Video Generation Models Understand Physics?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T23:39:22.070629Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07061"},"observation_digest":"sha256:f9336bae8cee44a23ddc1d234cd934a80e75d1f28bcfb2249f038050d886475f","observation_id":"68ca9a9f-28a4-45f6-97c7-000e535ba148","resolution":{"observed_at":"2026-06-30T23:45:08.203083Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07800","last_updated":"2026-06-09T17:19:58Z","snapshot_observed_at":"2026-08-02T01:30:37.148519Z","submitted_at":"2026-05-08T14:36:32Z","title":"SARA: Semantically Adaptive Relational Alignment for Video Diffusion Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T02:21:52.861714Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07800"},"observation_digest":"sha256:b4150c9f6757007121781338874071e8753900ddfd3ad0602fa08c2042053498","observation_id":"c4348ac0-7666-440b-b235-55241b7a45f7","resolution":{"observed_at":"2026-05-11T03:40:54.609388Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.07800","last_updated":"2026-06-09T17:19:58Z","snapshot_observed_at":"2026-08-02T01:30:37.148519Z","submitted_at":"2026-05-08T14:36:32Z","title":"SARA: Semantically Adaptive Relational Alignment for Video Diffusion Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-30T23:13:31.195562Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.07800"},"observation_digest":"sha256:942ebf7000728cebd5ba71b412d71ed976eec664e735c53842ef8a02aa1f08cf","observation_id":"ae39e17c-04af-48de-bf8a-868a40164b10","resolution":{"observed_at":"2026-06-30T23:15:08.007606Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-07-06T23:22:42.512039Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:99c3c2e8d5366197e0c79556c52372edf349a9ace5b113361697f3a1518318bd","observation_id":"42134aa3-572d-4f86-9217-752a2cba9445","resolution":{"observed_at":"2026-05-12T05:21:27.838360Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.11723","last_updated":"2026-05-28T12:50:43Z","snapshot_observed_at":"2026-08-01T23:43:09.636688Z","submitted_at":"2026-05-12T08:08:33Z","title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T06:00:31.582714Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.11723"},"observation_digest":"sha256:2ca09c33524753995566d7d5c78a1a2a11975c4d8808af86c0e6fc68db1dc90d","observation_id":"c6006889-d1ff-4cf3-818a-0b4733a448b9","resolution":{"observed_at":"2026-05-13T06:02:22.162702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.11723","last_updated":"2026-05-28T12:50:43Z","snapshot_observed_at":"2026-08-01T23:43:09.636688Z","submitted_at":"2026-05-12T08:08:33Z","title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T22:38:43.102769Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.11723"},"observation_digest":"sha256:05630e7d49a9ecd520991dab4bb72756c57154c6f4c9b4aee057a467026a9385","observation_id":"2dac5e39-8a77-4e2b-a560-ced9b61bfc1b","resolution":{"observed_at":"2026-07-01T13:55:45.294459Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.15116","last_updated":"2026-05-14T17:29:35Z","snapshot_observed_at":"2026-07-06T23:26:28.006734Z","submitted_at":"2026-05-14T17:29:35Z","title":"DriveCtrl: Conditioned Sim-to-Real Driving Video Generation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-30T21:06:06.548538Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.15116"},"observation_digest":"sha256:6a14230c980b6814d7e2cedb71bdcf7440ac2e48a41c46329a87f7080831bda8","observation_id":"b5c2cfd1-1775-4c66-bdd3-46d10217d19f","resolution":{"observed_at":"2026-06-30T21:15:04.778745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.18396","last_updated":"2026-05-19T06:23:43Z","snapshot_observed_at":"2026-07-06T23:29:14.916114Z","submitted_at":"2026-05-18T13:42:24Z","title":"NEWTON: Agentic Planning for Physically Grounded Video Generation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T10:56:22.343333Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.18396"},"observation_digest":"sha256:4771e7ce95c8a8b21ad6b59f027c8b73a74e2a4b7227c1a77b4fe16433a4a10c","observation_id":"881ec02c-7a65-4267-a803-556ef6b74eb5","resolution":{"observed_at":"2026-05-20T10:58:13.762531Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.19242","last_updated":"2026-05-19T01:28:52Z","snapshot_observed_at":"2026-08-06T22:30:04.170402Z","submitted_at":"2026-05-19T01:28:52Z","title":"PhyWorld: Physics-Faithful World Model for Video Generation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-20T07:28:20.248452Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.19242"},"observation_digest":"sha256:984eba82fe1e1dc6d82d7e7c18a1800fc14352976bb8c7ee59ee4e6787e499af","observation_id":"a98e831c-1b5f-4ca3-91d7-63573d7e2c39","resolution":{"observed_at":"2026-05-20T07:33:07.595472Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.23699","last_updated":"2026-05-22T14:51:22Z","snapshot_observed_at":"2026-07-06T23:33:53.870540Z","submitted_at":"2026-05-22T14:51:22Z","title":"CRONOS: Benchmarking Counterfactual Physical Consistency in Video Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-25T04:39:22.400458Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.23699"},"observation_digest":"sha256:694a4328e7271498eaf3b0a9939bf8cd105891dddb316b6124fdaac58df7f1fd","observation_id":"5cd09e5e-1276-445c-953d-e2da461e1bdb","resolution":{"observed_at":"2026-05-25T04:40:23.190574Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.23878","last_updated":"2026-05-22T17:34:42Z","snapshot_observed_at":"2026-08-02T04:44:48.004184Z","submitted_at":"2026-05-22T17:34:42Z","title":"LaMo: Self-Supervised Latent Motion Priors for Physical Realism in Video Generation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-25T04:42:32.717968Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.23878"},"observation_digest":"sha256:a01260e7b9e0374fca7dec69811c143b16bf94581b467ebceae470988029342b","observation_id":"c747a6d3-4c52-4ddd-b01f-5d5acc06334a","resolution":{"observed_at":"2026-05-25T04:45:20.481958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.24962","last_updated":"2026-05-24T09:28:05Z","snapshot_observed_at":"2026-08-02T19:14:30.533296Z","submitted_at":"2026-05-24T09:28:05Z","title":"Tempered Self-Similarity Alignment for Physically Plausible Video Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T11:39:06.597513Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.24962"},"observation_digest":"sha256:ca32da39ec2dd4fe77e5f96d8a368eaa4b1750e846caf6d245bac921ba496c41","observation_id":"3916976f-d93f-406c-9aca-fa3e7b81f9c7","resolution":{"observed_at":"2026-06-30T11:44:38.371299Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.25874","last_updated":"2026-05-25T14:01:31Z","snapshot_observed_at":"2026-07-06T23:35:46.157653Z","submitted_at":"2026-05-25T14:01:31Z","title":"WBench: A Comprehensive Multi-turn Benchmark for Interactive Video World Model Evaluation","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-29T22:57:08.381846Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.25874"},"observation_digest":"sha256:7b6c68f64fbfa6e64b8d737bd9522ae320572004f2f6a0839928672b71c55a0a","observation_id":"31fa3a2e-2dd7-4854-ba29-969d8e717dd3","resolution":{"observed_at":"2026-06-29T23:14:02.269296Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.27589","last_updated":"2026-05-26T19:02:26Z","snapshot_observed_at":"2026-08-05T09:42:53.562841Z","submitted_at":"2026-05-26T19:02:26Z","title":"What-If World: A Causal Benchmark for General World Models in Embodied Scenarios","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T18:23:22.987086Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.27589"},"observation_digest":"sha256:3e4b37719aaf39c4d050f7cff7e495c12c2273857ebdf7e680a99303f084c744","observation_id":"2dc8fd89-4e71-47a0-8c81-0f5c24d823d7","resolution":{"observed_at":"2026-06-29T18:23:50.403328Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.28230","last_updated":"2026-05-27T09:44:18Z","snapshot_observed_at":"2026-08-07T12:12:27.862303Z","submitted_at":"2026-05-27T09:44:18Z","title":"Proprio: Latent Self-Scoring and Inference-Time Refinement for Physically Plausible Video Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T12:55:24.689338Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.28230"},"observation_digest":"sha256:b9a35e48ed8b449075119e636e8556b96cec1ae7f4c93499f77b33e95bc7605c","observation_id":"928a2668-cecb-4338-97a9-034a1af2aa1c","resolution":{"observed_at":"2026-06-29T13:03:26.633426Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.30346","last_updated":"2026-05-28T17:59:51Z","snapshot_observed_at":"2026-08-04T10:14:05.552405Z","submitted_at":"2026-05-28T17:59:51Z","title":"YoCausal: How Far is Video Generation from World Model? A Causality Perspective","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T08:27:03.674229Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.30346"},"observation_digest":"sha256:d30d6470c707682484e5454f860546d0b7180f7db0282a58689733ca48cd708b","observation_id":"b092f7a9-95a8-4f84-9bd4-cd2b18a83278","resolution":{"observed_at":"2026-06-29T08:33:15.610723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.00499","last_updated":"2026-05-30T03:13:00Z","snapshot_observed_at":"2026-08-09T10:23:42.127989Z","submitted_at":"2026-05-30T03:13:00Z","title":"OptiWorld: Optimal Control for Video World Generation under Physical Constraints","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T19:02:51.848742Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.00499"},"observation_digest":"sha256:e77f414821f19e301a6135a7c2aa1e4933a42e539cd89af9488c09e6bd49503b","observation_id":"2a5c3853-3881-45ff-821b-facdf2267b1e","resolution":{"observed_at":"2026-06-28T19:32:35.562695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.01538","last_updated":"2026-06-11T05:58:14Z","snapshot_observed_at":"2026-08-02T12:35:46.473849Z","submitted_at":"2026-06-01T01:36:44Z","title":"MPMWorlds: Material-Point-Method Simulations for Inferring and Extrapolating Physical Dynamics","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T12:19:27.221596Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.01538"},"observation_digest":"sha256:42ae8c43d9cd0ac5c6ade3f94e7c44d06e2f712922e2687c2a582578d9e4209e","observation_id":"c199f755-4fc8-4cc0-8f1d-1a8b4870040a","resolution":{"observed_at":"2026-07-02T01:16:24.684102Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.04737","last_updated":"2026-06-03T11:20:00Z","snapshot_observed_at":"2026-07-06T23:44:47.848374Z","submitted_at":"2026-06-03T11:20:00Z","title":"Physics-Informed Video Generation via Mixture-of-Experts Latent Alignment","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T07:02:37.291472Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.04737"},"observation_digest":"sha256:4e5db0d453545968e3c12b449d9340d0fe8462f848a84bb8208536a029099356","observation_id":"d8879029-c4bb-428d-9030-f6d732fd5ae2","resolution":{"observed_at":"2026-07-02T07:16:44.758956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.20545","last_updated":"2026-06-18T17:55:15Z","snapshot_observed_at":"2026-08-09T08:58:43.424376Z","submitted_at":"2026-06-18T17:55:15Z","title":"Current World Models Lack a Persistent State Core","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-26T17:33:41.461245Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.20545"},"observation_digest":"sha256:23eb0d457e7a0d4ec97388aa2377e16a7013430bfd2bb8243c7868cdd5f8e763","observation_id":"db7ea51f-8389-43de-bceb-5e4a862c72a0","resolution":{"observed_at":"2026-07-04T03:49:31.017133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.22918","last_updated":"2026-06-22T06:58:39Z","snapshot_observed_at":"2026-07-06T23:57:42.632111Z","submitted_at":"2026-06-22T06:58:39Z","title":"Each Judge Its Own Yardstick: Discovering Per-VLM Taxonomies for Physical Video Evaluation","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-26T09:04:23.965554Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.22918"},"observation_digest":"sha256:b8dc65e0f2d3461d16e1b31e802e83152fb9163d440343bed9e20dff68d5343f","observation_id":"7e2c8aa6-6fec-46e8-a275-c5d9d023e5a7","resolution":{"observed_at":"2026-07-04T10:09:45.415761Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.26916","last_updated":"2026-06-25T11:53:27Z","snapshot_observed_at":"2026-07-07T00:01:10.711037Z","submitted_at":"2026-06-25T11:53:27Z","title":"PhysRAG: Enhancing Physics-Awareness in Video Generation via Retrieval-Augmented Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T05:16:53.011837Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.26916"},"observation_digest":"sha256:975a08aff5b58ab2bd625b5bd182db5fe65e1f8f21270ec81ddc4d9e085225b4","observation_id":"03148a37-6e8a-4ca4-a273-bb1a937b2c8b","resolution":{"observed_at":"2026-07-04T13:19:51.022182Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T02:03:45.564122Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:26834648cffe5998d654880a656f1b728e0b8dabed76e91247907dae02974585","observation_id":"5b8ddd1a-9cf9-47f0-88fd-f3abfedb3e5a","resolution":{"observed_at":"2026-07-01T18:25:57.779239Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-01T06:25:58.872140Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:cd010d628b96f17e06b587a5176cb45f0bf9ae8a6dc7b0abfb2259afb91d1d7e","observation_id":"776a5435-d44c-4177-9a53-4db6cf6b2e37","resolution":{"observed_at":"2026-07-01T09:35:40.499609Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-02T20:52:28.444524Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:80b1fea23013e7265b1de981640531a946ebd9d7feed786086f0bda82026fae7","observation_id":"9cd71609-d385-4009-b531-d27eaf95c46a","resolution":{"observed_at":"2026-07-02T20:57:22.806256Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-03T22:44:16.272541Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:5b4af4c50d956d87edac4785f859568ac14e799003001f6ddf8d52adc8eb2a44","observation_id":"fbfc599f-82d6-4ac7-9073-49c0a272493f","resolution":{"observed_at":"2026-07-03T22:49:00.899176Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-14T17:14:19.770867Z","title":"arXiv preprint arXiv:2503.06800 (2025) 3, 4","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-14T17:14:19.770867Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:a07dee5d68f71a603f894aa90624f559439356d8af53b5dd2b4c9150a6346aa2","observation_id":"06aee8cc-d380-4e35-81c5-407ff4d20c72","resolution":{"observed_at":"2026-07-14T17:14:19.770867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-02T10:00:09.187494Z","title":"arXiv preprint arXiv:2503.06800 (2025) 3, 4","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.27537","last_updated":"2026-07-19T21:49:02Z","snapshot_observed_at":"2026-08-05T10:16:29.451498Z","submitted_at":"2026-06-25T20:37:39Z","title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","version":6},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T10:00:09.187494Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2606.27537"},"observation_digest":"sha256:e6b06710060aa0e3f533210d3af6f9e2c3a26391cbd1323311f561fb996a1e0f","observation_id":"911c1311-d029-402b-81d2-40a9d4c148af","resolution":{"observed_at":"2026-08-02T10:00:09.187494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T21:07:22.906584Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16401","last_updated":"2026-07-17T18:00:05Z","snapshot_observed_at":"2026-08-08T02:26:45.240699Z","submitted_at":"2026-07-17T18:00:05Z","title":"Apple-$\\pi$: Benchmarking Thinking with Video Towards Law-Grounded Physical Intelligence","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T21:07:22.906584Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.16401"},"observation_digest":"sha256:a473c30214576b1ea399475c0a414add9e151a924dd0921216e253de576b6a44","observation_id":"42136782-eba8-4cb4-8ca0-4f81411e752f","resolution":{"observed_at":"2026-08-01T21:07:22.906584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T19:33:03.502183Z","title":"Videophy-2: A challenging action-centric physical com- monsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16947","last_updated":"2026-07-18T20:03:50Z","snapshot_observed_at":"2026-08-07T14:13:12.484072Z","submitted_at":"2026-07-18T20:03:50Z","title":"When Physical Preferences Meet Semantic Constraints: Physical and Semantic Direct Preference Optimization for Text-to-Video Generation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T19:33:03.502183Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.16947"},"observation_digest":"sha256:eb67e620de36174674d6919a4b186576f08a2947db5defa59cd2961c1faaee84","observation_id":"087d6028-ce69-4bf7-bf1d-e1f662b0f4e2","resolution":{"observed_at":"2026-08-01T19:33:03.502183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T17:47:06.272600Z","title":"Videophy-2: A challenging action-centric physical com- monsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17523","last_updated":"2026-07-20T03:56:43Z","snapshot_observed_at":"2026-08-09T15:11:31.679922Z","submitted_at":"2026-07-20T03:56:43Z","title":"Thinking in Video: Can Video Generators Really Reason About the Real World?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T17:47:06.272600Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.17523"},"observation_digest":"sha256:b9c25b2bcf9ac85c0205c14c2029e148707ffb1595efa9fed69168bc16136022","observation_id":"5920ead6-c2f8-48fb-9f7d-3462801e86f6","resolution":{"observed_at":"2026-08-01T17:47:06.272600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T14:02:10.808123Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation.arXiv preprint arXiv:2503.06800, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18924","last_updated":"2026-07-21T10:06:55Z","snapshot_observed_at":"2026-08-08T05:11:00.363961Z","submitted_at":"2026-07-21T10:06:55Z","title":"Learning Explicit Physical Parameter Control and Benchmarking for Video Generation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T14:02:10.808123Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.18924"},"observation_digest":"sha256:db393c5b0abfd549583cd83ac5e21e929099f6072208ce4929b9949212355655","observation_id":"a3f15e8b-0fbc-4760-a54c-db2a3b675253","resolution":{"observed_at":"2026-08-01T14:02:10.808123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T02:49:57.720734Z","title":"Bansal, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25321","last_updated":"2026-07-28T06:13:18Z","snapshot_observed_at":"2026-08-07T09:19:37.277038Z","submitted_at":"2026-07-28T06:13:18Z","title":"Physics-Grounded Fluid Video Generation with a Simulation Dataset and Dual-Stream Optical-Flow Supervision","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T02:49:57.720734Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.25321"},"observation_digest":"sha256:51688418f109517a1aaf5a5973d35327470518be47cfd5235155feeb2c17a6e5","observation_id":"a6a7e5c4-8726-45ce-a12b-a2e89635b934","resolution":{"observed_at":"2026-08-01T02:49:57.720734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-01T08:28:04.629823Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.27380","last_updated":"2026-07-29T18:38:23Z","snapshot_observed_at":"2026-08-02T23:26:16.524563Z","submitted_at":"2026-07-29T18:38:23Z","title":"VideoCoCo: Code-as-CoT for Physically-Consistent Video Generation via an Agentic Dual-Engine System","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T08:28:04.629823Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2607.27380"},"observation_digest":"sha256:832ce72e59576629442ebd966c7ed16d478fc964a134532232dd59c4f13ff2f7","observation_id":"73a0adac-4e87-49e6-b669-293c9e9c6aea","resolution":{"observed_at":"2026-08-01T08:28:04.629823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-08-07T20:36:25.525906Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05948","last_updated":"2026-08-06T12:19:38Z","snapshot_observed_at":"2026-08-09T23:12:35.041362Z","submitted_at":"2026-08-06T12:19:38Z","title":"GAUGE: A Measurement-Grounded Benchmark for Physical Fidelity in Simulation Engines and Video World Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T20:36:25.525906Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2608.05948"},"observation_digest":"sha256:c59cf4437c059423644413b23582d0620a90958895a175d4f67cde204f3baa06","observation_id":"824e9a33-f5e3-4323-b33b-9064aae12a26","resolution":{"observed_at":"2026-08-07T20:36:25.525906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.06800/citation-record","integrity":"/paper/2503.06800/integrity","json":"/paper/2503.06800/citation-record.json","paper":"/paper/2503.06800"},"outbound":[],"paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T23:44:57.579845Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 53 inbound Pith citation observations for arXiv:2503.06800."}