{"as_of":"2026-08-19T21:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:30d5d792f2b77bddb481bcb2215bc7f9d8ac856c585a69142dd281788a075c0b","coverage":[{"denominator":64,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":64,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T05:17:30.010064Z","state":"measured"},{"denominator":66,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":66,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T03:23:55.304755Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.10806","snapshot_observed_at":"2026-07-31T01:50:30.653972Z","title":"Phyground: Benchmarking physical reasoning in generative world models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28624","last_updated":"2026-07-30T17:59:46Z","snapshot_observed_at":"2026-08-17T09:28:25.316512Z","submitted_at":"2026-07-30T17:59:46Z","title":"PhiZero: A World Model Built Around Physical Language","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-31T01:50:30.653972Z"},"links":{"cited_paper":"/paper/2605.10806","citing_paper":"/paper/2607.28624"},"observation_digest":"sha256:28b620685d2864c6ec5e3d269414bc644cfc64fcb3c4bd0a0508b816d6f41ef7","observation_id":"ceff8e6a-3366-4083-b35a-93810768717e","resolution":{"observed_at":"2026-07-31T01:50:30.653972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.10806","snapshot_observed_at":"2026-08-04T03:23:55.304755Z","title":"Phyground: Benchmarking physical reasoning in generative world models.arXiv preprint arXiv:2605.10806, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.02603","last_updated":"2026-08-03T17:59:54Z","snapshot_observed_at":"2026-08-15T08:24:03.982326Z","submitted_at":"2026-08-03T17:59:54Z","title":"WorldExam: Benchmarking World Models from Apparent Appearance to Inherent Reactivity","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T03:23:55.304755Z"},"links":{"cited_paper":"/paper/2605.10806","citing_paper":"/paper/2608.02603"},"observation_digest":"sha256:902acffa15682fbb1595060abd7a2ea4ee7e36614811037614a23fad173285a2","observation_id":"17c98124-2781-4e15-b0a5-fd949c6f94bf","resolution":{"observed_at":"2026-08-04T03:23:55.304755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.10806/citation-record","integrity":"/paper/2605.10806/integrity","json":"/paper/2605.10806/citation-record.json","paper":"/paper/2605.10806"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-08-17T13:26:10.378579Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:a5509618a04c7e3a32f1748fac9569d4c0f472e6f197fd8247cfd0523990c4c1","observation_id":"e711beae-e78c-45ea-acdd-00f990d9cac0","resolution":{"observed_at":"2026-05-12T05:21:29.357124Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14378","last_updated":"2025-03-18T16:10:24Z","snapshot_observed_at":"2026-08-16T12:48:55.898946Z","submitted_at":"2025-03-18T16:10:24Z","title":"Impossible Videos","version":1},"cited_work":{"arxiv_id":"2503.14378","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.14378","snapshot_observed_at":"2026-07-01T10:25:41.408031Z","title":"Impossible videos","venue":null,"work_id":"5b20ddf4-07aa-4fa9-b1b0-e9f77b8d7345","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2503.14378","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:1df51c8f816b012a9532ce806e9e57e31e1d4e2d19c0c2c954dd42d4da00e8bf","observation_id":"ad7fc4c9-cdd6-467a-886f-318f9fa57741","resolution":{"observed_at":"2026-05-12T05:21:29.118097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.03520","last_updated":"2024-10-03T17:24:40Z","snapshot_observed_at":"2026-08-16T12:54:37.000371Z","submitted_at":"2024-06-05T17:53:55Z","title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","version":2},"cited_work":{"arxiv_id":"2406.03520","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.03520","snapshot_observed_at":"2026-07-07T15:43:53.689981Z","title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","venue":"cs.CV","work_id":"27ed795c-abbe-4de1-9a7a-2ecf39c354f3","year":2024},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2406.03520","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:176cf7b4fcfcf4d1d23ea1d02254c3471b89963e6622c5d217e42b2817496216","observation_id":"b5475a3e-9663-440d-8d76-0224bcdfbb6e","resolution":{"observed_at":"2026-05-20T11:34:37.994076Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06800","last_updated":"2025-03-09T22:49:12Z","snapshot_observed_at":"2026-08-16T12:51:43.372607Z","submitted_at":"2025-03-09T22:49:12Z","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","version":1},"cited_work":{"arxiv_id":"2503.06800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.06800","snapshot_observed_at":"2026-07-04T13:19:51.020711Z","title":"Videophy-2: A challenging action-centric physical commonsense evaluation in video generation","venue":null,"work_id":"4dfe3980-dfd5-4917-95e9-179c29e4eb18","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2503.06800","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:1c23301ce7a975176eef1985b9b29dfe03c597cd99d52e773ae00986ef274f69","observation_id":"42134aa3-572d-4f86-9217-752a2cba9445","resolution":{"observed_at":"2026-05-12T05:21:27.838360Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Campbell and Julian C","venue":null,"work_id":"a5b2e73a-66db-4d72-92e2-79f22c39137b","year":1963},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:2dee488cbb908c1f04aca76175f1a9717323a3d0218168c7fc5f7b04c0ffa5cf","observation_id":"fac1361f-d08a-494d-8aed-075a253046ef","resolution":{"observed_at":"2026-05-12T11:41:33.475040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01800","last_updated":"2024-12-02T18:47:25Z","snapshot_observed_at":"2026-08-16T13:00:36.187954Z","submitted_at":"2024-12-02T18:47:25Z","title":"PhysGame: Uncovering Physical Commonsense Violations in Gameplay Videos","version":1},"cited_work":{"arxiv_id":"2412.01800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.01800","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Physgame: Uncovering physical commonsense violations in gameplay videos","venue":null,"work_id":"46f4f776-4d46-4f0f-bd49-eec743b4e5ef","year":2024},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2412.01800","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:9b73f3657385662d6ce26a45be6744b5292c178ba289c196c616f68eb3f56115","observation_id":"805088b8-aac6-4b94-905e-e19c61f61bce","resolution":{"observed_at":"2026-05-12T05:21:27.757534Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nonnaïveté among amazon mechanical turk workers: Consequences and solutions for behavioral researchers.Behavior research methods, 46(1):112–130","venue":null,"work_id":"5a32035d-c477-4c09-9747-2bf6fe5afc60","year":2014},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:6b01ee855d2a7118ad6eaba30f7c13e087632e435f41ecde435f3b15ba0b34cb","observation_id":"837f6680-7e3a-4124-8b7c-7d5666031895","resolution":{"observed_at":"2026-05-12T11:41:33.471531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Understanding world or predicting future? a comprehensive survey of world models.ACM Computing Surveys, 58(3):1–38","venue":null,"work_id":"7c8b4f56-1dbd-411c-9db8-324d72daa029","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:38011892536e6d5a402d8dd4bf6c855738ca2e9c6018a02fcbd78b40c940c05c","observation_id":"04bbf226-7f1c-4a3f-be32-36f8bb923bb5","resolution":{"observed_at":"2026-05-12T11:41:33.480899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Veo 3.https://deepmind.google/models/veo/","venue":null,"work_id":"2b72fa44-bcf9-45ba-afdc-88af619805d6","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:af0f8ac24044bc4809292c4583aa5037b46aaff9f2c248360e07da6fe99bab00","observation_id":"2524dd35-b160-4e86-9027-72919e467c5a","resolution":{"observed_at":"2026-05-12T11:41:33.439820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13428","last_updated":"2026-05-26T08:34:24Z","snapshot_observed_at":"2026-08-06T16:22:35.916708Z","submitted_at":"2025-07-17T17:54:09Z","title":"\"PhyWorldBench\": A Comprehensive Evaluation of Physical Realism in Text-to-Video Models","version":3},"cited_work":{"arxiv_id":"2507.13428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.13428","snapshot_observed_at":"2026-07-08T07:14:45.185207Z","title":"InThe Thirteenth International Conference on Learning Representations","venue":"cs.CV","work_id":"7c164e09-5340-4f56-b63d-0e810f24f851","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2507.13428","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:f30b0d23c8f81792a835fd4aed06d3580756e5c8af4e52e093c89ce06085c2c0","observation_id":"603714e6-f940-4f6b-a3ab-832a175364c5","resolution":{"observed_at":"2026-05-27T02:05:04.304229Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cosmos world foundation models for physical ai","venue":null,"work_id":"96c69bdc-2745-4296-9981-af60af76ae3f","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:316b5d43b99d45324c050a01042556d81f194b6cff99082e5ef04178b037879f","observation_id":"e70236ce-1ae2-490c-a316-b0037ff20c03","resolution":{"observed_at":"2026-05-12T11:41:33.445334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.00337","last_updated":"2025-05-01T06:34:55Z","snapshot_observed_at":"2026-08-16T04:42:37.134626Z","submitted_at":"2025-05-01T06:34:55Z","title":"T2VPhysBench: A First-Principles Benchmark for Physical Consistency in Text-to-Video Generation","version":1},"cited_work":{"arxiv_id":"2505.00337","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.00337","snapshot_observed_at":"2026-07-04T13:59:51.792842Z","title":"T 2 V P hys B ench: A first-principles benchmark for physical consistency in text-to-video generation","venue":null,"work_id":"8e0819a9-997f-40ad-9d64-220808cd9d4c","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2505.00337","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:8c0ff09e7b93a8b4c0f50447d94c291181e0605fb1e1a03c6061b53bd93282ff","observation_id":"884c5c48-e4a7-46fc-bf9a-590b8b266281","resolution":{"observed_at":"2026-05-12T05:21:28.228199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03233","last_updated":"2026-01-06T18:24:41Z","snapshot_observed_at":"2026-08-18T20:29:58.857701Z","submitted_at":"2026-01-06T18:24:41Z","title":"LTX-2: Efficient Joint Audio-Visual Foundation Model","version":1},"cited_work":{"arxiv_id":"2601.03233","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.03233","snapshot_observed_at":"2026-07-10T01:46:41.047035Z","title":"LTX-2: Efficient Joint Audio-Visual Foundation Model","venue":"cs.CV","work_id":"1334671f-ee98-4c53-a1d4-0805281f8d2b","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2601.03233","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:b845f6ad5a8db87d5f99d59b0dfb6c954b17a400e1dff58c791a5fe2ae6f0e9d","observation_id":"01a9ba73-d263-4ccc-8df7-065601161be2","resolution":{"observed_at":"2026-05-12T05:21:28.818608Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T06:24:43.213996Z","title":"Video-bench: Human-aligned video generation benchmark","venue":null,"work_id":"53afc07f-f240-4533-adf6-304047280f56","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:3b4db49406a7c921868e8992bcc583c5ad8f653226f648e2781fadac2c3f45ce","observation_id":"6fae5349-abaa-4d26-9e63-925f2a51af26","resolution":{"observed_at":"2026-05-12T11:41:33.431298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.22799","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T10:27:56.742672Z","title":"Huynh-Thu, Q","venue":null,"work_id":"8b5960e1-cfbc-46a2-b0ae-e98f990faa4c","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:3a1fd6e602e05da7e4e01595b01e3b78b1aa85e197dd49f2a93b8e90c0ac602b","observation_id":"a1217bf1-d38d-4464-b3eb-9972861502a0","resolution":{"observed_at":"2026-05-12T05:21:28.907937Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T03:54:30.825321Z","title":"Videoscore: Building automatic metrics to simulate fine-grained human feedback for video generation","venue":null,"work_id":"a95a4387-dcbf-44b4-b3ed-2111b55ee80f","year":2024},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:74b2726fdb786351568f68354018a81894579ccec40ce810463899e142fad4bb","observation_id":"1ae48bb9-58da-4c9f-ad68-94609dc4b3a8","resolution":{"observed_at":"2026-05-12T11:41:33.422827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T01:20:38.029679Z","title":"Image quality metrics: Psnr vs","venue":null,"work_id":"fd5133ac-3e98-4ef0-84cc-81c2b24768cf","year":2010},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:4e591817980b9dd52c5ac0c385d799889cf506d0646644ca035e300d98218485","observation_id":"ea49ae1c-a838-426c-a815-a003f06da398","resolution":{"observed_at":"2026-05-12T11:41:33.414017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.02942","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T06:19:37.486447Z","title":"Benchmarking scientific understanding and reasoning for video generation using videoscience-bench","venue":null,"work_id":"d440513c-aa40-425f-8f07-255f2ebb82a1","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:8ddf28e135c33ac7a3a525d6ffb4bda70a397e3b10f98a6a7efe3f055df08042","observation_id":"4b8ca6e1-6940-4119-b055-265b16f8cd07","resolution":{"observed_at":"2026-05-12T05:21:28.798775Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cosmos-eval: Towards explainable evaluation of physics and semantics in text-to-video models","venue":null,"work_id":"c3d43549-7c80-4345-bc58-30dd26a23557","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:5f786b4218224562d24c98af50f3d7185052e53b708de60715db1c11e333b36f","observation_id":"458345e8-37d5-4f49-9f43-de0f074214e5","resolution":{"observed_at":"2026-05-12T11:41:33.427301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T01:07:50.611559Z","title":"Vbench: Comprehensive benchmark suite for video generative models","venue":null,"work_id":"2383dbc4-0e5d-40ed-a102-5aac37332eae","year":2024},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:3c79fc08dff2ddac12591d449343a45093c90017886021ada0bc543ebbe715d2","observation_id":"ac2e9143-d96d-4170-841b-7e3578e7c7d5","resolution":{"observed_at":"2026-05-12T11:41:33.449880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Krosnick","venue":null,"work_id":"74dc44ee-a825-433d-8fa1-909e2888f95b","year":1991},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:fd3dc1da9a76b4f7caa8349813a60ff9f298dae1624916d012d8a810a051db33","observation_id":"89149ddf-d2ee-4045-a6a3-7d19d0807453","resolution":{"observed_at":"2026-05-12T11:41:33.453722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20694","last_updated":"2025-02-28T03:58:23Z","snapshot_observed_at":"2026-08-16T12:54:26.116596Z","submitted_at":"2025-02-28T03:58:23Z","title":"WorldModelBench: Judging Video Generation Models As World Models","version":1},"cited_work":{"arxiv_id":"2502.20694","doi":"10.48550/arxiv.2502.20694","metadata_source":"pith","pith_arxiv_id":"2502.20694","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Li, S., Wu, K., Zhang, C., and Zhu, Y","venue":"cs.CV","work_id":"33a5087c-1633-4d52-b891-f34d10b9e7c8","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2502.20694","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:39d4343a111a8017987e03c1f3cdaebf5fe7b01f3eb3767cab0500f5800ed060","observation_id":"1e896341-996e-4c94-8a15-38b4b0eadf65","resolution":{"observed_at":"2026-05-12T05:21:28.258267Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-10T21:18:51.214081+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T21:18:51.214081+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13918","last_updated":"2025-10-27T08:22:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-23T18:55:41Z","title":"Improving Video Generation with Human Feedback","version":2},"cited_work":{"arxiv_id":"2501.13918","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.13918","snapshot_observed_at":"2026-07-04T13:19:51.115512Z","title":"Improving Video Generation with Human Feedback","venue":"cs.CV","work_id":"cfe4c01d-1cf7-4a00-ba86-06d583ca2cff","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2501.13918","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:00b17e57f1a5adf8f62ce982fe3ff7f01932d54b77c40dd601720bf2eee2a186","observation_id":"efb4bf65-7fd6-4b0c-81d7-a8c036db7991","resolution":{"observed_at":"2026-05-13T15:30:03.116566Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01255","last_updated":"2025-07-02T00:20:06Z","snapshot_observed_at":"2026-08-15T02:03:45.493317Z","submitted_at":"2025-07-02T00:20:06Z","title":"AIGVE-MACS: Unified Multi-Aspect Commenting and Scoring Model for AI-Generated Video Evaluation","version":1},"cited_work":{"arxiv_id":"2507.01255","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.01255","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aigve-macs: Unified multi-aspect commenting and scoring model for ai-generated video evaluation","venue":null,"work_id":"b0104470-8991-4cc5-be06-099a977c5288","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2507.01255","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:e06db13753a0997ee3325a5a5a55a43c80d2f876cf9596786110f8f26dd097c7","observation_id":"89572dfe-b377-4713-9158-53ddd38754ba","resolution":{"observed_at":"2026-05-12T05:21:28.617378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evalcrafter: Benchmarking and evaluating large video generation models","venue":null,"work_id":"6a9b41db-e552-408a-aa02-522edbaa304c","year":2024},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:04bddccb97948ecf82176486138f27ab27744d7d70475d7f79d7c6c117b7f5d6","observation_id":"cdf27064-5e3f-4410-beb6-6a8dfd50a003","resolution":{"observed_at":"2026-05-12T11:41:33.382340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05363","last_updated":"2024-10-07T17:56:04Z","snapshot_observed_at":"2026-08-08T09:48:34.424815Z","submitted_at":"2024-10-07T17:56:04Z","title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","version":1},"cited_work":{"arxiv_id":"2410.05363","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.05363","snapshot_observed_at":"2026-07-08T07:14:45.198485Z","title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","venue":"cs.CV","work_id":"644e2886-7387-436d-ac11-d848af5fcc71","year":2024},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2410.05363","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:664ad169ad8aebabc4f7a280eba1243e7e0fb227684b8db9b0d1cbb345a79393","observation_id":"b432c923-e7b5-489e-8d0e-4f3b68b14468","resolution":{"observed_at":"2026-05-18T14:40:00.221207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.07550","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T17:58:47.096428Z","title":"Travl: A recipe for making video-language mod- els better judges of physics implausibility","venue":null,"work_id":"c426a017-7a60-4f6b-90f6-7657c04fb4e7","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:c7020aace407d72051544cc2def03b15d7825d8c3d2ed89715c00dec12f0d546","observation_id":"9924653a-6535-4f4b-8da0-ce2497b56f79","resolution":{"observed_at":"2026-05-12T05:21:28.586302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Do generative video models understand physical principles? InProceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision, pages 948–958","venue":null,"work_id":"20863b0b-1454-4a9f-b6b8-775eb09bd754","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:94e9abccdaf3e7414b5c5fda47dcf489f659f611db83f28b0f6a127a4d5497b4","observation_id":"7931133a-73ab-4a23-b457-f6e7a32c27e1","resolution":{"observed_at":"2026-05-12T11:41:33.396219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6634f8c6-592a-4daf-b1c6-ac5000d5a1e6","year":1962},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:3c9871b715f20b4199cf797e99d4a014084eeb0652b8d56e72569fb20d871697","observation_id":"abd03aff-b5ff-4218-a812-ade3627a9d27","resolution":{"observed_at":"2026-05-12T11:41:33.324995Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.24458","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T00:07:28.009236Z","title":"Omniweaving: Towards unified video generation with free-form composition and reasoning.https://arxiv.org/abs/2603.24458","venue":null,"work_id":"c3fe3b8c-232b-435d-bf50-69ec6af6db0f","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:0904e07bf04a85ed834ced468fc0c28057dcb9c31488fc1b219b1537771c0359","observation_id":"4b51e551-f5f7-45a6-a2cd-55404c8e421d","resolution":{"observed_at":"2026-05-12T05:21:28.757546Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Running experiments on amazon mechanical turk.Judgment and Decision making, 5(5):411–419","venue":null,"work_id":"e610fd36-70fc-4087-b2de-e4604518f859","year":2010},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:80266669ee810cbe74968077495727987ef4cc8fd73ae602175002bc4a292e09","observation_id":"24eedc12-6c68-4e90-a1dc-2e76b2d9e580","resolution":{"observed_at":"2026-05-12T11:41:33.320458Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:16:04.687562Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"ad3e05b3-af3a-4fa2-ab30-c45f9f403277","year":2021},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:ecec5d7c87c2868e6d85cbada4da5f52f2ce0f0c1b7ec7a5d6273f74c65a85d6","observation_id":"c47d1df4-cfe3-4925-998a-a9a584b4c51f","resolution":{"observed_at":"2026-05-12T11:41:33.370350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reis and Charles M","venue":null,"work_id":"595a8605-13a1-4f4e-bf0e-764f3dce1148","year":2000},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:dfa927ccf991c98cdcfb8a9fb427e27d80e6cf0056210251a6955ec1f33abbff","observation_id":"52d2bf37-3b2c-4150-aae3-8fe00f3164a9","resolution":{"observed_at":"2026-05-12T11:41:33.328732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.16592","last_updated":"2026-04-17T17:51:46Z","snapshot_observed_at":"2026-07-06T23:03:52.631613Z","submitted_at":"2026-04-17T17:51:46Z","title":"Human Cognition in Machines: A Unified Perspective of World Models","version":1},"cited_work":{"arxiv_id":"2604.16592","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16592","snapshot_observed_at":"2026-07-02T07:16:45.104866Z","title":"Human Cognition in Machines: A Unified Perspective of World Models","venue":"cs.RO","work_id":"e9f711d2-6321-4819-9a27-738f10976c6f","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2604.16592","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:8abed706df9085b52a6d84c0ecde3273179bf193da396586ef50c21e53a7c5c4","observation_id":"b4dcd527-a9ee-4efe-92da-744280dbdd2d","resolution":{"observed_at":"2026-05-12T05:21:28.570348Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Shadish, Thomas D","venue":null,"work_id":"85a27950-7c09-47af-8056-3d8ab87fba6f","year":2002},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:d9d433ccf9c2d72748212ecb6f27b6219ca3453f749f2fded2665491e5acbc18","observation_id":"a453b392-dd78-4048-96f7-ff2938345fba","resolution":{"observed_at":"2026-05-12T11:41:33.351301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vf-eval: Evaluating multimodal llms for generating feedback on aigc videos","venue":null,"work_id":"e67281cd-6f24-4765-9f5b-e66f2bef0ff9","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:42c061e91a9ec3ee322bd32a9e5bf1209a9fb11a0a3e9453c972254d5ec7b09b","observation_id":"39fda1bd-4801-4ae8-99d8-fe8a5464b9f2","resolution":{"observed_at":"2026-05-12T11:41:33.337033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04076","last_updated":"2025-02-06T13:41:24Z","snapshot_observed_at":"2026-08-17T06:39:00.829436Z","submitted_at":"2025-02-06T13:41:24Z","title":"Content-Rich AIGC Video Quality Assessment via Intricate Text Alignment and Motion-Aware Consistency","version":1},"cited_work":{"arxiv_id":"2502.04076","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.04076","snapshot_observed_at":"2026-07-04T16:39:57.276136Z","title":"Content-rich aigc video quality assessment via intricate text alignment and motion-aware consistency","venue":null,"work_id":"32a17dec-b6a9-453f-8bdd-e5c22ff6fa8d","year":2023},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2502.04076","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:67e02cfb212a83d1f10b573ce8bbed03c9183438ed223eab371688a14a56a457","observation_id":"fa1bf4dd-37e6-41c9-81ea-bc7a5c8936cd","resolution":{"observed_at":"2026-05-12T05:21:28.553668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:16fd5ad30bac9435827539d0fa4298feacf83ec2ab3da2e0a8f99f1e7202409a","observation_id":"f50cbe9e-451c-46dd-b4bf-025d89287f95","resolution":{"observed_at":"2026-05-12T05:21:28.578263Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01719","last_updated":"2025-02-07T03:54:34Z","snapshot_observed_at":"2026-08-16T20:05:23.668951Z","submitted_at":"2025-02-03T18:56:33Z","title":"MJ-VIDEO: Fine-Grained Benchmarking and Rewarding Video Preferences in Video Generation","version":3},"cited_work":{"arxiv_id":"2502.01719","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01719","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mj-video: Fine-grained benchmarking and rewarding video preferences in video generation","venue":null,"work_id":"24ffa43d-520e-494a-87bd-8c4559ac4136","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2502.01719","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:2f5ad625c0306c2e61d796ff833a082c1245c36be576a675f4055536896ed04d","observation_id":"28313f88-cbbb-403d-bfea-a7e042f23e29","resolution":{"observed_at":"2026-05-12T05:21:28.398672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T17:35:10.823088Z","title":"Fvd: A new metric for video generation","venue":null,"work_id":"bff3762d-6c6a-4b69-a92b-f3cdc2df44be","year":2019},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:c70995844e944e37397c471846e0dc3350ec6f96a2a9b0e3f51ecb4f98cbdc9e","observation_id":"1c845899-99ac-46f4-b31d-26d97d78121f","resolution":{"observed_at":"2026-05-12T11:41:33.364802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20314","last_updated":"2025-04-19T02:22:42Z","snapshot_observed_at":"2026-08-12T22:34:51.361078Z","submitted_at":"2025-03-26T08:25:43Z","title":"Wan: Open and Advanced Large-Scale Video Generative Models","version":2},"cited_work":{"arxiv_id":"2503.20314","doi":"10.1109/19.492748","metadata_source":"pith","pith_arxiv_id":"2503.20314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Wan: Open and Advanced Large-Scale Video Generative Models","venue":"cs.CV","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2503.20314","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:8d8041b6c0172c782be7684c06603f19b982dc3667b260c8af11afdf7e36737e","observation_id":"9923985f-d3cb-46ba-805f-dae2b5773e0c","resolution":{"observed_at":"2026-05-12T05:21:28.487243Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.12098","last_updated":"2025-05-17T17:49:26Z","snapshot_observed_at":"2026-08-17T17:30:06.121487Z","submitted_at":"2025-05-17T17:49:26Z","title":"LOVE: Benchmarking and Evaluating Text-to-Video Generation and Video-to-Text Interpretation","version":1},"cited_work":{"arxiv_id":"2505.12098","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.12098","snapshot_observed_at":"2026-07-02T13:56:59.272327Z","title":"Love: Benchmarking and evaluating text-to-video generation and video-to- text interpretation","venue":null,"work_id":"65b796ec-b5c4-4663-a486-1ac45eb1e875","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2505.12098","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:7f9b205a7f169f3b8c71c94f4c29ce13691cb4e25f711048c7cbd8b02efd7003","observation_id":"f03c7216-4506-48bb-a9b4-1fe2e7189b63","resolution":{"observed_at":"2026-05-12T05:21:28.543375Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aigv-assessor: Benchmarking and evaluating the perceptual quality of text-to-video generation with lmm","venue":null,"work_id":"55c8933a-47a8-4210-9804-99693fff58fb","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:16d94fa9d1fd5b2b19008c495014e26561ae894100e3adc0d54f54b60020ad88","observation_id":"25d614c6-d84d-4116-858c-d4707edf8d47","resolution":{"observed_at":"2026-05-12T11:41:33.389562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.20159","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T01:46:40.917858Z","title":"A very big video reasoning suite","venue":null,"work_id":"4cb52917-7539-4575-a350-92645383c4ef","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:0640418b25ee762d39eab5f0b0e7577da2a65416919d1b30a86f27ba22290d86","observation_id":"70a87992-5f73-4b2b-bd90-1636d5db86d2","resolution":{"observed_at":"2026-05-12T05:21:28.563098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.01843","last_updated":"2026-05-18T11:10:13Z","snapshot_observed_at":"2026-08-08T17:35:04.569091Z","submitted_at":"2025-12-01T16:28:13Z","title":"PhyDetEx: Detecting and Explaining the Physical Plausibility of T2V Models","version":3},"cited_work":{"arxiv_id":"2512.01843","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2512.01843","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Phydetex: Detecting and explaining the physical plausibility of t2v models","venue":null,"work_id":"994ad55a-f0eb-44ed-b5a4-dc6dfac22e29","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2512.01843","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:ad3bc41a912fb7a04e13d9e77cc028a555320ff4495b927e34ed1a971bf9fe7e","observation_id":"01aec354-73a1-4843-abe6-8df5100cbf10","resolution":{"observed_at":"2026-05-20T00:05:38.433674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T03:54:30.812941Z","title":"Visionreward: Fine-grained multi-dimensional human preference learning for image and video generation","venue":null,"work_id":"00f4fdb0-f1bf-459c-877a-6ac01a6ea3ff","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:586ae317b02185b28b1396c80bc948c13dcfbe3cadf2ed1090fd4d235da9f84b","observation_id":"029bc0e7-cef6-4d6c-b4ab-5c9bdad21d38","resolution":{"observed_at":"2026-05-12T11:41:33.401111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02918","last_updated":"2026-06-29T01:14:54Z","snapshot_observed_at":"2026-08-16T12:44:13.632005Z","submitted_at":"2025-04-03T15:21:17Z","title":"Evaluating Newtonian Mechanics in Video Generative Models with Real Physical Systems","version":3},"cited_work":{"arxiv_id":"2504.02918","doi":"10.48550/arxiv.2504.02918","metadata_source":"pith","pith_arxiv_id":"2504.02918","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zhang, D","venue":"cs.CV","work_id":"a5729cf7-9ce0-49a1-a57d-9886470c40b7","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2504.02918","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:223751f5062a76cb8381f11b288d4ad0e1037315012c252c974ba60adfaecd65","observation_id":"7536e8b1-71cb-44bb-b736-70e487934b53","resolution":{"observed_at":"2026-06-30T03:17:04.940243Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-10T21:18:48.706484+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T21:18:48.706484+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.19607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T10:54:36.504697Z","title":"Zhang, P","venue":null,"work_id":"82f208c5-735f-4958-97f5-738ebffd7e4d","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:2c9db66c59037b3c8713538b88cfd60406d4bebe3a2ec6d97b88fc2707ddc00f","observation_id":"492143e2-cd04-4275-b062-7558d610b776","resolution":{"observed_at":"2026-05-12T05:21:28.811518Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vq-insight: Teaching vlms for ai-generated video quality understanding via progressive visual reinforcement learning","venue":null,"work_id":"4b6d5001-67a2-4415-b04c-c39dd2f6339f","year":2026},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:0cf15465719e2c4367cbe4f2e978fe321ffb32877175beb1a68384b9a6757f9e","observation_id":"47de9737-c504-4582-9d94-b904377ac17d","resolution":{"observed_at":"2026-05-12T11:41:33.461840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Q-bench-video: Benchmark the video quality understanding of lmms","venue":null,"work_id":"8320ca5f-a265-43cc-ae8d-98ec5436a32c","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:6db7cd06188adbb8e8446c18daf3b3bac9e7c56e2958acef09a80244a060251a","observation_id":"75c319f0-613f-4fe6-ab55-e798c2cbb578","resolution":{"observed_at":"2026-05-12T11:41:33.467477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21755","last_updated":"2025-08-20T15:49:30Z","snapshot_observed_at":"2026-08-15T12:21:16.936063Z","submitted_at":"2025-03-27T17:57:01Z","title":"VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness","version":2},"cited_work":{"arxiv_id":"2503.21755","doi":"10.48550/arxiv.2503.21755","metadata_source":"pith","pith_arxiv_id":"2503.21755","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness","venue":"cs.CV","work_id":"14060202-ac5f-48e9-b91a-24d150775431","year":2025},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"cited_paper":"/paper/2503.21755","citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:3076b5c0c1f5cf7b5488b67463270c1d43d1dc68b7f271699f76330b0c49e772","observation_id":"876ce071-c6a5-494f-b23c-0f48aed884a6","resolution":{"observed_at":"2026-05-14T18:42:03.548540Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"spaghetti breaks into pieces","venue":null,"work_id":"90eb0b1c-600c-4d58-b623-92f8274cd58b","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:b68a6e04aed02abaaf68549bf8f173e5a83c1fae677cc51d9b090cd435a7a49e","observation_id":"4042779e-fc7a-4f2b-bbd1-0d876bf2575b","resolution":{"observed_at":"2026-05-12T11:41:33.457412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e3faef25-fbee-4c25-a2cf-84b8da6500fb","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:05567e7facd714b7b159b069b5f120b3f3c242ff0e88bb96fc8bbd041ea0dc62","observation_id":"da9a4e9a-5cd8-4148-a69d-11a69e08d042","resolution":{"observed_at":"2026-05-12T11:41:33.307689Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"rotary filling machine","venue":null,"work_id":"7a50d709-5ff7-452c-931c-7d7503c30ee2","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:b6687a9f26befae1951381c99536546383d9643d9cb869919b96bbd1fad2d81f","observation_id":"d1a8e8b0-7f7f-454d-8e8e-91f83023362a","resolution":{"observed_at":"2026-05-12T11:41:33.292309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prompts that are only superficially associated through keywords are excluded","venue":null,"work_id":"9cd9607a-eb4e-45ce-b757-71a672fc6e68","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:ceb541525f6e85abbaaf72990d889191955564aa7a6c50374b51da6350f91563","observation_id":"b3bdd210-3ca5-40f1-a717-96d039fdb3cc","resolution":{"observed_at":"2026-05-12T11:41:33.295500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The athlete throws a javelin","venue":null,"work_id":"ef4075ba-0282-4f0e-bfb4-ad345228e79c","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:d14425dc5fa7bdcfa1e9e4a7d329a5e8506cc65020c8801d587f79c05e91dc09","observation_id":"b6a112b9-d6a4-41b5-9cbe-8952c45410cc","resolution":{"observed_at":"2026-05-12T11:41:33.314157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The presentation order of videos is randomized independently for each annotator","venue":null,"work_id":"1dc40aff-8544-4fce-9b7b-7842bcffe461","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:9ad366a6c2627959395192c849a6db2f0d28f24146a76375174c0a7cbf3f7c06","observation_id":"086ad89c-0463-418b-a49c-2ab194d21d0f","resolution":{"observed_at":"2026-05-12T11:41:33.332304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The task is described without revealing our hypotheses, and model identities are hidden from annotators","venue":null,"work_id":"34c0b2ff-6357-4ed3-9e5c-38b3811d23a0","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:78f926fe2c7916cb08af653c7997fc6695acb26d48347ad2a819d0047f777a15","observation_id":"74c157d8-d976-446e-92c8-673133bc481a","resolution":{"observed_at":"2026-05-12T11:41:33.377798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The module presents example videos that illustrate different levels of physical realism and explains how to apply the five-point Likert scale","venue":null,"work_id":"b32f7f9f-56ef-4086-8c79-78d5a8049761","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:33423b36d70a8dc3456ba3377e09ba69646dd9fd81a2902d76c15082c7d73f8f","observation_id":"a296956f-5f00-4996-9df8-302c71552216","resolution":{"observed_at":"2026-05-12T11:41:33.342583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Annotators evaluate three general dimensions, semantic alignment, physical temporal validity, and object persistence, as well as the applicable physical laws for that video","venue":null,"work_id":"d335b989-aec7-48d9-9732-37bc336c75f3","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:4eb5395631e3644366965dcc141b190020727d70140afc195a2568162609976d","observation_id":"a925d8a5-e5f4-444a-9a27-5be0fc119fa8","resolution":{"observed_at":"2026-05-12T11:41:33.355422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"std= 0 means assigning the same score to every dimension of every video and provides no discriminating information whatsoever; std<0.3 is treated as near-constant","venue":null,"work_id":"93c96285-ee2f-43b1-ac94-57b4620f154b","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:ab6060df304dfeca5c0a6863f9fd7c8e348396f62d1334f7bb4532531b70618f","observation_id":"a2449bc4-6e59-4804-bb03-06e8485e1f4f","resolution":{"observed_at":"2026-05-12T11:41:33.361808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A 100% copy-paste rate means the annotator does not differentiate evaluation dimensions (e.g., gravity vs","venue":null,"work_id":"692bfee7-6924-40cb-aebf-a06a8ada8d15","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:5bfa7363a9a8af7738e435b17c9cb68dafb265b41f0d76d4a2d7bbc397f14e5e","observation_id":"fac7664f-e408-42e1-9f53-f27680acdd48","resolution":{"observed_at":"2026-05-12T11:41:33.406474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fef2e394-bb24-41a6-b62a-a54d5f97adea","year":null},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:68094f0637f013847dfb5f225cd9e49d73e3ff3b7d1f1699b2709c8e7e2a02bf","observation_id":"2c593aa0-ce88-4673-841a-6713409258f2","resolution":{"observed_at":"2026-05-12T11:41:33.347236Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1000.0794","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"skip when hard","venue":null,"work_id":"16cc253e-e267-4cbd-9f5d-6ccb195ad79f","year":2037},"citing_paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-12T05:17:30.010064Z"},"links":{"citing_paper":"/paper/2605.10806"},"observation_digest":"sha256:5c43121b0046b9fdffd68e4ba0bec1880a5793d440efbf513e9d8c6a6ed4ba73","observation_id":"9aa514a1-9e6b-47f4-a857-4a95ae8d313d","resolution":{"observed_at":"2026-05-12T05:21:28.187296Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.10806","last_updated":"2026-05-11T16:30:51Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T17:32:52.255062Z","submitted_at":"2026-05-11T16:30:51Z","title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models"},"reference_resolution":{"displayed":64,"state_counts":{"malformed_identifier":1,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":3,"verified_exact":23,"verified_fuzzy":33},"total_outbound_references":64},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 64 of 64 outbound references and 2 inbound Pith citation observations for arXiv:2605.10806."}