{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NXBVB5S5UKYDM6NP3SKPYVRM7P","short_pith_number":"pith:NXBVB5S5","schema_version":"1.0","canonical_sha256":"6dc350f65da2b03679afdc94fc562cfbc9ac4d33b4f20fda5f4ba6ea14255a1d","source":{"kind":"arxiv","id":"2312.08514","version":2},"attestation_state":"computed","paper":{"title":"TAM-VT: Transformation-Aware Multi-scale Video Transformer for Segmentation and Tracking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Leonid Sigal, Mennatullah Siam, Raghav Goyal, Wan-Cyuan Fan","submitted_at":"2023-12-13T21:02:03Z","abstract_excerpt":"Video Object Segmentation (VOS) has emerged as an increasingly important problem with availability of larger datasets and more complex and realistic settings, which involve long videos with global motion (e.g, in egocentric settings), depicting small objects undergoing both rigid and non-rigid (including state) deformations. While a number of recent approaches have been explored for this task, these data characteristics still present challenges. In this work we propose a novel, clip-based DETR-style encoder-decoder architecture, which focuses on systematically analyzing and addressing aforemen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.08514","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-12-13T21:02:03Z","cross_cats_sorted":[],"title_canon_sha256":"a06fc80319b0ce1d0e22e7fcb7519dd7472f8ea3ae1732cfe0099e81fa6a494d","abstract_canon_sha256":"844f1b162b4df55b1fbc025a91bc0bfaae53ecebe986e1171a9c69e2acf23db1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:12.150270Z","signature_b64":"6rw9y8Jk+/a5zppGVFhxuJBMUV6eyHA5U3s0hMwX/InBD9izrE1s07XCMHjMZS6wC+NV95OaCW4KIkkAJ3b8AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6dc350f65da2b03679afdc94fc562cfbc9ac4d33b4f20fda5f4ba6ea14255a1d","last_reissued_at":"2026-07-05T08:06:12.149815Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:12.149815Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TAM-VT: Transformation-Aware Multi-scale Video Transformer for Segmentation and Tracking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Leonid Sigal, Mennatullah Siam, Raghav Goyal, Wan-Cyuan Fan","submitted_at":"2023-12-13T21:02:03Z","abstract_excerpt":"Video Object Segmentation (VOS) has emerged as an increasingly important problem with availability of larger datasets and more complex and realistic settings, which involve long videos with global motion (e.g, in egocentric settings), depicting small objects undergoing both rigid and non-rigid (including state) deformations. While a number of recent approaches have been explored for this task, these data characteristics still present challenges. In this work we propose a novel, clip-based DETR-style encoder-decoder architecture, which focuses on systematically analyzing and addressing aforemen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.08514","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.08514/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.08514","created_at":"2026-07-05T08:06:12.149872+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.08514v2","created_at":"2026-07-05T08:06:12.149872+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.08514","created_at":"2026-07-05T08:06:12.149872+00:00"},{"alias_kind":"pith_short_12","alias_value":"NXBVB5S5UKYD","created_at":"2026-07-05T08:06:12.149872+00:00"},{"alias_kind":"pith_short_16","alias_value":"NXBVB5S5UKYDM6NP","created_at":"2026-07-05T08:06:12.149872+00:00"},{"alias_kind":"pith_short_8","alias_value":"NXBVB5S5","created_at":"2026-07-05T08:06:12.149872+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2408.00714","citing_title":"SAM 2: Segment Anything in Images and Videos","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P","json":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P.json","graph_json":"https://pith.science/api/pith-number/NXBVB5S5UKYDM6NP3SKPYVRM7P/graph.json","events_json":"https://pith.science/api/pith-number/NXBVB5S5UKYDM6NP3SKPYVRM7P/events.json","paper":"https://pith.science/paper/NXBVB5S5"},"agent_actions":{"view_html":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P","download_json":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P.json","view_paper":"https://pith.science/paper/NXBVB5S5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.08514&json=true","fetch_graph":"https://pith.science/api/pith-number/NXBVB5S5UKYDM6NP3SKPYVRM7P/graph.json","fetch_events":"https://pith.science/api/pith-number/NXBVB5S5UKYDM6NP3SKPYVRM7P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P/action/storage_attestation","attest_author":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P/action/author_attestation","sign_citation":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P/action/citation_signature","submit_replication":"https://pith.science/pith/NXBVB5S5UKYDM6NP3SKPYVRM7P/action/replication_record"}},"created_at":"2026-07-05T08:06:12.149872+00:00","updated_at":"2026-07-05T08:06:12.149872+00:00"}