{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:SMN7LSC2V3NMHHOSZ33I6FUPAU","short_pith_number":"pith:SMN7LSC2","schema_version":"1.0","canonical_sha256":"931bf5c85aaedac39dd2cef68f168f051061a2198e1f1e63ce29df644b11b897","source":{"kind":"arxiv","id":"2010.00263","version":1},"attestation_state":"computed","paper":{"title":"RefVOS: A Closer Look at Referring Expressions for Video Object Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Carina Silberer, Carles Ventura, Ioannis Kazakos, Jordi Torres, Miriam Bellver, Xavier Giro-i-Nieto","submitted_at":"2020-10-01T09:10:53Z","abstract_excerpt":"The task of video object segmentation with referring expressions (language-guided VOS) is to, given a linguistic phrase and a video, generate binary masks for the object to which the phrase refers. Our work argues that existing benchmarks used for this task are mainly composed of trivial cases, in which referents can be identified with simple phrases. Our analysis relies on a new categorization of the phrases in the DAVIS-2017 and Actor-Action datasets into trivial and non-trivial REs, with the non-trivial REs annotated with seven RE semantic categories. We leverage this data to analyze the re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.00263","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-10-01T09:10:53Z","cross_cats_sorted":[],"title_canon_sha256":"f3bf335220ad768fc476a5b3615c13f293b8fa379bae85193a8009f809aad8d2","abstract_canon_sha256":"4c9040cba82686e0969002bc1d62cbcb94dbbd88f3390cc3d182c528282e1ba3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:39:31.954452Z","signature_b64":"wfQ5NZ8CI2s3Jp0qmZ4+uxk18Q86yWLvbnVw1pAbzF6Nq04iJApg3VDbiQXWhvRprExzvENaKlpxTnaYjZYMDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"931bf5c85aaedac39dd2cef68f168f051061a2198e1f1e63ce29df644b11b897","last_reissued_at":"2026-07-05T01:39:31.953956Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:39:31.953956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RefVOS: A Closer Look at Referring Expressions for Video Object Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Carina Silberer, Carles Ventura, Ioannis Kazakos, Jordi Torres, Miriam Bellver, Xavier Giro-i-Nieto","submitted_at":"2020-10-01T09:10:53Z","abstract_excerpt":"The task of video object segmentation with referring expressions (language-guided VOS) is to, given a linguistic phrase and a video, generate binary masks for the object to which the phrase refers. Our work argues that existing benchmarks used for this task are mainly composed of trivial cases, in which referents can be identified with simple phrases. Our analysis relies on a new categorization of the phrases in the DAVIS-2017 and Actor-Action datasets into trivial and non-trivial REs, with the non-trivial REs annotated with seven RE semantic categories. We leverage this data to analyze the re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.00263","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.00263/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.00263","created_at":"2026-07-05T01:39:31.954016+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.00263v1","created_at":"2026-07-05T01:39:31.954016+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.00263","created_at":"2026-07-05T01:39:31.954016+00:00"},{"alias_kind":"pith_short_12","alias_value":"SMN7LSC2V3NM","created_at":"2026-07-05T01:39:31.954016+00:00"},{"alias_kind":"pith_short_16","alias_value":"SMN7LSC2V3NMHHOS","created_at":"2026-07-05T01:39:31.954016+00:00"},{"alias_kind":"pith_short_8","alias_value":"SMN7LSC2","created_at":"2026-07-05T01:39:31.954016+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.10611","citing_title":"Molmo2: Open Weights and Data for Vision-Language Models with Video Understanding and Grounding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09569","citing_title":"Automatic Mind Wandering Detection in Educational Settings: A Systematic Review and Multimodal Benchmarking","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU","json":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU.json","graph_json":"https://pith.science/api/pith-number/SMN7LSC2V3NMHHOSZ33I6FUPAU/graph.json","events_json":"https://pith.science/api/pith-number/SMN7LSC2V3NMHHOSZ33I6FUPAU/events.json","paper":"https://pith.science/paper/SMN7LSC2"},"agent_actions":{"view_html":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU","download_json":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU.json","view_paper":"https://pith.science/paper/SMN7LSC2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.00263&json=true","fetch_graph":"https://pith.science/api/pith-number/SMN7LSC2V3NMHHOSZ33I6FUPAU/graph.json","fetch_events":"https://pith.science/api/pith-number/SMN7LSC2V3NMHHOSZ33I6FUPAU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU/action/storage_attestation","attest_author":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU/action/author_attestation","sign_citation":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU/action/citation_signature","submit_replication":"https://pith.science/pith/SMN7LSC2V3NMHHOSZ33I6FUPAU/action/replication_record"}},"created_at":"2026-07-05T01:39:31.954016+00:00","updated_at":"2026-07-05T01:39:31.954016+00:00"}