{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2ZCSNOQ5JLC757NPFFCY27S4KW","short_pith_number":"pith:2ZCSNOQ5","schema_version":"1.0","canonical_sha256":"d64526ba1d4ac5fefdaf29458d7e5c55bd06d357e74941fa40b6288f01fac8c0","source":{"kind":"arxiv","id":"2506.02356","version":3},"attestation_state":"computed","paper":{"title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jaeho Lee, Seongchan Kim, Seungryong Kim, Woojeong Jin","submitted_at":"2025-06-03T01:16:13Z","abstract_excerpt":"Referring video object segmentation (RVOS) aims to segment objects in a video described by a natural language expression. However, most existing approaches focus on segmenting only the referred object (typically the actor), even when the expression clearly describes an interaction involving multiple objects with distinct roles. For instance, \"A throwing B\" implies a directional interaction, but standard RVOS segments only the actor (A), neglecting other involved target objects (B). In this paper, we introduce Interaction-aware Referring Video Object Segmentation (InterRVOS), a novel task that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02356","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-03T01:16:13Z","cross_cats_sorted":[],"title_canon_sha256":"c9f0708fefa2b7eaa4e674cabe2b72d7c679e7145e2c50102a04f00096f80841","abstract_canon_sha256":"d5346c71ec55f531ad03d87a93ccdbdc3985efc8974a1834312110231bc3a23b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:19.734396Z","signature_b64":"80ceAUZuuy6mE4wrKjnu6UrTp7KZ+3Nfsms1JuH9Jw7quH0EUiOKRy42hhySuK+nsxAtPEunBKFhOxUJ0EtRDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d64526ba1d4ac5fefdaf29458d7e5c55bd06d357e74941fa40b6288f01fac8c0","last_reissued_at":"2026-07-05T11:55:19.733961Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:19.733961Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jaeho Lee, Seongchan Kim, Seungryong Kim, Woojeong Jin","submitted_at":"2025-06-03T01:16:13Z","abstract_excerpt":"Referring video object segmentation (RVOS) aims to segment objects in a video described by a natural language expression. However, most existing approaches focus on segmenting only the referred object (typically the actor), even when the expression clearly describes an interaction involving multiple objects with distinct roles. For instance, \"A throwing B\" implies a directional interaction, but standard RVOS segments only the actor (A), neglecting other involved target objects (B). In this paper, we introduce Interaction-aware Referring Video Object Segmentation (InterRVOS), a novel task that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02356","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02356","created_at":"2026-07-05T11:55:19.734011+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02356v3","created_at":"2026-07-05T11:55:19.734011+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02356","created_at":"2026-07-05T11:55:19.734011+00:00"},{"alias_kind":"pith_short_12","alias_value":"2ZCSNOQ5JLC7","created_at":"2026-07-05T11:55:19.734011+00:00"},{"alias_kind":"pith_short_16","alias_value":"2ZCSNOQ5JLC757NP","created_at":"2026-07-05T11:55:19.734011+00:00"},{"alias_kind":"pith_short_8","alias_value":"2ZCSNOQ5","created_at":"2026-07-05T11:55:19.734011+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW","json":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW.json","graph_json":"https://pith.science/api/pith-number/2ZCSNOQ5JLC757NPFFCY27S4KW/graph.json","events_json":"https://pith.science/api/pith-number/2ZCSNOQ5JLC757NPFFCY27S4KW/events.json","paper":"https://pith.science/paper/2ZCSNOQ5"},"agent_actions":{"view_html":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW","download_json":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW.json","view_paper":"https://pith.science/paper/2ZCSNOQ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02356&json=true","fetch_graph":"https://pith.science/api/pith-number/2ZCSNOQ5JLC757NPFFCY27S4KW/graph.json","fetch_events":"https://pith.science/api/pith-number/2ZCSNOQ5JLC757NPFFCY27S4KW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW/action/storage_attestation","attest_author":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW/action/author_attestation","sign_citation":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW/action/citation_signature","submit_replication":"https://pith.science/pith/2ZCSNOQ5JLC757NPFFCY27S4KW/action/replication_record"}},"created_at":"2026-07-05T11:55:19.734011+00:00","updated_at":"2026-07-05T11:55:19.734011+00:00"}