{"id":"W4401414239","doi":"10.1109/icra57147.2024.10611647","title":"CLIPUNetr: Assisting Human-robot Interface for Uncalibrated Visual Servoing Control with CLIP-driven Referring Expression Segmentation","year":2024,"lang":"en","type":"article","venue":"","topic":"Visual Attention and Saliency Detection","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Visual servoing; Computer vision; Computer science; Artificial intelligence; Robot; Segmentation; Interface (matter); Expression (computer science); Image segmentation; Human–robot interaction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003165003,0.0008205029,0.0003664707,0.0003532256,0.0002706997,0.0004469902,0.001352653,0.0005200266,0.003931677],"category_scores_gemma":[0.001253832,0.0002679267,0.0003037282,0.000191132,0.0005215304,0.0006285281,0.0009086442,0.0007060553,0.000795783],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005270678,"about_ca_system_score_gemma":0.00072727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004087816,"about_ca_topic_score_gemma":0.005373797,"domain_scores_codex":[0.9997442,0.00002725974,0.000007981056,0.00009656899,0.00009195189,0.00003195032],"domain_scores_gemma":[0.9996618,0.0001068943,0.00003238281,0.00005766516,0.00009808968,0.00004315121],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007162743,0.0001813163,0.001057134,0.0002392024,0.00005319871,0.0004738579,0.0004562474,0.05509507,0.3797734,0.005700377,0.014887,0.541367],"study_design_scores_gemma":[0.00003617964,0.0002452478,0.001021299,0.00001414728,0.00002571929,0.000189267,0.00005582101,0.8526484,0.1346108,0.002037121,0.00908022,0.00003564591],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02426865,0.0001519549,0.9598351,0.00009120422,0.0000805942,0.0001150337,0.0001190123,0.01223869,0.003099858],"genre_scores_gemma":[0.4833866,0.0001400012,0.5064846,0.0003742078,0.00005281997,0.0002193136,0.0005724607,0.0009655678,0.007804473],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004087816,"threshold_uncertainty_score":0.01315278,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03028229144140852,"score_gpt":0.3464324341251153,"score_spread":0.3161501426837068,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}