{"id":"W3133424709","doi":"10.1109/iros45743.2020.9340905","title":"Understanding Contexts Inside Robot and Human Manipulation Tasks through Vision-Language Model and Ontology System in Video Streams","year":2020,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Ontology; Robot; Scheme (mathematics); Artificial intelligence; Human–computer interaction; Process (computing); Object (grammar); Mobile robot; Computer vision","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004848038,0.0004765002,0.0003022207,0.0008403755,0.0003613598,0.001013841,0.0007471303,0.0007380381,0.001301382],"category_scores_gemma":[0.002257382,0.0001907002,0.0007503844,0.0006543644,0.0005336054,0.002272192,0.001017072,0.0009307186,0.000261291],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001294641,"about_ca_system_score_gemma":0.001051233,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03030813,"about_ca_topic_score_gemma":0.03186375,"domain_scores_codex":[0.9996156,0.00007171397,0.00001834191,0.0001693341,0.00007208465,0.00005289803],"domain_scores_gemma":[0.9995937,0.0001435619,0.00005833241,0.00007515882,0.00008533293,0.00004386393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001129326,0.0008439869,0.02691825,0.0004208089,0.0001963688,0.00124562,0.00230862,0.2499876,0.09413972,0.02928099,0.008367439,0.5851613],"study_design_scores_gemma":[0.00001374215,0.0000648069,0.008964447,0.00002377188,0.00004069405,0.0001385206,0.0004014184,0.962418,0.01354029,0.01122122,0.003142722,0.00003048134],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3568425,0.0008650151,0.6336544,0.0006894585,0.0001077693,0.0002227065,0.001391544,0.001987941,0.00423885],"genre_scores_gemma":[0.8921807,0.0003362276,0.1038128,0.0001224794,0.00002510769,0.00009451673,0.001618522,0.00007568755,0.001734079],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03030813,"threshold_uncertainty_score":0.06026345,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08615830094156307,"score_gpt":0.3290150866216928,"score_spread":0.2428567856801297,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}