{"id":"W4316813279","doi":"10.3390/jimaging9020020","title":"Verification, Evaluation, and Validation: Which, How &amp; Why, in Medical Augmented Reality System Design","year":2023,"lang":"en","type":"article","venue":"Journal of Imaging","topic":"Human-Automation Interaction and Safety","field":"Psychology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Augmented reality; Virtual reality; Human–computer interaction; Key (lock); Observer (physics); Software engineering; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1026314,0.001372226,0.001728955,0.002532024,0.0019535,0.01028993,0.002454677,0.003870348,0.001750293],"category_scores_gemma":[0.1758168,0.001317967,0.001832241,0.001483304,0.01551639,0.01851089,0.004959889,0.005109013,0.0006208015],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003693023,"about_ca_system_score_gemma":0.006778433,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003568881,"about_ca_topic_score_gemma":0.001953569,"domain_scores_codex":[0.8697819,0.08747958,0.008164955,0.006815345,0.02577586,0.001982414],"domain_scores_gemma":[0.8147315,0.1363759,0.009899817,0.01480668,0.02290173,0.001284297],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006044607,0.0002151872,0.00803985,0.002927492,0.0003918415,0.0002452098,0.004891712,0.01714153,0.005995381,0.4426736,0.003059525,0.5138142],"study_design_scores_gemma":[0.000314153,0.002379803,0.008317846,0.006332713,0.0005048024,0.001623091,0.003773618,0.1170817,0.02420106,0.7771704,0.05771236,0.0005884715],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01617268,0.01728546,0.9444932,0.01083642,0.0006345487,0.0005038937,0.00007325182,0.0003548643,0.009645678],"genre_scores_gemma":[0.5262544,0.006936792,0.4606719,0.002141595,0.000794037,0.0009333037,0.0001059796,0.0002700084,0.001892062],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1026314,"threshold_uncertainty_score":0.5427733,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1014873274066052,"score_gpt":0.4263075144564643,"score_spread":0.324820187049859,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}