{"id":"W2049871613","doi":"10.1007/bf03016408","title":"Expert versus non-expert raters in simulation performance evaluation","year":2008,"lang":"en","type":"article","venue":"Canadian Journal of Anesthesia/Journal canadien d anesthésie","topic":"Simulation Techniques and Applications","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"The Wilson Centre; University of Toronto; Women's College Hospital","funders":"","keywords":"Computer science; Psychology; Artificial intelligence; Natural language processing; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004700339,0.0002713582,0.0004977091,0.001844989,0.0007469764,0.0002925408,0.001046569,0.0001977218,0.0006253713],"category_scores_gemma":[0.0009929372,0.0002319423,0.000232739,0.001431927,0.0002111162,0.001226695,0.000009815419,0.0005532881,0.00004815708],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001357305,"about_ca_system_score_gemma":0.004477527,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00426236,"about_ca_topic_score_gemma":0.005301322,"domain_scores_codex":[0.9955331,0.0002700721,0.001554847,0.0003160577,0.001687393,0.0006385283],"domain_scores_gemma":[0.9948039,0.0004577761,0.0009352639,0.0004250285,0.002060786,0.00131722],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"case_report","study_design_scores_codex":[0.0009869797,0.0001698179,0.06845392,0.000008158622,0.00009795823,0.1198331,0.02194331,0.5638486,0.0002174116,0.0003013799,0.0331875,0.1909519],"study_design_scores_gemma":[0.005602316,0.001662258,0.1616801,0.0002411974,0.00005161737,0.6123326,0.003349879,0.1325558,0.0005067302,0.001358146,0.07956581,0.001093557],"study_design_candidate":"case_report","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.994762,0.0004651132,0.0003806979,0.002828494,0.0004170281,0.0002711656,0.000001539396,0.000008466218,0.0008655126],"genre_scores_gemma":[0.9981589,0.0001139628,0.0007968316,0.0001521432,0.000508533,0.0000102242,0.000003784069,0.00003006928,0.0002256066],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4924996,"threshold_uncertainty_score":0.9458331,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0959764257871998,"score_gpt":0.3544521935698364,"score_spread":0.2584757677826366,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}