{"id":"W3214660453","doi":"10.7202/1083182ar","title":"Analytic rubric scoring versus comparative judgment: a comparison of two approaches to assessing spoken-language interpreting","year":2021,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Interpreting and Communication in Healthcare","field":"Health Professions","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Rubric; Set (abstract data type); Reliability (semiconductor); Psychology; Natural language processing; Rank (graph theory); Computer science; Quality (philosophy); Artificial intelligence; Linguistics; Cognitive psychology; Mathematics education; Mathematics; Epistemology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.156473,0.001722188,0.001646645,0.01307496,0.002352814,0.005171672,0.0033853,0.001275976,0.0033271],"category_scores_gemma":[0.3740647,0.0008560299,0.001565913,0.007597732,0.005745391,0.006181885,0.005275158,0.002480166,0.0007756182],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004247809,"about_ca_system_score_gemma":0.004236004,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005670989,"about_ca_topic_score_gemma":0.01406588,"domain_scores_codex":[0.6985718,0.2189228,0.0153472,0.01114871,0.05423117,0.001778276],"domain_scores_gemma":[0.5100525,0.3708904,0.03176462,0.02104887,0.06188561,0.004358061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003110174,0.001294675,0.06486754,0.005985397,0.0009800649,0.0001843226,0.06620646,0.002449674,0.009695552,0.02644945,0.003951705,0.8148249],"study_design_scores_gemma":[0.002135528,0.02388262,0.5295987,0.009080653,0.002279444,0.00402665,0.1434897,0.08440301,0.03581849,0.08189376,0.07977212,0.003619389],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5580277,0.01048981,0.3256652,0.001975276,0.001204143,0.009718744,0.0006434302,0.001255454,0.09102034],"genre_scores_gemma":[0.6160167,0.002364082,0.3732247,0.0005235437,0.000257647,0.004290522,0.0002698819,0.0002133169,0.002839659],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.156473,"threshold_uncertainty_score":0.827518,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5182363850557781,"score_gpt":0.5173268214474924,"score_spread":0.0009095636082856462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}