{"id":"W3200503707","doi":"10.2196/25110","title":"Predicting the Easiness and Complexity of English Health Materials for International Tertiary Students With Linguistically Enhanced Machine Learning Algorithms: Development and Validation Study","year":2021,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Health Literacy and Information Accessibility","field":"Health Professions","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Interpretability; Random forest; Machine learning; Artificial intelligence; Support vector machine; Computer science; Feature selection; Gradient boosting; Boosting (machine learning); Algorithm","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00900428,0.0009162744,0.0006621106,0.001308187,0.000272999,0.0007963075,0.0008516825,0.000736577,0.00109525],"category_scores_gemma":[0.0229904,0.000241817,0.001131259,0.0007745243,0.0003725984,0.0007361741,0.001057248,0.0009214099,0.0005606349],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007118467,"about_ca_system_score_gemma":0.0009317491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002149031,"about_ca_topic_score_gemma":0.00178618,"domain_scores_codex":[0.9976414,0.001233271,0.0002390954,0.0004319482,0.0003413379,0.0001129608],"domain_scores_gemma":[0.9842625,0.01025291,0.000957455,0.001025193,0.003177145,0.0003248522],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003048134,0.006296329,0.5850747,0.0005254128,0.0007913741,0.0003055169,0.0012707,0.05221225,0.01192827,0.0006204275,0.002470562,0.3354564],"study_design_scores_gemma":[0.0002772256,0.004128875,0.3153639,0.0001927956,0.0004936567,0.000346501,0.0007025164,0.6573405,0.01841372,0.0007053878,0.001947484,0.00008731097],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9828318,0.0001611375,0.01554199,0.00005611794,0.00001917091,0.0003055108,0.0003196877,0.000125198,0.0006392992],"genre_scores_gemma":[0.9618343,0.000104708,0.03583379,0.00003968444,0.00001494082,0.0004502106,0.001216166,0.00002464476,0.0004816064],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00900428,"threshold_uncertainty_score":0.04761976,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05936797026151921,"score_gpt":0.4488036022491286,"score_spread":0.3894356319876094,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}