{"id":"W2791245005","doi":"10.5430/wje.v8n1p111","title":"Effects of Analytical and Holistic Scoring Patterns on Scorer Reliability in Biology Essay Tests","year":2018,"lang":"en","type":"article","venue":"World Journal of Education","topic":"Digital Imaging for Blood Diseases","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Pearson product-moment correlation coefficient; Reliability (semiconductor); Test (biology); Data collection; Moment (physics); Scoring system; Mathematics education; Biology; Psychology; Statistics; Medicine; Mathematics; Internal medicine; Ecology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03078958,0.000467037,0.0005462356,0.000924426,0.0007965906,0.00135763,0.0005816416,0.0004411369,0.001907393],"category_scores_gemma":[0.1758454,0.0004940847,0.0006232873,0.0009940369,0.001179192,0.001278571,0.001465214,0.0007109549,0.0003465929],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005948073,"about_ca_system_score_gemma":0.0008674852,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001445811,"about_ca_topic_score_gemma":0.002684815,"domain_scores_codex":[0.9506025,0.02596101,0.004870618,0.003962371,0.01359365,0.001009827],"domain_scores_gemma":[0.7383916,0.1877019,0.03284566,0.01699918,0.02140476,0.002656924],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001600176,0.0003541015,0.9154124,0.0001663089,0.0002648724,0.0001634737,0.0093756,0.000554421,0.003089125,0.0003388331,0.0006467365,0.06803381],"study_design_scores_gemma":[0.00003856823,0.001568504,0.9870727,0.0001015713,0.0001243436,0.0002504915,0.004480487,0.00206149,0.002894986,0.0004241224,0.0009328616,0.00004990813],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9954287,0.0001989935,0.002076463,0.0001276832,0.00002973523,0.0000564078,0.00003398105,0.00002536846,0.002022607],"genre_scores_gemma":[0.9980277,0.00005889991,0.001253807,0.00002517481,0.00001240514,0.00004649224,0.00004451277,0.00001317705,0.000517763],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03078958,"threshold_uncertainty_score":0.1628327,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01652793913598873,"score_gpt":0.3327994318309083,"score_spread":0.3162714926949196,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}