{"id":"W4213141158","doi":"10.7820/vli.v10.2.mclean","title":"The internal consistency and accuracy of automatically scored written receptive meaning-recall data: A preliminary study","year":2021,"lang":"en","type":"article","venue":"Vocabulary Learning and Instruction","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"Japan Society for the Promotion of Science","keywords":"Recall; Consistency (knowledge bases); Meaning (existential); Internal consistency; Computer science; Natural language processing; Precision and recall; Psychology; Artificial intelligence; Cognitive psychology; Psychometrics; Developmental psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0372174,0.0006705064,0.001060105,0.002276311,0.0007193866,0.002048883,0.001420195,0.001048167,0.001615588],"category_scores_gemma":[0.08160827,0.0005987172,0.002269783,0.001420014,0.00150849,0.001605519,0.00174827,0.001238017,0.0008611353],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006630218,"about_ca_system_score_gemma":0.000643788,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009691895,"about_ca_topic_score_gemma":0.001365062,"domain_scores_codex":[0.9741002,0.008392019,0.004314699,0.003430783,0.00914095,0.0006212521],"domain_scores_gemma":[0.8820125,0.0718283,0.01281792,0.009709579,0.02242536,0.001206283],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001165309,0.001004264,0.915759,0.0005284944,0.001021986,0.0001817226,0.0112631,0.001439107,0.003138537,0.001039757,0.001579048,0.06187969],"study_design_scores_gemma":[0.0001253022,0.001835084,0.9773406,0.0002538093,0.000328259,0.0005727464,0.00304664,0.007226387,0.004041378,0.001519408,0.003604031,0.0001063253],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9861333,0.0005055677,0.007948698,0.0001428496,0.0001351656,0.0005129925,0.0004827136,0.0001087595,0.004029976],"genre_scores_gemma":[0.9908842,0.0001400447,0.006257243,0.0000974869,0.00005118972,0.0008335109,0.0007545438,0.00007136063,0.0009105337],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9627826,"threshold_uncertainty_score":0.1968268,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02340135669395839,"score_gpt":0.3193501383761505,"score_spread":0.2959487816821921,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}