{"id":"W4239260231","doi":"10.31219/osf.io/mxr8s","title":"A thorough evaluation of the Language Environment Analysis (LENATM) system","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Language Development and Disorders","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba","funders":"Economic and Social Research Council","keywords":"Set (abstract data type); Psychology; Language acquisition; Key (lock); Computer science; Annotation; Natural language processing; Mathematics education; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001144424,0.0002051152,0.0003835614,0.0001569168,0.00002895957,0.0000170235,0.0004545732,0.0002675382,0.008465831],"category_scores_gemma":[0.00002168802,0.0001279966,0.000388053,0.0002182712,0.00003535747,0.00001620382,0.0004253887,0.00021275,0.0003770838],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001877336,"about_ca_system_score_gemma":0.00009452679,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001055136,"about_ca_topic_score_gemma":0.0001639601,"domain_scores_codex":[0.997734,0.0005023209,0.0003760368,0.0004485943,0.0007610979,0.0001780129],"domain_scores_gemma":[0.9982106,0.00004139975,0.0003483962,0.001319243,0.00005539419,0.00002496922],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000241903,0.001455139,0.4194221,0.001708655,0.04372583,0.0000289732,0.3143334,0.07234955,0.0003898917,0.02311852,0.01686856,0.1063575],"study_design_scores_gemma":[0.003202127,0.00005339937,0.851684,0.0002367516,0.02060249,0.000005199568,0.1010374,0.01937656,0.0006514749,0.0005174979,0.001450559,0.001182509],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7553321,0.002038041,0.001583557,0.0001395354,0.001184722,0.00128801,0.00004252128,0.00005197232,0.2383396],"genre_scores_gemma":[0.9875317,0.00000469951,0.0002123012,0.00006276197,0.00004057645,0.0001516446,0.00014371,0.00001658424,0.01183605],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4322619,"threshold_uncertainty_score":0.9924406,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02782759158065745,"score_gpt":0.3177921741999969,"score_spread":0.2899645826193394,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}