{"id":"W2918865870","doi":"","title":"Towards international standards for evaluating machine learning.","year":2019,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; Toronto Rehabilitation Institute","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001077568,0.000140458,0.0001329612,0.0001811172,0.0001980799,0.0002624878,0.0008027729,0.00007796086,0.000638304],"category_scores_gemma":[0.0003947166,0.0001385834,0.00009191778,0.0002938344,0.00004125686,0.0002831087,0.0001116025,0.0002235453,0.0002582564],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002019142,"about_ca_system_score_gemma":0.0004417399,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001954518,"about_ca_topic_score_gemma":0.00000955731,"domain_scores_codex":[0.997936,0.00004609539,0.0003579776,0.0004672111,0.001000585,0.0001922042],"domain_scores_gemma":[0.9976951,0.0001740492,0.0001652877,0.0002178585,0.001685662,0.00006197292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002581064,0.0000412703,0.00001731188,0.000003230868,0.000007379256,1.185167e-7,0.00005108702,0.001223493,0.001941206,0.7321508,0.00008279204,0.2644555],"study_design_scores_gemma":[0.00003814025,0.0002923552,0.00005393883,0.0000158642,0.000001762205,0.000002159572,0.00003273449,0.7093372,0.02709944,0.2476254,0.01535226,0.0001487412],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001356094,0.00001010592,0.9560742,0.00352112,0.0003957706,0.0004500326,0.00006874726,0.0002338845,0.03789008],"genre_scores_gemma":[0.9546888,0.00001797125,0.0437584,0.0003920937,0.00009794578,0.0001638183,0.0000313881,0.000009902954,0.0008396204],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9533328,"threshold_uncertainty_score":0.6988981,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1343354412105469,"score_gpt":0.4156434660308341,"score_spread":0.2813080248202873,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}