{"id":"W2918865870","doi":"","title":"Towards international standards for evaluating machine learning.","year":2019,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; Toronto Rehabilitation Institute","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1535138,0.003159991,0.002569278,0.01892104,0.003452401,0.01630671,0.01098782,0.006467129,0.007383289],"category_scores_gemma":[0.2551908,0.001435915,0.002409469,0.01421361,0.006464452,0.01356856,0.01069194,0.009695997,0.007611118],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006381877,"about_ca_system_score_gemma":0.01545389,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005791544,"about_ca_topic_score_gemma":0.00630708,"domain_scores_codex":[0.847952,0.06380273,0.02835706,0.004585154,0.05282353,0.002479471],"domain_scores_gemma":[0.6902821,0.1039547,0.01477813,0.05332105,0.1298603,0.007803747],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004187065,0.0006680426,0.01424407,0.002257936,0.0002975606,0.0001539618,0.001654554,0.008702926,0.003077587,0.3034811,0.1509644,0.5140792],"study_design_scores_gemma":[0.0002106307,0.001025633,0.02537925,0.009375817,0.0003686538,0.0007321568,0.004076351,0.03872314,0.01450329,0.4118188,0.4935127,0.0002734747],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01161656,0.02123011,0.7927224,0.01668417,0.006039893,0.002739754,0.005143507,0.007105411,0.1367182],"genre_scores_gemma":[0.07469292,0.00629187,0.8835603,0.001966562,0.001065988,0.005902139,0.01208639,0.001534182,0.01289968],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8464862,"threshold_uncertainty_score":0.8118682,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1343354412105469,"score_gpt":0.4156434660308341,"score_spread":0.2813080248202873,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}