{"id":"W138034719","doi":"","title":"Constructing new and better evaluation measures for machine learning","year":2007,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University; University of Ottawa","funders":"","keywords":"Computer science; Machine learning; Artificial intelligence; Measure (data warehouse); Construct (python library); Artificial neural network; Greedy algorithm; Mean squared error; Data mining; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04810335,0.002485279,0.003315483,0.008585648,0.001075199,0.005407913,0.002541433,0.003565749,0.001118758],"category_scores_gemma":[0.1634883,0.0006835099,0.001742875,0.005946608,0.003279501,0.01265139,0.003286178,0.004730194,0.000498291],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00302507,"about_ca_system_score_gemma":0.002079384,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009330821,"about_ca_topic_score_gemma":0.0008992004,"domain_scores_codex":[0.9581048,0.01982046,0.005321661,0.004023315,0.01195809,0.0007717048],"domain_scores_gemma":[0.8331583,0.1083535,0.01522611,0.02069158,0.02051237,0.002058187],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004993595,0.0007771702,0.02156008,0.001701464,0.0009576891,0.0001566023,0.000592501,0.2176818,0.01031557,0.2026711,0.009006666,0.53408],"study_design_scores_gemma":[0.0001050741,0.0009177722,0.006869111,0.0003830648,0.0001985844,0.0002383758,0.0002373242,0.787868,0.009457588,0.1856552,0.007845814,0.0002242195],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02126337,0.002717753,0.9732134,0.0004980438,0.0002449896,0.0002039453,0.0001428915,0.0004395772,0.001275989],"genre_scores_gemma":[0.2007523,0.0007010382,0.7961552,0.0002640659,0.0003460078,0.0005521149,0.0006228219,0.0002759378,0.0003306391],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04810335,"threshold_uncertainty_score":0.2543979,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04615306708286241,"score_gpt":0.3140166668024245,"score_spread":0.267863599719562,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}