{"id":"W7026617208","doi":"","title":"Accounting for variance and hyperparameter optimization in machine learning benchmarks","year":2022,"lang":"fr","type":"dissertation","venue":"Papyrus : Institutional Repository (Université de Montréal)","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Agence Nationale de la Recherche; Compute Canada; Canadian Institute for Advanced Research","keywords":"Test (biology); Decision tree","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009195318,0.0012567,0.001181163,0.001619469,0.0007578589,0.003467512,0.002271831,0.001485043,0.002228514],"category_scores_gemma":[0.06123414,0.0005499468,0.0009308459,0.001921183,0.0008725879,0.002889404,0.00188836,0.002185017,0.0006243034],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001774245,"about_ca_system_score_gemma":0.002539597,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007385991,"about_ca_topic_score_gemma":0.01046806,"domain_scores_codex":[0.9892036,0.005853209,0.0006566527,0.001342414,0.00234566,0.0005983794],"domain_scores_gemma":[0.9710363,0.01828581,0.001437568,0.005527726,0.003364633,0.0003480272],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003102462,0.0001779582,0.01545887,0.0002174921,0.0002022989,0.0001054877,0.0001964578,0.7078291,0.004371388,0.02322349,0.004517219,0.24339],"study_design_scores_gemma":[0.00003189762,0.0001442765,0.004721157,0.00005704818,0.00004554346,0.00006188153,0.00008310258,0.9646168,0.006257769,0.0196122,0.004334779,0.00003339391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2569033,0.004401696,0.7128314,0.001888359,0.0004257933,0.0002502785,0.000905824,0.004848131,0.01754519],"genre_scores_gemma":[0.8316839,0.0004762014,0.1599129,0.0003072015,0.0001503458,0.0003294736,0.00130956,0.001162791,0.004667648],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009195318,"threshold_uncertainty_score":0.04863006,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00776089845587945,"score_gpt":0.1938290518964026,"score_spread":0.1860681534405232,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}