{"id":"W4393086187","doi":"10.1038/s41592-024-02234-5","title":"Comparing classifier performance with baselines","year":2024,"lang":"en","type":"article","venue":"Nature Methods","topic":"Statistical Methods and Applications","field":"Mathematics","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canada's Michael Smith Genome Sciences Centre","funders":"","keywords":"Classifier (UML); Computer science; Computational biology; George (robot); Artificial intelligence; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01122588,0.002438251,0.002463674,0.004546127,0.001925234,0.004670191,0.002609224,0.003862876,0.007157236],"category_scores_gemma":[0.03042899,0.0004884313,0.002296633,0.002453328,0.0008267644,0.005309895,0.002164583,0.003496455,0.01085123],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002031735,"about_ca_system_score_gemma":0.002246539,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007743906,"about_ca_topic_score_gemma":0.008913456,"domain_scores_codex":[0.9889094,0.002587204,0.000857659,0.003318711,0.003531907,0.0007951726],"domain_scores_gemma":[0.9793402,0.01086032,0.0007524505,0.003782522,0.004561835,0.0007026619],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007379336,0.001863606,0.03698742,0.002348376,0.003808141,0.0003792493,0.0003468104,0.04595366,0.01662059,0.004069855,0.1388447,0.7413982],"study_design_scores_gemma":[0.001007011,0.007017443,0.05180236,0.0008020736,0.003709617,0.002467763,0.001832697,0.7398177,0.07267097,0.03106404,0.0872894,0.0005188522],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6240066,0.05782833,0.1876273,0.007501769,0.01368815,0.001092663,0.0338253,0.02739528,0.04703464],"genre_scores_gemma":[0.8265869,0.004889776,0.08598384,0.001258176,0.001977765,0.0003812168,0.0586176,0.001685643,0.01861914],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9887741,"threshold_uncertainty_score":0.05936885,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1483906697713013,"score_gpt":0.5096182819771905,"score_spread":0.3612276122058892,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}