{"id":"W4379106406","doi":"10.1101/2023.05.29.542750","title":"Classifying high-dimensional phenotypes with ensemble learning","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Morphological variations and asymmetry","field":"Mathematics","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Victoria; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; National Institutes of Health","keywords":"Artificial intelligence; Ensemble learning; Machine learning; Computer science; Linear discriminant analysis; Preprocessor; Class (philosophy); Binary classification; Set (abstract data type); Task (project management); Pattern recognition (psychology); Support vector machine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0008270134,0.0005816326,0.0007235015,0.0002882277,0.0003642328,0.0002228223,0.0004316327,0.0005951669,0.0001589854],"category_scores_gemma":[0.000776505,0.0004947364,0.0001462557,0.0006136338,0.00009786322,0.0001042733,0.0006383888,0.001452736,0.0002799117],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000186136,"about_ca_system_score_gemma":0.0003384218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006987575,"about_ca_topic_score_gemma":0.000004189467,"domain_scores_codex":[0.9969524,0.0002072271,0.0005637945,0.001040865,0.0006070848,0.0006286472],"domain_scores_gemma":[0.9973302,0.0005236309,0.000551174,0.0009500162,0.0004135564,0.0002313462],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.0003155647,0.001645847,0.03509441,0.003519042,0.002469365,0.001545304,0.00006702011,0.008146923,0.6360394,0.2916385,0.01949503,0.00002358272],"study_design_scores_gemma":[0.01024428,0.001783253,0.5745406,0.01209265,0.003860627,6.627688e-7,0.0001046821,0.02943538,0.3196406,0.006546419,0.02528995,0.0164609],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9416916,0.0002049989,0.05319852,0.0005134928,0.001398922,0.0007807153,0.0000889532,0.002056087,0.00006668249],"genre_scores_gemma":[0.8898205,0.00002809859,0.1091518,0.0001005926,0.0004391922,0.0001472679,8.293192e-7,0.0001991374,0.0001125164],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5394462,"threshold_uncertainty_score":0.9997504,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04269285077485603,"score_gpt":0.2526215481243801,"score_spread":0.2099286973495241,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}