{"id":"W4400889427","doi":"10.21203/rs.3.rs-4646752/v1","title":"Optimizing Model Performance and Interpretability: an application to biological data classification","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Metabolomics and Mass Spectrometry Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"Southern University of Science and Technology; National Natural Science Foundation of China","keywords":"Interpretability; Computer science; Artificial intelligence; Machine learning; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01613333,0.002176978,0.002238777,0.002086009,0.0009135451,0.00340324,0.002102284,0.002896482,0.001572929],"category_scores_gemma":[0.05084084,0.0008110359,0.001970709,0.001852977,0.001080662,0.002619321,0.002245335,0.003301318,0.0004595385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001841221,"about_ca_system_score_gemma":0.002008461,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007555643,"about_ca_topic_score_gemma":0.004324035,"domain_scores_codex":[0.9956263,0.002427317,0.0003342753,0.0007648789,0.0006202241,0.0002269206],"domain_scores_gemma":[0.9664569,0.02727214,0.001219969,0.002491232,0.002132272,0.0004274545],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000940873,0.0004813826,0.009505834,0.0001998308,0.0004209105,0.0002293021,0.0001950231,0.7093971,0.006575943,0.003714258,0.005026719,0.2633127],"study_design_scores_gemma":[0.00001544967,0.00002989372,0.0004162279,0.000005870674,0.00001933835,0.00001464628,0.000008145271,0.9950777,0.001198418,0.003094648,0.0001132699,0.000006352442],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1420474,0.001085957,0.8476515,0.001591935,0.0001283276,0.0001549897,0.0006002029,0.005459498,0.001280141],"genre_scores_gemma":[0.6179386,0.000314461,0.3766968,0.0002944656,0.0001564987,0.0001926458,0.001384328,0.001311818,0.001710296],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01613333,"threshold_uncertainty_score":0.08532226,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1544728795180428,"score_gpt":0.4294303644014495,"score_spread":0.2749574848834067,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}