{"id":"W4296252071","doi":"10.1101/2022.09.16.508347","title":"Supervised Machine Learning Enables Geospatial Microbial Provenance","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Microbial Community Ecology and Physiology","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institutes of Health; York University; Vallee Foundation; WorldQuant Foundation; Pershing Square Foundation; Alfred P. Sloan Foundation","keywords":"Geospatial analysis; Metadata; Metagenomics; Citizen science; Random forest; Computer science; Identification (biology); Classifier (UML); Machine learning; Geography; Artificial intelligence; Cartography; Ecology; Biology; World Wide Web","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004728697,0.0006040579,0.0007772926,0.002438932,0.0008249843,0.002166941,0.001659492,0.0009325165,0.001238317],"category_scores_gemma":[0.01716251,0.0004302363,0.001041585,0.002503375,0.0009142935,0.002631601,0.002332315,0.001702146,0.001020337],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001006652,"about_ca_system_score_gemma":0.001395417,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00476461,"about_ca_topic_score_gemma":0.005961687,"domain_scores_codex":[0.9973507,0.0008547684,0.0002227697,0.0008650591,0.0005700816,0.0001365322],"domain_scores_gemma":[0.98808,0.004807012,0.001241772,0.003557891,0.002067387,0.0002459838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006244452,0.0004265198,0.06131245,0.0006092984,0.0003393704,0.0005622405,0.0005943318,0.380558,0.01341415,0.02647922,0.01793152,0.4971485],"study_design_scores_gemma":[0.00002211033,0.00002865542,0.001713428,0.00003844885,0.00001712955,0.00006222227,0.0000664791,0.9600585,0.004438998,0.02923698,0.004300433,0.00001654588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07809613,0.001036575,0.8995731,0.001038025,0.000287383,0.0001352451,0.003862622,0.01331892,0.002652073],"genre_scores_gemma":[0.5143461,0.0002909209,0.4758744,0.0002700391,0.0001343043,0.0001587747,0.007530398,0.0004186554,0.0009764609],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00476461,"threshold_uncertainty_score":0.02500802,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009540743175552879,"score_gpt":0.194951196830604,"score_spread":0.1854104536550512,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}