{"id":"W4283781120","doi":"10.1101/2022.06.29.498112","title":"A scalable open-source framework for machine learning based image collection, annotation and classification: a case study for automatic fish species identification","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Water Quality Monitoring Technologies","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Positive Living North","funders":"","keywords":"Computer science; Data collection; Identification (biology); Scalability; Citizen science; Data science; Convolutional neural network; Modular design; Automatic image annotation; Artificial intelligence; Image processing; Database; Ecology; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002915332,0.001353391,0.0008200607,0.002037111,0.0008293611,0.002090626,0.003615788,0.001591101,0.007872467],"category_scores_gemma":[0.005137374,0.0008567063,0.001696872,0.001195242,0.001151988,0.002785093,0.004139239,0.001828401,0.007427585],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001038447,"about_ca_system_score_gemma":0.001713408,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008294731,"about_ca_topic_score_gemma":0.01169955,"domain_scores_codex":[0.9983335,0.0002141811,0.0001328518,0.0004685609,0.000667473,0.0001834123],"domain_scores_gemma":[0.9979929,0.0005043009,0.000121402,0.0005690061,0.0005608933,0.0002515708],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001999749,0.00103773,0.00791158,0.00198781,0.0004387361,0.003105387,0.001791039,0.04199865,0.09990088,0.02573168,0.1979223,0.6161746],"study_design_scores_gemma":[0.0003032882,0.0002917423,0.008312472,0.0004550517,0.0001166869,0.001647229,0.0003615726,0.6451809,0.1014693,0.02411334,0.2174025,0.0003458564],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007520373,0.0003703833,0.8005107,0.000453172,0.0001159622,0.0005969466,0.003270651,0.1846428,0.002519067],"genre_scores_gemma":[0.10395,0.0004836181,0.8495421,0.0005099755,0.00007201804,0.0009954701,0.02029151,0.01599367,0.008161648],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008294731,"threshold_uncertainty_score":0.02633601,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04250894668956279,"score_gpt":0.2819618623967635,"score_spread":0.2394529157072007,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}