{"id":"W4283781120","doi":"10.1101/2022.06.29.498112","title":"A scalable open-source framework for machine learning based image collection, annotation and classification: a case study for automatic fish species identification","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Water Quality Monitoring Technologies","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Positive Living North","funders":"","keywords":"Computer science; Data collection; Identification (biology); Scalability; Citizen science; Data science; Convolutional neural network; Modular design; Automatic image annotation; Artificial intelligence; Image processing; Database; Ecology; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.002049107,0.0004042164,0.0004446698,0.0002182045,0.001307707,0.00100849,0.0007862325,0.0002846109,0.0001132686],"category_scores_gemma":[0.001667844,0.0004713227,0.00008319358,0.0007489129,0.0002030108,0.000377191,0.001253605,0.0006102995,0.000007636027],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008737749,"about_ca_system_score_gemma":0.000108375,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004366105,"about_ca_topic_score_gemma":0.00003197048,"domain_scores_codex":[0.9969892,0.0003160825,0.0006848063,0.001231448,0.0003914716,0.0003869755],"domain_scores_gemma":[0.9974285,0.0004803081,0.0007524119,0.00108043,0.0001556985,0.0001025931],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.0004152721,0.003620677,0.463688,0.002953935,0.0005748284,0.0001944951,0.001781701,0.009357224,0.5107961,0.0007944658,0.005712418,0.0001107805],"study_design_scores_gemma":[0.003379627,0.001047754,0.5995501,0.0003634799,0.0006425536,8.088566e-7,0.002091874,0.2543555,0.1232698,0.0002113958,0.01268238,0.00240474],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8790273,0.00003273634,0.1135372,0.0007378915,0.0004672128,0.005120508,0.0002773975,0.0007983107,0.000001430287],"genre_scores_gemma":[0.8862222,0.00001112367,0.1072993,0.00004501466,0.00008090804,0.006172352,0.000004183282,0.0001033993,0.00006154375],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3875264,"threshold_uncertainty_score":0.9999924,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04250894668956279,"score_gpt":0.2819618623967635,"score_spread":0.2394529157072007,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}