{"id":"W4283719127","doi":"10.1101/2022.06.26.497684","title":"Improving the workflow to crack Small, Unbalanced, Noisy, but Genuine (SUNG) datasets in bioacoustics: the case of bonobo calls","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Animal Vocal Communication and Behavior","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Chicoutimi","funders":"LabEx ASLAN; Université de Lyon; Social Sciences and Humanities Research Council of Canada; Institut Universitaire de France; Agence Nationale de la Recherche","keywords":"Bonobo; Computer science; Workflow; Artificial intelligence; Robustness (evolution); Cluster analysis; Machine learning; Feature vector; Support vector machine; Feature (linguistics); Pattern recognition (psychology); Biology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004595868,0.001516678,0.001033552,0.00231644,0.001612445,0.002499796,0.002006434,0.002030499,0.005017183],"category_scores_gemma":[0.0175926,0.0005192398,0.001698223,0.001224175,0.001149916,0.002008733,0.003401904,0.002704265,0.005812251],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000596141,"about_ca_system_score_gemma":0.001636008,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004292461,"about_ca_topic_score_gemma":0.00721128,"domain_scores_codex":[0.9973806,0.0007648402,0.000219958,0.0008318335,0.0005571909,0.0002456553],"domain_scores_gemma":[0.9931176,0.002449837,0.0004168923,0.001717635,0.001870424,0.0004276446],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001207385,0.0006007716,0.02567345,0.0005693869,0.0003309677,0.001275931,0.001929749,0.07158212,0.09970458,0.006275181,0.02814525,0.7627052],"study_design_scores_gemma":[0.00006242426,0.0001514743,0.008086817,0.00008438619,0.00004131445,0.000246186,0.0009830353,0.922908,0.04025575,0.0147123,0.01239181,0.00007660039],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08463919,0.000231192,0.8732386,0.0009461045,0.0003224055,0.0003604798,0.001374859,0.03729235,0.001594855],"genre_scores_gemma":[0.205916,0.00008069251,0.785052,0.0003729914,0.00007923901,0.0003960379,0.00350698,0.002711928,0.001884029],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005017183,"threshold_uncertainty_score":0.02430552,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01996363930098752,"score_gpt":0.2550388992871567,"score_spread":0.2350752599861692,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}