{"id":"W2259311060","doi":"10.1093/bib/bbw022","title":"The BRaliBase dent—a tale of benchmark design and interpretation","year":2016,"lang":"en","type":"article","venue":"Briefings in Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Agence Nationale de la Recherche","keywords":"Benchmark (surveying); Computer science; Interpretation (philosophy); Range (aeronautics); Artificial intelligence; Machine learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000219065,0.00006790282,0.00007225453,0.00001929202,0.00004826519,0.00001350228,0.00009059167,0.00004137066,0.000001089999],"category_scores_gemma":[0.0001217757,0.00004040424,0.00002256099,0.0000335299,0.000126398,0.000001918479,0.00009054685,0.00001810454,9.790466e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000005066865,"about_ca_system_score_gemma":0.00002082464,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001064692,"about_ca_topic_score_gemma":0.00001346368,"domain_scores_codex":[0.9995226,0.00001812362,0.0002280268,0.00006647948,0.00005345335,0.0001112936],"domain_scores_gemma":[0.9996476,0.00006971817,0.00009809027,0.0001297091,0.00003488736,0.00001999207],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004240436,0.00006783974,0.01898636,0.0001585044,0.0001717937,0.000001334856,0.002925389,0.0001885813,0.5643573,0.002653316,0.01070705,0.3993585],"study_design_scores_gemma":[0.004927991,0.001609602,0.07959445,0.0004565602,0.00007851354,0.00008298104,0.001480865,0.01759644,0.755666,0.008355215,0.1290408,0.001110602],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9648663,0.001135026,0.03256066,0.00068255,0.00007359035,0.0002120836,0.0000109887,0.00000185336,0.0004569791],"genre_scores_gemma":[0.9869783,0.001388562,0.01130948,0.0002097886,0.00001298267,0.00000938696,0.000002383957,0.000005208205,0.00008393979],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3982479,"threshold_uncertainty_score":0.1647637,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007002299377866998,"score_gpt":0.2156237908336125,"score_spread":0.2086214914557455,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}