{"id":"W2259311060","doi":"10.1093/bib/bbw022","title":"The BRaliBase dent—a tale of benchmark design and interpretation","year":2016,"lang":"en","type":"article","venue":"Briefings in Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Agence Nationale de la Recherche","keywords":"Benchmark (surveying); Computer science; Interpretation (philosophy); Range (aeronautics); Artificial intelligence; Machine learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06699935,0.002892294,0.00216348,0.008055306,0.00363262,0.01315886,0.00854652,0.004080761,0.006421804],"category_scores_gemma":[0.170052,0.001711215,0.001984565,0.004125083,0.009272743,0.01224266,0.01132739,0.01194439,0.0034558],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004049692,"about_ca_system_score_gemma":0.005380708,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002019198,"about_ca_topic_score_gemma":0.002495787,"domain_scores_codex":[0.9566026,0.02680524,0.002677114,0.00295466,0.009568775,0.001391647],"domain_scores_gemma":[0.9249206,0.03769528,0.002975577,0.01392638,0.01786122,0.002621003],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004591491,0.0002674126,0.003092796,0.0009630959,0.0002981201,0.0003428891,0.001374141,0.02137459,0.004443731,0.6401836,0.1239007,0.2032998],"study_design_scores_gemma":[0.0001902977,0.0004853091,0.0009881789,0.0009467131,0.00009297652,0.0002760416,0.0008915787,0.08100436,0.01204223,0.7211635,0.1816511,0.0002677579],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005918343,0.002323258,0.9400315,0.02868653,0.002196754,0.0005393054,0.0005764774,0.005811155,0.01391662],"genre_scores_gemma":[0.07613179,0.001523372,0.9021626,0.007586674,0.001003924,0.001276829,0.001151234,0.004122514,0.005041068],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.06699935,"threshold_uncertainty_score":0.3543307,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007002299377866998,"score_gpt":0.2156237908336125,"score_spread":0.2086214914557455,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}