{"id":"W6996773961","doi":"","title":"Tartarus:A Benchmarking Platform for Realistic And Practical Inverse Molecular Design","year":2023,"lang":"en","type":"article","venue":"University of Groningen research database (University of Groningen / Centre for Information Technology)","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Vector Institute; Cégep de Lévis; University of Toronto","funders":"","keywords":"Benchmark (surveying); Benchmarking; Suite; Set (abstract data type); Field (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005472687,0.002537058,0.001386961,0.002410837,0.0007780274,0.001590257,0.005924912,0.001937279,0.009082457],"category_scores_gemma":[0.01097285,0.000770382,0.001424017,0.003656126,0.0008250723,0.001912869,0.002001867,0.002355504,0.00327594],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001309263,"about_ca_system_score_gemma":0.002476085,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00651945,"about_ca_topic_score_gemma":0.007548063,"domain_scores_codex":[0.9965014,0.00156372,0.0003187357,0.0003511088,0.0009999861,0.0002649922],"domain_scores_gemma":[0.9949512,0.002675817,0.0002931343,0.0008993225,0.0008963053,0.0002842329],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009690475,0.001060112,0.005047387,0.003060175,0.0004954218,0.0003370981,0.0001844997,0.6764891,0.008024083,0.03542826,0.165897,0.1030078],"study_design_scores_gemma":[0.0003820999,0.0004288025,0.001293387,0.0001444427,0.00005593181,0.0001139823,0.00006796402,0.9148805,0.00856737,0.01331131,0.06068157,0.00007264627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1602279,0.01163361,0.5095899,0.002783745,0.001756888,0.001695097,0.05580045,0.1695263,0.0869862],"genre_scores_gemma":[0.3697467,0.004058954,0.4729725,0.0008860607,0.0001429782,0.002314293,0.1288295,0.01432149,0.006727411],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009082457,"threshold_uncertainty_score":0.03038388,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05731104528014077,"score_gpt":0.304857145735837,"score_spread":0.2475461004556962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}