{"id":"W2132705868","doi":"10.5194/gmd-5-1009-2012","title":"Assessing climate model software quality: a defect density analysis of three models","year":2012,"lang":"en","type":"article","venue":"Geoscientific model development","topic":"Software Engineering Research","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Executable; Context (archaeology); Software; Climate model; Computer science; Software quality; Quality (philosophy); Trustworthiness; Climate change; Software development; Geography; Geology; Programming language; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.009447434,0.0005984734,0.0005518211,0.003189106,0.0005392467,0.001458514,0.001151961,0.0007516043,0.0006863856],"category_scores_gemma":[0.04079361,0.0003706637,0.001623697,0.002268321,0.001089078,0.00153869,0.001187791,0.0008626152,0.00009640273],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002262609,"about_ca_system_score_gemma":0.0007650126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01482206,"about_ca_topic_score_gemma":0.01014058,"domain_scores_codex":[0.997364,0.001147655,0.0002265501,0.0003287641,0.0007764718,0.0001565592],"domain_scores_gemma":[0.8988746,0.07827451,0.007366083,0.007661925,0.006550012,0.001272832],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009321744,0.0008771336,0.6692533,0.0001302076,0.0005898238,0.0002447278,0.001043414,0.2982815,0.002416968,0.003341451,0.001016725,0.02187271],"study_design_scores_gemma":[0.00004489837,0.0004445448,0.1886891,0.00002185248,0.0001473202,0.0001144743,0.0003355581,0.8062556,0.001418156,0.002164125,0.0003111878,0.00005301418],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9959235,0.00003658054,0.003350141,0.00004582354,0.000002285811,0.00002252791,0.0001435701,0.00009390382,0.0003816962],"genre_scores_gemma":[0.9973453,0.00002067859,0.002253297,0.000005258614,0.000001746099,0.00001828037,0.0002871043,0.00001578062,0.00005260157],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9905525,"threshold_uncertainty_score":0.04996341,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1135921490724538,"score_gpt":0.3292275387615579,"score_spread":0.215635389689104,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}