{"id":"W2068919549","doi":"10.1016/j.jocs.2010.12.002","title":"Examining random and designed tests to detect code mistakes in scientific software","year":2010,"lang":"en","type":"article","venue":"Journal of Computational Science","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa; Royal Military College of Canada","funders":"","keywords":"Computer science; Mistake; Code (set theory); Test (biology); Software; Code coverage; Software bug; Programming language; Reliability engineering; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008245439,0.0007926486,0.0004662006,0.002064261,0.0005319313,0.001091429,0.001807945,0.001790525,0.001058001],"category_scores_gemma":[0.208067,0.000550252,0.0005419822,0.00096033,0.00100327,0.001546741,0.0008922172,0.001051235,0.0003681114],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009844356,"about_ca_system_score_gemma":0.001709475,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001763137,"about_ca_topic_score_gemma":0.003472924,"domain_scores_codex":[0.9849923,0.007263479,0.001400505,0.00236558,0.003496588,0.0004816998],"domain_scores_gemma":[0.566781,0.3537556,0.03125981,0.01751795,0.02827123,0.00241424],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005283432,0.002699036,0.6421588,0.00110209,0.0007906306,0.002097375,0.005067284,0.02777955,0.04915625,0.003890308,0.004962594,0.2550126],"study_design_scores_gemma":[0.0009639808,0.01437722,0.3711093,0.0005137505,0.001007245,0.003893373,0.003402483,0.4372827,0.1493418,0.01135208,0.006409273,0.0003468005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9632151,0.0001575678,0.03382785,0.000106538,0.00006992443,0.0001320555,0.0001485272,0.001233509,0.001108941],"genre_scores_gemma":[0.9717492,0.00002838237,0.02707299,0.0001287073,0.00001291388,0.0000830152,0.000177424,0.0001565315,0.0005907516],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9917545,"threshold_uncertainty_score":0.04360658,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02964945215008588,"score_gpt":0.2961040621995382,"score_spread":0.2664546100494523,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}