{"id":"W2101700102","doi":"10.1109/secse.2009.5069163","title":"Testing for trustworthiness in scientific software","year":2009,"lang":"en","type":"article","venue":"","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"Royal Military College of Canada; Queen's University","funders":"","keywords":"Computer science; Software engineering; Correctness; Software construction; Software reliability testing; Trustworthiness; Software; Software testing; Verification and validation; Code (set theory); Software bug; Software quality; Software development; Programming language; Computer security; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09173316,0.0008655178,0.001801491,0.004203366,0.00193872,0.004205653,0.002439127,0.002427823,0.002130927],"category_scores_gemma":[0.488721,0.000678886,0.001497165,0.002698293,0.007959807,0.008148429,0.004137136,0.002855317,0.0004392842],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00219656,"about_ca_system_score_gemma":0.003640376,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001525208,"about_ca_topic_score_gemma":0.0008652976,"domain_scores_codex":[0.8524598,0.08408593,0.01123098,0.01126317,0.03883347,0.002126605],"domain_scores_gemma":[0.3344633,0.5276973,0.04846233,0.05451621,0.03094408,0.00391665],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.003080574,0.001069031,0.3322372,0.002900478,0.001449785,0.002648904,0.01910425,0.06511579,0.02390731,0.2328863,0.004118741,0.3114818],"study_design_scores_gemma":[0.0003743258,0.002640583,0.07582331,0.001121361,0.0003389242,0.001723397,0.002820803,0.2708698,0.02351193,0.6109425,0.009540115,0.0002928764],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4580768,0.001696177,0.5201591,0.003632298,0.0002854174,0.0007576537,0.0002680486,0.0004695422,0.0146549],"genre_scores_gemma":[0.917709,0.0002088392,0.08079445,0.0001990138,0.00009778672,0.0003883914,0.0001297533,0.00007964088,0.0003931794],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9082668,"threshold_uncertainty_score":0.4851371,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2322874908001109,"score_gpt":0.4142692574484899,"score_spread":0.1819817666483789,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}