{"id":"W1980089213","doi":"10.1109/compsac.2012.23","title":"Evaluating Reliability-Testing Usage Models","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Reliability (semiconductor); Computer science; Reliability engineering; Goodness of fit; Markov chain; Markov model; Non-regression testing; Software quality; Markov process; Data mining; Machine learning; Statistics; Software; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02364416,0.001918294,0.001338469,0.005109079,0.0005691965,0.002152843,0.002337583,0.002084521,0.0009149387],"category_scores_gemma":[0.1368029,0.0007429403,0.001322939,0.002977265,0.001202846,0.003644833,0.001508579,0.001219315,0.0003758391],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002066844,"about_ca_system_score_gemma":0.001483739,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006319282,"about_ca_topic_score_gemma":0.004227322,"domain_scores_codex":[0.9732757,0.01390124,0.001868525,0.001909316,0.008203348,0.0008418697],"domain_scores_gemma":[0.8135199,0.144353,0.01060563,0.01699064,0.01329314,0.001237654],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004869655,0.0002962046,0.07055682,0.0001981753,0.0003553229,0.0001572113,0.0003254257,0.8597446,0.002546825,0.005654907,0.0006479621,0.05902964],"study_design_scores_gemma":[0.00001257208,0.0002052022,0.00645846,0.00003024364,0.00004014782,0.0001089216,0.00005631836,0.9879565,0.001415175,0.003431372,0.0002588141,0.00002633664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6358522,0.0005253099,0.3564141,0.0002540787,0.00003484471,0.0003505559,0.0008954172,0.001895689,0.003777882],"genre_scores_gemma":[0.9402466,0.0001238256,0.05794114,0.00003762365,0.00001708546,0.000224214,0.0009781731,0.0001404195,0.0002908325],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02364416,"threshold_uncertainty_score":0.1250438,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2197938598611162,"score_gpt":0.4020445771824858,"score_spread":0.1822507173213697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}