{"id":"W2141291123","doi":"10.1145/1108768.1108795","title":"An empirical framework for comparing effectiveness of testing and property-based formal analysis","year":2005,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Debugging; Computer science; Software engineering; Software bug; Formal methods; Software testing; Empirical research; Formal specification; Software; Reliability engineering; Programming language; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1722868,0.001375582,0.001412534,0.01003786,0.001097006,0.004410813,0.002288627,0.002707656,0.002724656],"category_scores_gemma":[0.5867899,0.0004670708,0.001789841,0.007516942,0.007457251,0.00841538,0.003714948,0.003114407,0.0005881826],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002254438,"about_ca_system_score_gemma":0.00198233,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009997279,"about_ca_topic_score_gemma":0.0005247871,"domain_scores_codex":[0.6888434,0.2460248,0.01406906,0.006463504,0.04254306,0.002056119],"domain_scores_gemma":[0.1667859,0.7691811,0.02063598,0.02476238,0.01721584,0.001418884],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006552163,0.005702456,0.3123976,0.004136627,0.004175294,0.000424333,0.007571692,0.05168827,0.006725843,0.309523,0.006178304,0.2849244],"study_design_scores_gemma":[0.002851192,0.03236476,0.2973026,0.003036693,0.001992297,0.001840038,0.007322393,0.2909147,0.01243884,0.3153972,0.03390117,0.0006381673],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3302804,0.005224709,0.6128856,0.002738899,0.0003971628,0.006173338,0.002505371,0.0005606965,0.03923381],"genre_scores_gemma":[0.8599052,0.0006723963,0.1289898,0.000597795,0.0002312564,0.007729822,0.001007874,0.0001816094,0.0006843313],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8277133,"threshold_uncertainty_score":0.9111504,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04273890736513571,"score_gpt":0.3093231757345271,"score_spread":0.2665842683693914,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}