{"id":"W2080002385","doi":"10.1145/1370143.1370148","title":"The benefits and challenges of executable acceptance testing","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Executable; Computer science; Acceptance testing; Software engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001831827,0.00004287291,0.00005462329,0.00002403344,0.0001115014,0.00002213314,0.000420667,0.00001510579,0.000001158806],"category_scores_gemma":[0.0006424038,0.00002870961,0.000008007959,0.0001851532,0.00004615675,0.0001442541,0.0002057478,0.00005200502,0.000003156455],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000005735027,"about_ca_system_score_gemma":0.00002199756,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001508144,"about_ca_topic_score_gemma":0.000003233927,"domain_scores_codex":[0.9994669,0.000009390178,0.00007131355,0.0001281478,0.0001665306,0.0001577382],"domain_scores_gemma":[0.9982018,0.001397677,0.00001577396,0.0002802262,0.00007030588,0.00003424926],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00000180016,0.00002259662,0.04855765,0.00004254955,0.00001221037,0.00001072423,0.000825879,0.0008488502,0.0005064456,0.1035465,0.0003800099,0.8452448],"study_design_scores_gemma":[0.0002426559,0.0001399681,0.9409702,0.00005602308,0.000001079227,0.0001554461,0.00006779753,0.04536103,0.008980656,0.0009527393,0.002866289,0.0002060494],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8311017,0.08382382,0.06956474,0.003854804,0.0003132017,0.0002962792,8.149042e-7,0.000739684,0.01030494],"genre_scores_gemma":[0.9588568,0.002756675,0.03820051,0.000008262232,0.00001493228,0.000005254841,1.386638e-8,0.000003755229,0.0001538312],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8924126,"threshold_uncertainty_score":0.1170744,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08238491627779673,"score_gpt":0.2579861573770468,"score_spread":0.17560124109925,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}