{"id":"W1554594814","doi":"10.1109/metrics.2004.2","title":"A prototype empirical evaluation of test driven development","year":2004,"lang":"en","type":"article","venue":"IEEE International Software Metrics Symposium","topic":"Software Engineering Research","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; University of Calgary","funders":"","keywords":"Computer science; Debugging; Test-driven development; Software development; Software development process; Software engineering; Test strategy; Software quality; New product development; Manual testing; Test (biology); Software; Empirical research; Software construction; Operating system; Business","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07918895,0.0007779818,0.000740386,0.002100538,0.00192242,0.003238209,0.003831539,0.002554388,0.008509725],"category_scores_gemma":[0.310636,0.0006982367,0.0008333157,0.002425368,0.003678822,0.00504167,0.003268574,0.002095321,0.001629079],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004356789,"about_ca_system_score_gemma":0.004331052,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002280847,"about_ca_topic_score_gemma":0.001836355,"domain_scores_codex":[0.9074475,0.07311607,0.003806659,0.003219321,0.01074603,0.001664501],"domain_scores_gemma":[0.46357,0.4313927,0.01358535,0.03324799,0.0525969,0.005607124],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.02319307,0.08125424,0.1395811,0.008365877,0.0006916628,0.001765322,0.05038581,0.01863723,0.01466966,0.06568946,0.02709552,0.5686711],"study_design_scores_gemma":[0.03272934,0.2406631,0.2798174,0.005055981,0.001227224,0.001918747,0.0471492,0.09361948,0.02650941,0.04830713,0.22206,0.0009430012],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8817163,0.0006869697,0.04517604,0.002316989,0.0003535317,0.01924765,0.001207581,0.0005235975,0.04877137],"genre_scores_gemma":[0.9193952,0.000354884,0.06010311,0.000848046,0.00009142565,0.01541695,0.0008410799,0.00009465862,0.002854628],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07918895,"threshold_uncertainty_score":0.4187962,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06676157119916065,"score_gpt":0.3508707110518867,"score_spread":0.284109139852726,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}