{"id":"W3202264790","doi":"10.1007/s10664-021-10016-2","title":"Rotten green tests in Java, Pharo and Python","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Python (programming language); Java; Computer science; Programming language; Unit testing; Test (biology); Software; Biology; Ecology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01407781,0.0007869022,0.0009235585,0.00218819,0.001157593,0.002144177,0.002147014,0.001102272,0.01667463],"category_scores_gemma":[0.1634448,0.0003701055,0.001747201,0.0034462,0.002264532,0.006842692,0.003052938,0.002629067,0.002203228],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001340887,"about_ca_system_score_gemma":0.002077682,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01438768,"about_ca_topic_score_gemma":0.0146463,"domain_scores_codex":[0.984265,0.009902264,0.00103529,0.001741267,0.002036264,0.001020051],"domain_scores_gemma":[0.6003603,0.3712649,0.008244867,0.01172428,0.005660813,0.002744964],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01238982,0.004223412,0.3444031,0.00143487,0.002169643,0.0007824774,0.008722796,0.06147906,0.003709337,0.1996755,0.05490549,0.3061045],"study_design_scores_gemma":[0.001182385,0.005191617,0.5143338,0.0004995439,0.001099139,0.0004605421,0.008830825,0.2678427,0.009870294,0.1602843,0.03007212,0.0003328338],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9049903,0.0004254703,0.05048542,0.0007602078,0.0002397995,0.0001303496,0.002782343,0.001349985,0.03883611],"genre_scores_gemma":[0.9804277,0.0000694858,0.01109056,0.0001700156,0.00005661647,0.0002065219,0.001833714,0.0007945406,0.00535088],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9859222,"threshold_uncertainty_score":0.07445151,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02234213422314712,"score_gpt":0.284636052570599,"score_spread":0.2622939183474519,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}