{"id":"W1794751911","doi":"","title":"Deriving high-level abstractions from legacy software using example-driven clustering","year":2011,"lang":"en","type":"article","venue":"Conference of the Centre for Advanced Studies on Collaborative Research","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Cluster analysis; Program comprehension; Software maintenance; Graph; Programming language; Theoretical computer science; Software; Software system; Software engineering; Data mining; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006793615,0.0002643968,0.000422854,0.0002569319,0.0008207207,0.0001625649,0.001763293,0.00008142249,0.00002379028],"category_scores_gemma":[0.00658934,0.0002091209,0.0000951539,0.001304062,0.0004359793,0.0008056292,0.001342759,0.000505693,0.00001173791],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004455734,"about_ca_system_score_gemma":0.0005932768,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002365268,"about_ca_topic_score_gemma":0.0006056374,"domain_scores_codex":[0.9969361,0.0003081063,0.0004075415,0.0006951776,0.00088708,0.0007660001],"domain_scores_gemma":[0.9895666,0.004819743,0.0002085639,0.001086235,0.004175574,0.000143273],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002313829,0.002595808,0.02966946,0.002052803,0.004940987,0.00012999,0.1986236,0.1974737,0.1964238,0.1632541,0.004969006,0.197553],"study_design_scores_gemma":[0.007903177,0.002164867,0.07375216,0.005882479,0.0001243556,0.000009952411,0.07056648,0.09040094,0.6868434,0.05334762,0.006117391,0.002887171],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3336024,0.0004992204,0.6610897,0.0007643211,0.001522876,0.001945489,0.0002845633,0.0001932289,0.00009817141],"genre_scores_gemma":[0.7642345,0.00009114548,0.2352517,0.000009659239,0.00005516913,0.0001089824,0.000002840977,0.00002697389,0.000219011],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4904196,"threshold_uncertainty_score":0.8527701,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3262199322302999,"score_gpt":0.3984240914621571,"score_spread":0.07220415923185719,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}