{"id":"W4411950637","doi":"10.1109/forge66646.2025.00023","title":"MaRV: A Manually Validated Refactoring Dataset","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"EMI","keywords":"Code refactoring; Computer science; Programming language; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004579863,0.001995594,0.0007697263,0.008038322,0.001127234,0.00163195,0.003663277,0.002869667,0.003044923],"category_scores_gemma":[0.02500091,0.0006094534,0.001724188,0.005741796,0.0008176145,0.001723052,0.001840699,0.00230153,0.004568854],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001680091,"about_ca_system_score_gemma":0.002646118,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00903334,"about_ca_topic_score_gemma":0.017461,"domain_scores_codex":[0.9925171,0.001591238,0.001163924,0.001675997,0.002719079,0.0003326807],"domain_scores_gemma":[0.976712,0.008049036,0.002203735,0.006273925,0.006204417,0.0005570071],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00111381,0.001172811,0.04415685,0.007581396,0.0005837409,0.001590697,0.00121945,0.02460168,0.02291272,0.005515052,0.6486185,0.2409333],"study_design_scores_gemma":[0.0007281227,0.0008538166,0.07752848,0.001430877,0.0003878681,0.002363015,0.0008186717,0.0689455,0.04008267,0.007951399,0.7984896,0.0004199512],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.1169294,0.004157558,0.03633072,0.0008276259,0.0004017566,0.0009215232,0.7915484,0.03660627,0.01227676],"genre_scores_gemma":[0.03723201,0.0003602282,0.03250774,0.000186002,0.00003083963,0.0006835969,0.9260148,0.00123587,0.001748955],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.00903334,"threshold_uncertainty_score":0.02422088,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01278864663343112,"score_gpt":0.3091754388331368,"score_spread":0.2963867921997057,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}