{"id":"W7125949815","doi":"10.1109/ase63991.2025.00192","title":"SPICE: An Automated SWE-Bench Labeling Pipeline for Issue Clarity, Test Coverage, and Effort Estimation","year":2025,"lang":"","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Spice; Pipeline (software); Software; Test (biology); Code (set theory); Test data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004157213,0.002020583,0.0008738875,0.006329111,0.001389264,0.002274425,0.002328299,0.002100813,0.007965272],"category_scores_gemma":[0.02753567,0.000720125,0.001547675,0.003532553,0.00077451,0.003325524,0.003642183,0.002493898,0.007392619],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001401929,"about_ca_system_score_gemma":0.00287264,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007033782,"about_ca_topic_score_gemma":0.02600858,"domain_scores_codex":[0.995747,0.0008535919,0.0004436623,0.001491243,0.00118515,0.0002793376],"domain_scores_gemma":[0.9800221,0.008152458,0.001696102,0.004879345,0.004653837,0.00059619],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006347236,0.0004701443,0.03814307,0.003089024,0.0002308115,0.0006487354,0.001929867,0.01334609,0.0278417,0.01039758,0.6043104,0.2989577],"study_design_scores_gemma":[0.0004007914,0.0004004368,0.05679619,0.0008627982,0.0002001969,0.0007487612,0.001820923,0.1863595,0.05074955,0.02899871,0.6724016,0.0002605757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09514207,0.00250365,0.3567652,0.002975056,0.0007834169,0.001296542,0.318856,0.1941076,0.02757057],"genre_scores_gemma":[0.1154464,0.000479797,0.3508419,0.001092781,0.0001656677,0.001852875,0.5142293,0.008032863,0.007858393],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007965272,"threshold_uncertainty_score":0.02664649,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01586443149369384,"score_gpt":0.3423301053617524,"score_spread":0.3264656738680586,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}