{"id":"W2760730809","doi":"10.1109/re.2017.64","title":"Panel: Context-Dependent Evaluation of Tools for NL RE Tasks: Recall vs. Precision, and Beyond","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Recall; Context (archaeology); Computer science; Precision and recall; Cognitive psychology; Artificial intelligence; Natural language processing; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03812086,0.00154945,0.0009536666,0.003025976,0.002667911,0.004962645,0.002510624,0.006515563,0.02488231],"category_scores_gemma":[0.08293048,0.0005832732,0.001741256,0.002319832,0.001257939,0.005871559,0.004149736,0.002590006,0.01210182],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002276034,"about_ca_system_score_gemma":0.002151834,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002591948,"about_ca_topic_score_gemma":0.005408498,"domain_scores_codex":[0.9820299,0.007580176,0.001028215,0.002776988,0.005985369,0.0005993783],"domain_scores_gemma":[0.891456,0.05742722,0.003532897,0.00602889,0.03815194,0.003402988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00351898,0.0007371689,0.01795409,0.001843871,0.0005260248,0.0002180718,0.001000023,0.004694594,0.02358678,0.007486591,0.6725286,0.2659052],"study_design_scores_gemma":[0.002480131,0.006711781,0.1439084,0.007983575,0.001744793,0.001378637,0.00508395,0.0667453,0.1547351,0.06639355,0.5418344,0.001000384],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1784362,0.03081478,0.2906107,0.1631554,0.01508626,0.01109066,0.03108487,0.007659441,0.2720617],"genre_scores_gemma":[0.616226,0.006854848,0.1984069,0.02875465,0.005489893,0.006576628,0.02240393,0.002849595,0.1124376],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03812086,"threshold_uncertainty_score":0.2016047,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1336489168214906,"score_gpt":0.3628396132054447,"score_spread":0.2291906963839541,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}