{"id":"W3195694877","doi":"10.1145/3447548.3470802","title":"Mixed Method Development of Evaluation Metrics","year":2021,"lang":"en","type":"article","venue":"","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Google (Canada)","funders":"","keywords":"Computer science; Metric (unit); Context (archaeology); Data science; Selection (genetic algorithm); Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2897027,0.003393647,0.002303526,0.0110761,0.002321345,0.01098806,0.005851686,0.002504665,0.008020867],"category_scores_gemma":[0.477009,0.002047642,0.003221066,0.006690186,0.003427989,0.007767215,0.009631464,0.004342679,0.002348762],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007081155,"about_ca_system_score_gemma":0.01164109,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003076892,"about_ca_topic_score_gemma":0.003661274,"domain_scores_codex":[0.5827049,0.3476511,0.02192943,0.01187358,0.0340018,0.001839264],"domain_scores_gemma":[0.4274033,0.3925147,0.02136861,0.04725162,0.1089097,0.002552212],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0006647017,0.0006173786,0.00822892,0.005438886,0.0009272318,0.0002647361,0.01698714,0.01650199,0.004902147,0.257176,0.009583162,0.6787077],"study_design_scores_gemma":[0.0008702099,0.002792844,0.007879393,0.008614869,0.0007339154,0.0007135543,0.01329688,0.2946486,0.03106907,0.4569735,0.1817372,0.000669871],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002389975,0.0003709163,0.9894032,0.0004232759,0.0001060267,0.003259839,0.0002772236,0.0004148305,0.00335475],"genre_scores_gemma":[0.02061941,0.000156339,0.9677863,0.0001402195,0.00003262164,0.01010977,0.0002505291,0.0002544331,0.0006502743],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7102973,"threshold_uncertainty_score":0.875923,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1300077186787937,"score_gpt":0.3621521690396146,"score_spread":0.232144450360821,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}