{"id":"W4392681658","doi":"10.22318/icls2023.645029","title":"Situating Evaluation: Theory Driven Evaluation in Practice","year":2023,"lang":"en","type":"article","venue":"Proceedings.","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Trent University","funders":"","keywords":"Autoethnography; Knowledge management; Computer science; Sociology; Engineering ethics; Management science; Social science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3061263,0.001715174,0.001843517,0.008159262,0.005319653,0.02572139,0.004588854,0.00639909,0.007954432],"category_scores_gemma":[0.3151244,0.001165552,0.001258844,0.005171824,0.02318176,0.01996971,0.01397282,0.005582233,0.002202211],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01388727,"about_ca_system_score_gemma":0.02321883,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001984938,"about_ca_topic_score_gemma":0.001860067,"domain_scores_codex":[0.5720358,0.3877054,0.00769868,0.007469011,0.02204896,0.003042101],"domain_scores_gemma":[0.6004511,0.3096447,0.01062898,0.03620195,0.03755941,0.005513938],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003117102,0.0007982384,0.004309859,0.003413717,0.0001688678,0.000325735,0.06584799,0.005148522,0.002855083,0.5041252,0.01276183,0.3999331],"study_design_scores_gemma":[0.0006029805,0.001368671,0.002714659,0.0100557,0.0002008832,0.0004876967,0.05870332,0.0240303,0.008585288,0.7061302,0.1868074,0.0003129024],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02157348,0.002242929,0.8753506,0.02464216,0.0009655246,0.006586008,0.0001289895,0.001182924,0.06732744],"genre_scores_gemma":[0.334587,0.001331853,0.6490722,0.002305841,0.000281926,0.006530677,0.0001674074,0.0004964722,0.005226597],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3061263,"threshold_uncertainty_score":0.8556698,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3177210124840651,"score_gpt":0.5597948644746129,"score_spread":0.2420738519905478,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}