{"id":"W2912962050","doi":"10.1109/beliv.2018.8634420","title":"How to Evaluate an Evaluation Study? Comparing and Contrasting Practices in Vis with Those of Other Disciplines : Position Paper","year":2018,"lang":"en","type":"article","venue":"","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Checklist; Position (finance); Position paper; Resource (disambiguation); Empirical research; Psychology; Qualitative research; Engineering ethics; Management science; Sociology; Applied psychology; Social science; Computer science; Epistemology; Engineering; Cognitive psychology; Business; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4447128,0.0009173926,0.001724831,0.007662059,0.006943785,0.02923617,0.003032054,0.003732094,0.006722671],"category_scores_gemma":[0.6574951,0.0008666695,0.001223944,0.007125297,0.0157238,0.02955164,0.007967123,0.005466008,0.002192957],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01103122,"about_ca_system_score_gemma":0.02802377,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003112927,"about_ca_topic_score_gemma":0.00442364,"domain_scores_codex":[0.4589839,0.4788696,0.02069671,0.007966938,0.03111861,0.002364298],"domain_scores_gemma":[0.2980613,0.4500789,0.03456424,0.05821963,0.1486111,0.01046485],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003629441,0.0003696401,0.01729784,0.005205063,0.0004482022,0.0001821365,0.07447491,0.0009136942,0.002275559,0.1917101,0.06342111,0.6433387],"study_design_scores_gemma":[0.0003539168,0.001017866,0.01211286,0.02948645,0.0006301017,0.0004241574,0.1574027,0.005047358,0.009032369,0.2753136,0.5087286,0.000450053],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08315197,0.02368413,0.4049185,0.3181719,0.01575079,0.00581114,0.0006523358,0.00137748,0.1464818],"genre_scores_gemma":[0.5470358,0.009414487,0.3966284,0.02313714,0.002983799,0.008178005,0.000423015,0.001566657,0.0106328],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5552871,"threshold_uncertainty_score":0.6847679,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1263585004118408,"score_gpt":0.4299767174871811,"score_spread":0.3036182170753403,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}