{"id":"W4297674479","doi":"10.5281/zenodo.1179009","title":"What Does 'Evaluation' Mean For The Nime Community?","year":2015,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","sts","scholarly_communication","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1504474,0.0004349444,0.0005671851,0.0003410643,0.001619809,0.003961435,0.004994439,0.0003355601,0.001326264],"category_scores_gemma":[0.01933396,0.0002720701,0.0004181252,0.0007101581,0.0004624377,0.00088445,0.002532623,0.001209086,0.0002363224],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003191096,"about_ca_system_score_gemma":0.001835679,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007382005,"about_ca_topic_score_gemma":0.01478181,"domain_scores_codex":[0.9667079,0.02659523,0.001421704,0.000947073,0.003855361,0.0004727189],"domain_scores_gemma":[0.9511418,0.02120558,0.001381864,0.006174651,0.0198335,0.00026262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008218992,0.001220549,0.002837638,0.0001399154,0.0003721456,7.637457e-7,0.1487564,0.004884667,0.0003467963,0.0877204,0.05868631,0.6949522],"study_design_scores_gemma":[0.001687757,0.000001930099,0.003738573,0.00118113,0.0002657337,0.000006207851,0.01933747,0.4835222,0.004850037,0.2902184,0.1944163,0.0007743108],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2617058,0.009703647,0.3279569,0.2618178,0.009768524,0.008616624,0.0004781932,0.000571701,0.1193809],"genre_scores_gemma":[0.9479511,0.001166046,0.01224643,0.0008977141,0.0001327522,0.0009575168,0.0008190769,0.00006493823,0.03576447],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6941779,"threshold_uncertainty_score":0.9999732,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.180340118858431,"score_gpt":0.4219220950226665,"score_spread":0.2415819761642355,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}