{"id":"W6966966793","doi":"10.48448/n4d3-w687","title":"Towards Reproducible Machine Learning Research in Natural Language Processing Part 2","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Natural language; Support vector machine; Feature (linguistics); Natural (archaeology); Training set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02988407,0.001348323,0.001457328,0.004659181,0.002153435,0.0111991,0.003320408,0.003180306,0.02722079],"category_scores_gemma":[0.05227262,0.0008728739,0.001662662,0.004406746,0.007185551,0.01145602,0.006158214,0.004197089,0.01648218],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00337081,"about_ca_system_score_gemma":0.008775387,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001584722,"about_ca_topic_score_gemma":0.001362326,"domain_scores_codex":[0.98528,0.007539864,0.0008808775,0.002143404,0.003668612,0.0004872222],"domain_scores_gemma":[0.9239085,0.03248747,0.002023346,0.02683612,0.01318614,0.001558428],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001843642,0.000353399,0.002020795,0.001347157,0.00008490689,0.0001915608,0.0006799473,0.004610372,0.01659301,0.4603427,0.07769582,0.435896],"study_design_scores_gemma":[0.00006265397,0.0002189501,0.001906235,0.0005941822,0.00004463769,0.0002786883,0.0003814049,0.02947378,0.02502115,0.6843694,0.2575801,0.00006877053],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.006881671,0.01019694,0.8933046,0.02995423,0.005040842,0.0006211261,0.001067208,0.00345826,0.04947508],"genre_scores_gemma":[0.1114476,0.01097077,0.7743824,0.005991469,0.006755847,0.001700607,0.004749884,0.003862884,0.08013856],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9701159,"threshold_uncertainty_score":0.158044,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05891084852940384,"score_gpt":0.3753735924228179,"score_spread":0.3164627438934141,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}