{"id":"W4313644201","doi":"10.31219/osf.io/2t53f","title":"An Empirical Study of Supervised and Active Learning Methodologies for Classifying Research Papers","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Machine learning; Artificial intelligence; Computer science; Random forest; Naive Bayes classifier; Supervised learning; Semi-supervised learning; Class (philosophy); Active learning (machine learning); Oversampling; Set (abstract data type); Empirical research; Instance-based learning; Ensemble learning; Support vector machine; Artificial neural network; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07805445,0.0006275005,0.000618671,0.00585767,0.001192043,0.003442904,0.002105537,0.001915007,0.001472014],"category_scores_gemma":[0.3819462,0.0003695442,0.0007226329,0.005315422,0.002639384,0.006759707,0.001432286,0.002034134,0.0007790643],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001161085,"about_ca_system_score_gemma":0.0009422255,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008609294,"about_ca_topic_score_gemma":0.001321921,"domain_scores_codex":[0.9066756,0.06469371,0.005420723,0.003857402,0.01861547,0.000737137],"domain_scores_gemma":[0.285477,0.6189834,0.033765,0.02700835,0.03251348,0.002252738],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002265482,0.004658633,0.4929821,0.001638664,0.001009749,0.0004032178,0.006345485,0.02539206,0.00322495,0.03511522,0.01177542,0.415189],"study_design_scores_gemma":[0.0007882729,0.004916413,0.3413118,0.001369345,0.0004652334,0.00297435,0.01104579,0.5070047,0.01342638,0.0777403,0.03860967,0.0003477253],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8667512,0.00402402,0.1103227,0.002276652,0.0002999821,0.0009558757,0.001363281,0.0003732736,0.01363305],"genre_scores_gemma":[0.9575597,0.000560612,0.03746578,0.0002465817,0.0002394457,0.0004449088,0.001545655,0.00007235017,0.001864863],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07805445,"threshold_uncertainty_score":0.4127963,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.536340079485524,"score_gpt":0.5289477743316955,"score_spread":0.007392305153828493,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}