{"id":"W4313644201","doi":"10.31219/osf.io/2t53f","title":"An Empirical Study of Supervised and Active Learning Methodologies for Classifying Research Papers","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Machine learning; Artificial intelligence; Computer science; Random forest; Naive Bayes classifier; Supervised learning; Semi-supervised learning; Class (philosophy); Active learning (machine learning); Oversampling; Set (abstract data type); Empirical research; Instance-based learning; Ensemble learning; Support vector machine; Artificial neural network; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002867525,0.0002050495,0.0004336316,0.0006352291,0.000290717,0.0003049027,0.001712692,0.000332155,0.000005923935],"category_scores_gemma":[0.001798927,0.0001685001,0.0000700434,0.0005007811,0.0002171201,0.0002689286,0.003177597,0.0009612576,0.000002300051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006477104,"about_ca_system_score_gemma":0.0001305985,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009598468,"about_ca_topic_score_gemma":0.00004591534,"domain_scores_codex":[0.9968194,0.0009213862,0.0003796495,0.00103029,0.0004898368,0.0003594233],"domain_scores_gemma":[0.9952046,0.003226309,0.0001676556,0.001015306,0.0003246153,0.00006153304],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001768034,0.0007630183,0.04500103,0.0004655656,0.0003630828,0.00000910195,0.03423765,0.001339686,0.01143114,0.02257133,0.001040159,0.8826014],"study_design_scores_gemma":[0.002596906,0.006789252,0.2587274,0.0001671545,0.00007898883,0.000002480422,0.3602987,0.1564542,0.02172798,0.1900688,0.00176613,0.001322058],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6360734,0.00006638997,0.3541363,0.003957956,0.0003141433,0.002129056,0.000005464946,0.002503653,0.0008136121],"genre_scores_gemma":[0.8684962,0.00008934256,0.1303748,0.00001302267,0.00002080707,0.0004780485,0.000007496121,0.0000186797,0.0005015377],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8812793,"threshold_uncertainty_score":0.6871234,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.536340079485524,"score_gpt":0.5289477743316955,"score_spread":0.007392305153828493,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}