{"id":"W4392906991","doi":"10.32920/25412863.v1","title":"Incremental Text Clustering Algorithm Using Incremental Learning in COVID-19 Research Papers","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Cluster analysis; Word2vec; Computer science; Data mining; Correlation clustering; Artificial intelligence; Document clustering; tf–idf; Machine learning; Embedding","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002379056,0.00131876,0.001637965,0.01039483,0.001501603,0.003473628,0.002891695,0.001364814,0.004133109],"category_scores_gemma":[0.009590814,0.0005928018,0.001802107,0.00856024,0.000602746,0.00301773,0.002416169,0.001197375,0.004156512],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001434896,"about_ca_system_score_gemma":0.003674483,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004656559,"about_ca_topic_score_gemma":0.006237023,"domain_scores_codex":[0.997147,0.0003423828,0.0003138354,0.001086785,0.0008941423,0.0002158593],"domain_scores_gemma":[0.9951508,0.001065513,0.0005186345,0.0007643045,0.002183993,0.000316743],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003246963,0.0003325448,0.005557755,0.000507012,0.0002435344,0.0003258279,0.000381954,0.03030377,0.01012392,0.006683517,0.02262156,0.9225938],"study_design_scores_gemma":[0.0002374076,0.0003865428,0.004956251,0.0001249421,0.0002868017,0.000871841,0.0005172526,0.8872808,0.02223294,0.03153949,0.05143949,0.0001262974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04765324,0.003027237,0.9260066,0.0007713593,0.0009273626,0.001007364,0.002841334,0.01191295,0.005852585],"genre_scores_gemma":[0.1238524,0.001070921,0.8556521,0.0002679164,0.0005460769,0.0008120298,0.01042666,0.0005972727,0.006774647],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01039483,"threshold_uncertainty_score":0.01382661,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1361018834047257,"score_gpt":0.3970309097637997,"score_spread":0.260929026359074,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}