{"id":"W3080743972","doi":"10.11606/d.55.2020.tde-20082020-093906","title":"Interactive keyterm-based document clustering and visualization via neural language models","year":2020,"lang":"en","type":"dissertation","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico; Dalhousie University; Indiana Corn Marketing Council","keywords":"Cluster analysis; Visualization; Computer science; Natural language processing; Artificial neural network; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.00006102021,0.0002697803,0.000296914,0.0002310174,0.0000679014,0.0002478508,0.0004450651,0.0001097374,0.00001848722],"category_scores_gemma":[0.00001605798,0.0002580259,0.00008536639,0.0002774493,0.00001100447,0.001044716,0.0001400453,0.0001833198,0.00000506446],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008212366,"about_ca_system_score_gemma":0.00002917261,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007340778,"about_ca_topic_score_gemma":0.0002149819,"domain_scores_codex":[0.9986259,0.00005831026,0.0002979478,0.0005789991,0.0002710058,0.0001678206],"domain_scores_gemma":[0.999211,0.00003898072,0.0002454625,0.0003311649,0.00009288886,0.00008057451],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003900863,0.0002694472,0.00006701922,0.001184324,0.0003724826,0.0002926714,0.05178226,0.02826974,0.05533118,0.05008271,0.0006422905,0.8113158],"study_design_scores_gemma":[0.0001304345,0.00006310082,0.00002825051,0.00006136158,0.00002620993,0.000001776513,0.0002091873,0.9743262,0.02202541,0.002814088,0.00004140047,0.0002725743],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003047761,0.0001062745,0.9935104,0.00009958326,0.0001191804,0.0002609439,9.665162e-7,0.0005656173,0.002289305],"genre_scores_gemma":[0.9130115,0.00001903065,0.08525413,0.0004902459,0.00003711827,0.0000571583,0.0003282432,0.00003298699,0.0007695705],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9460565,"threshold_uncertainty_score":0.9999872,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009807791098549257,"score_gpt":0.3188874140990868,"score_spread":0.3090796230005375,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}