{"id":"W3210704534","doi":"10.20944/preprints202110.0382.v1","title":"Employing Statistical Machine Reading for Inferring Key Concepts of a Research Field From a Body of Abstracts and Blog Posts","year":2021,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reading (process); Field (mathematics); Data science; Function (biology); Computer science; Comprehension; Conversation; Key (lock); Management science; Political science; Psychology; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.00184388,0.0002293606,0.0006080092,0.0002125881,0.00008692015,0.00007203443,0.001063385,0.0002906676,0.00007240937],"category_scores_gemma":[0.002595463,0.0002541244,0.0001021138,0.0001314569,0.0001178048,0.0001585695,0.005099945,0.00105462,0.000004572344],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006432929,"about_ca_system_score_gemma":0.000346351,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003413336,"about_ca_topic_score_gemma":0.00006278397,"domain_scores_codex":[0.9968026,0.0002301899,0.0008622525,0.001129107,0.000567288,0.0004085537],"domain_scores_gemma":[0.995258,0.002341819,0.0003478805,0.001422289,0.0004782142,0.0001518421],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001915409,0.0004000783,0.7864633,0.003313011,0.0005381247,0.0001664605,0.01957828,0.003273688,0.1351245,0.03574054,0.00001853685,0.01519195],"study_design_scores_gemma":[0.001421159,0.0001615164,0.4468025,0.003467404,0.00008719001,0.00002086136,0.0004170022,0.1544887,0.3419068,0.05033822,0.00008388847,0.0008046606],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7737905,0.0001759495,0.2243937,0.0001853903,0.0002367524,0.0004399697,0.0000451175,0.00004924209,0.0006834798],"genre_scores_gemma":[0.9190676,0.00006165749,0.08063747,0.00003326943,0.00007182687,0.00005863251,0.0000214934,0.00001955891,0.00002851009],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3396607,"threshold_uncertainty_score":0.9999911,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2271925651014114,"score_gpt":0.4556321232008874,"score_spread":0.228439558099476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}