{"id":"W4400910582","doi":"10.1109/ic3se62002.2024.10593288","title":"Development of Language Model on Biomedical Domain to Pretrain Natural Language Processing","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Computer science; Natural language processing; Domain (mathematical analysis); Natural language; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003013067,0.00009710314,0.0001035105,0.0001666095,0.00003533741,0.00008452834,0.0004470907,0.00004194265,0.00001266495],"category_scores_gemma":[0.00001985019,0.00007273448,0.00002854711,0.0003174361,0.00001436224,0.0001396638,0.0001773401,0.0001244754,0.0000293275],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005843324,"about_ca_system_score_gemma":0.000204906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000005611004,"about_ca_topic_score_gemma":0.00001070113,"domain_scores_codex":[0.9988743,0.00001313914,0.000214852,0.0003235537,0.000370804,0.0002033164],"domain_scores_gemma":[0.999624,0.00002456256,0.00001666805,0.0002334778,0.00001591603,0.00008543483],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003567013,0.00003796447,0.000001224758,0.0001319335,0.00001048373,0.00004600109,0.08616405,0.0005207638,0.04268066,0.01905295,0.0003806922,0.8509697],"study_design_scores_gemma":[0.00007188496,0.00001521425,0.00001050446,0.0001543732,0.000001080924,0.000004740707,0.0006568753,0.9899145,0.008277798,0.0002420831,0.0005363922,0.0001145979],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2080948,0.0002905852,0.788536,0.0006012739,0.0001698307,0.00008758613,9.169455e-7,0.0002091867,0.002009821],"genre_scores_gemma":[0.5440713,9.426235e-8,0.4551473,0.0002160239,0.00003160955,0.000005217108,0.000001175289,0.000004542499,0.0005227202],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9893937,"threshold_uncertainty_score":0.2966026,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01513722655410556,"score_gpt":0.2907106947186042,"score_spread":0.2755734681644986,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}