{"id":"W4404395182","doi":"10.2196/60272","title":"Enhancing Bias Assessment for Complex Term Groups in Language Embedding Models: Quantitative Comparison of Methods","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Word embedding; Measure (data warehouse); Word2vec; Machine learning; Embedding; Association test; Term (time); Data mining","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002061912,0.000142914,0.0004057668,0.0002541189,0.00004229548,0.0001193028,0.0006525986,0.000118438,0.00001740991],"category_scores_gemma":[0.0001841472,0.0001198655,0.00009089307,0.0003620591,0.00005268782,0.0007719271,0.0002929493,0.0003535654,0.000003489213],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001031776,"about_ca_system_score_gemma":0.0002135862,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001666151,"about_ca_topic_score_gemma":0.00002251737,"domain_scores_codex":[0.99778,0.0001052389,0.00113219,0.0001566225,0.0005335123,0.0002923996],"domain_scores_gemma":[0.9980928,0.001242193,0.0001774846,0.0003080582,0.00005876258,0.0001207223],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008032961,0.0001584925,0.0001327634,0.002510405,0.00006518095,0.000009639945,0.222484,0.008690675,0.001185841,0.4424312,0.00040571,0.3219181],"study_design_scores_gemma":[0.0002629466,0.00008053245,0.0000462552,0.0005447987,0.000005699433,0.00000429774,0.008134656,0.9863446,0.0006212695,0.003634047,0.0001928916,0.0001279957],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05292315,0.000187141,0.9450016,0.0001837324,0.0002899966,0.0003595207,0.000004571486,0.0001188364,0.0009314721],"genre_scores_gemma":[0.3921781,0.00000544333,0.607613,0.0001105019,0.00002434286,0.00004827862,0.00000768348,0.00000588876,0.000006767671],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9776539,"threshold_uncertainty_score":0.4887972,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1731788120296335,"score_gpt":0.5031536617832251,"score_spread":0.3299748497535915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}