{"id":"W4386805215","doi":"10.1016/j.jbi.2023.104486","title":"A self-supervised language model selection strategy for biomedical question answering","year":2023,"lang":"en","type":"article","venue":"Journal of Biomedical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"Toronto Metropolitan University; University of Waterloo","funders":"","keywords":"Computer science; Leverage (statistics); Artificial intelligence; Classifier (UML); Machine learning; Language model; Question answering; Retraining; Domain (mathematical analysis); Task (project management); Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004926242,0.001330315,0.001843078,0.002703797,0.001170701,0.00139421,0.002834563,0.002274892,0.003031885],"category_scores_gemma":[0.008215294,0.0007162982,0.002028528,0.001614397,0.0005523678,0.0021833,0.002130087,0.00246274,0.002877838],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006397426,"about_ca_system_score_gemma":0.001886016,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003374189,"about_ca_topic_score_gemma":0.00730303,"domain_scores_codex":[0.9966438,0.001626076,0.00027631,0.0007274629,0.0005220174,0.0002043079],"domain_scores_gemma":[0.9936453,0.004139532,0.0001749979,0.0005575519,0.001269014,0.0002136691],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001010072,0.001045959,0.003816852,0.000395935,0.0006983469,0.0004414515,0.0004806351,0.05287415,0.0364962,0.007269218,0.03511446,0.8603568],"study_design_scores_gemma":[0.00005955956,0.0001017294,0.0005821037,0.00001395121,0.0001011678,0.0001422382,0.00005941093,0.9827771,0.007311553,0.006710719,0.002114009,0.00002639116],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01754406,0.0005233087,0.9751521,0.0004570632,0.0001150879,0.0001913068,0.0006012279,0.004698533,0.0007173549],"genre_scores_gemma":[0.3384112,0.0004123156,0.6433012,0.001010877,0.0006148472,0.0008056576,0.008536793,0.0008686908,0.006038349],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004926242,"threshold_uncertainty_score":0.02605283,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02976819324708699,"score_gpt":0.3077633349504817,"score_spread":0.2779951417033948,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}