{"id":"W4401043387","doi":"10.18653/v1/2024.semeval-1.188","title":"BD-NLP at SemEval-2024 Task 2: Investigating Generative and Discriminative Models for Clinical Inference with Knowledge Augmentation","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Discriminative model; SemEval; Artificial intelligence; Computer science; Generative grammar; Inference; Natural language processing; Task (project management); Generative model; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01442277,0.002297137,0.001563166,0.001804197,0.0009654487,0.002671025,0.003882147,0.00389164,0.01000721],"category_scores_gemma":[0.04397611,0.0008952031,0.002235976,0.001352913,0.001221841,0.004137338,0.003582045,0.006040727,0.004766928],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00211949,"about_ca_system_score_gemma":0.003638365,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007716339,"about_ca_topic_score_gemma":0.0152224,"domain_scores_codex":[0.9919543,0.005139522,0.0003470379,0.001739488,0.0006044281,0.0002152322],"domain_scores_gemma":[0.9557919,0.03803346,0.0007556354,0.003521487,0.001214517,0.0006830218],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0042117,0.002354763,0.01329842,0.003619885,0.001061026,0.001298726,0.000987167,0.2125647,0.01027098,0.014493,0.126483,0.6093566],"study_design_scores_gemma":[0.0008543658,0.0004815897,0.002725147,0.0002125065,0.0001977639,0.00059604,0.0002548524,0.9295089,0.008702934,0.03038345,0.02597254,0.0001099908],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2223524,0.01458495,0.6107101,0.02020333,0.001906521,0.002633667,0.04676663,0.05922131,0.02162101],"genre_scores_gemma":[0.5514298,0.001176658,0.3712492,0.004109917,0.0005891741,0.001256075,0.06110337,0.001228027,0.007857719],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01442277,"threshold_uncertainty_score":0.07627583,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1736812373090174,"score_gpt":0.4105354476230899,"score_spread":0.2368542103140725,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}