{"id":"W4401042030","doi":"10.18653/v1/2024.semeval-1.239","title":"CLaC at SemEval-2024 Task 2: Faithful Clinical Trial Inference","year":2024,"lang":"en","type":"article","venue":"","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"SemEval; Inference; Computer science; Task (project management); Natural language processing; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03530348,0.003993926,0.002035248,0.00577786,0.002327348,0.005285045,0.005119608,0.006764268,0.03473715],"category_scores_gemma":[0.1484964,0.001412257,0.004386643,0.002395335,0.001857336,0.004842472,0.006705671,0.006178225,0.01299182],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003126526,"about_ca_system_score_gemma":0.009626063,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008939264,"about_ca_topic_score_gemma":0.02076932,"domain_scores_codex":[0.9742276,0.01551848,0.001730759,0.004483924,0.003350421,0.00068882],"domain_scores_gemma":[0.8979431,0.07488727,0.002488769,0.01387778,0.008950097,0.001852945],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003091238,0.001270996,0.01066303,0.006913346,0.00153727,0.001561146,0.001621574,0.06951039,0.01052433,0.03823089,0.5379645,0.3171114],"study_design_scores_gemma":[0.002205384,0.0009793267,0.005044842,0.001484604,0.0007225128,0.001945957,0.0007164388,0.4848341,0.03180069,0.1373848,0.3325914,0.0002899208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02765807,0.002989389,0.7611367,0.007764452,0.001637059,0.006591864,0.09536342,0.07356953,0.02328954],"genre_scores_gemma":[0.1390858,0.000501115,0.6997485,0.003147466,0.0005454806,0.004348443,0.1381328,0.006018802,0.008471609],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03530348,"threshold_uncertainty_score":0.1867049,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06425536072042741,"score_gpt":0.4095967690482611,"score_spread":0.3453414083278337,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}