{"id":"W4416037466","doi":"10.18653/v1/2025.emnlp-demos.2","title":"ROBOTO2: An Interactive System and Dataset for LLM-assisted Clinical Trial Risk of Bias Assessment","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Alberta; University of Washington","keywords":"Clinical trial; Risk assessment; Natural (archaeology); MEDLINE; Clinical Practice","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.008558153,0.0004145031,0.001280357,0.0003284035,0.000415433,0.0004313573,0.001391055,0.000365551,0.00003175634],"category_scores_gemma":[0.002777412,0.0003606734,0.0002770659,0.0005954995,0.0002223351,0.0006510915,0.001150242,0.001107341,0.000004104387],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002304938,"about_ca_system_score_gemma":0.001274232,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002576008,"about_ca_topic_score_gemma":0.0004638454,"domain_scores_codex":[0.989352,0.005352921,0.002685894,0.001605419,0.000483358,0.000520432],"domain_scores_gemma":[0.9873802,0.008008589,0.001681113,0.002002408,0.0005663871,0.0003612749],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.03157366,0.004118435,0.1059305,0.004784319,0.001381734,0.00002341099,0.0008976812,0.000885678,0.000009777576,0.07717318,0.01213217,0.7610895],"study_design_scores_gemma":[0.02989638,0.007693908,0.06864487,0.0006987131,0.0002877933,0.000008785222,0.0006347925,0.8868758,0.00001575023,0.0001789107,0.004746514,0.0003177909],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08445913,0.000184427,0.8943238,0.002041742,0.00862498,0.005569308,0.00376911,0.0001462146,0.0008813289],"genre_scores_gemma":[0.8886358,0.00006133679,0.1097175,0.000315141,0.0003206394,0.0001633487,0.0005349587,0.00001972883,0.0002315309],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8859901,"threshold_uncertainty_score":0.9998845,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1740284308516769,"score_gpt":0.5129896321549351,"score_spread":0.3389612013032582,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}