{"id":"W4416037466","doi":"10.18653/v1/2025.emnlp-demos.2","title":"ROBOTO2: An Interactive System and Dataset for LLM-assisted Clinical Trial Risk of Bias Assessment","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Alberta; University of Washington","keywords":"Clinical trial; Risk assessment; Natural (archaeology); MEDLINE; Clinical Practice","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02496358,0.001872437,0.002300984,0.003294582,0.0007907908,0.002645432,0.002847079,0.002364749,0.08691348],"category_scores_gemma":[0.1040271,0.001018309,0.002851759,0.001669364,0.0006503661,0.001337813,0.003672026,0.00156829,0.0147878],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001116097,"about_ca_system_score_gemma":0.005002833,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002326852,"about_ca_topic_score_gemma":0.009003152,"domain_scores_codex":[0.9887476,0.006764029,0.001880518,0.001148486,0.001253802,0.0002056298],"domain_scores_gemma":[0.9167317,0.06691702,0.00530035,0.006798357,0.002858144,0.001394494],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.007396721,0.0002323754,0.007456761,0.01219804,0.002834951,0.0003617368,0.0003278994,0.003785514,0.001966733,0.004995978,0.858638,0.0998053],"study_design_scores_gemma":[0.02771907,0.001559748,0.02128931,0.004742782,0.004427786,0.001531441,0.0002174997,0.03687189,0.007550217,0.04911076,0.8443816,0.0005978805],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"software","genre_scores_codex":[0.009913154,0.0033047,0.08982688,0.00432548,0.0008407426,0.006504567,0.815975,0.06178899,0.007520414],"genre_scores_gemma":[0.08275715,0.001779898,0.3104129,0.005759018,0.0008386656,0.06662288,0.5057797,0.01472291,0.01132695],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.9750364,"threshold_uncertainty_score":0.2907546,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1740284308516769,"score_gpt":0.5129896321549351,"score_spread":0.3389612013032582,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}