{"id":"W4403345667","doi":"10.48550/arxiv.2410.07025","title":"CheXalign: Preference fine-tuning in chest X-ray interpretation models without human feedback","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Medical Imaging Techniques and Applications","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research; Advanced Research Projects Agency; National Institutes of Health; Deutsche Forschungsgemeinschaft; University of Oxford; Göran Gustafssons Stiftelser; National Defense Science and Engineering Graduate; U.S. Department of Defense","keywords":"Interpretation (philosophy); Preference; X-ray; Computer science; Physics; Optics; Economics; Microeconomics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006108785,0.002941678,0.001747594,0.001374289,0.0007090545,0.002469979,0.004743322,0.003246335,0.008201707],"category_scores_gemma":[0.02211618,0.0008567057,0.001407836,0.001035548,0.000837208,0.003406617,0.002887097,0.003809701,0.004713407],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001746095,"about_ca_system_score_gemma":0.002329496,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007660896,"about_ca_topic_score_gemma":0.01374471,"domain_scores_codex":[0.9963934,0.001326494,0.000252606,0.001190816,0.0005665446,0.0002701164],"domain_scores_gemma":[0.9907653,0.004971093,0.0005912235,0.002123483,0.001051527,0.0004973262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002518004,0.001004878,0.008500662,0.0007002236,0.0003222284,0.0004412878,0.000257961,0.1604475,0.01448376,0.002909959,0.06982478,0.7385888],"study_design_scores_gemma":[0.0002903661,0.0003714296,0.001042377,0.000056785,0.000052869,0.0001961566,0.00005112314,0.9795181,0.007801251,0.005793503,0.004779716,0.00004637747],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1720644,0.006966108,0.6583116,0.002516494,0.0008914895,0.0009837821,0.006227783,0.1422295,0.009808877],"genre_scores_gemma":[0.6142429,0.0007015832,0.3550784,0.001983746,0.0003984509,0.0005333425,0.0133897,0.003371422,0.01030054],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008201707,"threshold_uncertainty_score":0.03230673,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1520923606936568,"score_gpt":0.2637400344583088,"score_spread":0.111647673764652,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}