{"id":"W7082289850","doi":"10.48448/vmpg-kd28","title":"Improving Automatic Evaluation of Large Language Models (LLMs) in Biomedical Relation Extraction via LLMs-as-the-Judge","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Relation (database); Relationship extraction; Disk formatting; Benchmark (surveying); Task (project management); Adaptation (eye)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004906077,0.0001913593,0.0002319635,0.0006510393,0.0001525034,0.00007893269,0.001225293,0.0002650566,0.000355633],"category_scores_gemma":[0.0008847208,0.0001552506,0.00004847777,0.001594761,0.0002930623,0.000435165,0.0003762672,0.0003601411,0.00002574948],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000181525,"about_ca_system_score_gemma":0.001140034,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000680145,"about_ca_topic_score_gemma":0.0002000219,"domain_scores_codex":[0.9971131,0.0001658606,0.0004504998,0.0006049452,0.001297986,0.0003675839],"domain_scores_gemma":[0.9983204,0.0001364914,0.0004662926,0.000763242,0.0002511654,0.00006239517],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008559497,0.00087859,0.0001619666,0.0009763075,0.00006198593,0.00003201044,0.005300685,0.01038221,0.04705944,0.07653067,0.01047957,0.848128],"study_design_scores_gemma":[0.0003223242,0.00002325431,0.0001282864,0.0002171041,0.00002262046,0.000008993521,0.0001773224,0.9866442,0.0007782865,0.01023713,0.001287327,0.0001531603],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001639138,0.000449752,0.7749055,0.001762834,0.0007980423,0.0007116164,0.000008443983,0.0002450871,0.2194796],"genre_scores_gemma":[0.932046,0.00001765438,0.02599361,0.0002196302,0.0001382491,0.00005194873,0.00005371091,0.00002123333,0.04145797],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.976262,"threshold_uncertainty_score":0.6330937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01870475802595813,"score_gpt":0.3100892437054796,"score_spread":0.2913844856795215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}