{"id":"W7082289850","doi":"10.48448/vmpg-kd28","title":"Improving Automatic Evaluation of Large Language Models (LLMs) in Biomedical Relation Extraction via LLMs-as-the-Judge","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Relation (database); Relationship extraction; Disk formatting; Benchmark (surveying); Task (project management); Adaptation (eye)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01894305,0.002892759,0.00167965,0.003192774,0.001244585,0.004381968,0.003216298,0.003106113,0.007882962],"category_scores_gemma":[0.08231788,0.0007639857,0.002189853,0.001598837,0.001123149,0.005609959,0.004226878,0.003515113,0.007664166],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001454998,"about_ca_system_score_gemma":0.003081758,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003315774,"about_ca_topic_score_gemma":0.008088117,"domain_scores_codex":[0.970606,0.01854155,0.002292982,0.004479759,0.003482687,0.0005969898],"domain_scores_gemma":[0.9317896,0.05202277,0.002081027,0.006604097,0.006327077,0.001175348],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002293571,0.0013868,0.01508154,0.004692263,0.0008672259,0.001270933,0.003230901,0.03854164,0.06264403,0.008697766,0.1187784,0.7425149],"study_design_scores_gemma":[0.0007249457,0.001613706,0.009334208,0.0004708601,0.0004056672,0.00156283,0.002330948,0.7964694,0.0998528,0.01889067,0.0679258,0.0004181074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2206353,0.008354395,0.5983583,0.003844741,0.002201521,0.002505141,0.01919896,0.125538,0.01936373],"genre_scores_gemma":[0.4430658,0.000795099,0.4865363,0.002357307,0.0004170663,0.001614327,0.05369583,0.003870791,0.007647395],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01894305,"threshold_uncertainty_score":0.1001816,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01870475802595813,"score_gpt":0.3100892437054796,"score_spread":0.2913844856795215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}