{"id":"W4408688975","doi":"10.32388/b69sky","title":"DAFE: LLM-Based Evaluation Through Dynamic Arbitration for Free-Form Question-Answering","year":2025,"lang":"en","type":"preprint","venue":"Qeios","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Arbitration; Question answering; Computer science; Information retrieval; Political science; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02686263,0.002268611,0.001526314,0.00363922,0.0009698757,0.004117431,0.00329166,0.003029236,0.01031192],"category_scores_gemma":[0.1056031,0.0005996838,0.001182113,0.001247914,0.001677972,0.006269299,0.007627098,0.002829913,0.003872497],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001687825,"about_ca_system_score_gemma":0.001898555,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001997539,"about_ca_topic_score_gemma":0.003320475,"domain_scores_codex":[0.9669176,0.02187807,0.001919053,0.003648807,0.004986011,0.0006503663],"domain_scores_gemma":[0.9517157,0.03482217,0.001950202,0.004926826,0.00551625,0.001068888],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001498138,0.000770656,0.008476933,0.001728805,0.0004410998,0.0003285421,0.004548387,0.05463694,0.02786935,0.04699679,0.03447393,0.8182305],"study_design_scores_gemma":[0.0002315486,0.0003808032,0.002583295,0.0001922471,0.00007290838,0.0001962621,0.0006551268,0.8548994,0.01915394,0.09818199,0.0232872,0.0001652538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01605271,0.0004099628,0.9672564,0.0004991677,0.0001398939,0.0006632912,0.0007760727,0.01080151,0.003400967],"genre_scores_gemma":[0.3487172,0.0001625563,0.641537,0.0005822728,0.0001281844,0.001871426,0.002572527,0.00123897,0.003189904],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02686263,"threshold_uncertainty_score":0.1420648,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03800586777782443,"score_gpt":0.3589599952477819,"score_spread":0.3209541274699575,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}