{"id":"W4411531987","doi":"10.1007/978-3-031-96235-6_22","title":"Enhancing Answer Reliability Through Inter-Model Consensus of Large Language Models","year":2025,"lang":"en","type":"book-chapter","venue":"IFIP advances in information and communication technology","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Reliability (semiconductor); Computer science; Natural language processing; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003848844,0.0001813913,0.0003390191,0.0004641673,0.000074624,0.00002698631,0.001108185,0.000380839,0.000005547883],"category_scores_gemma":[0.0001221691,0.0001918786,0.00004354153,0.0001459302,0.0001918629,0.001325296,0.0009213822,0.0005293937,0.00000386227],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008289598,"about_ca_system_score_gemma":0.0001019237,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001579782,"about_ca_topic_score_gemma":0.0001220621,"domain_scores_codex":[0.9985815,0.00002686848,0.0008874928,0.0002083183,0.0001380566,0.0001577892],"domain_scores_gemma":[0.9974197,0.0001521689,0.0005138671,0.001689099,0.0002084472,0.00001667685],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000004336258,0.00000891655,0.000005906474,0.0001085542,0.000004466249,1.661226e-7,0.001731943,0.001955865,0.000004353748,0.9191965,0.00002280767,0.07695614],"study_design_scores_gemma":[0.0002943124,0.00001774428,4.393881e-7,0.0003817066,0.000004937748,0.000003918279,0.0005103034,0.3318118,0.0003728369,0.6320398,0.03439105,0.0001710405],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0001603403,0.006636288,0.8112412,0.000936348,0.0000787688,0.0002577167,0.00001802717,0.0001803314,0.180491],"genre_scores_gemma":[0.4317272,0.02327043,0.536362,0.001017305,0.00000851409,0.00009706906,0.00008437596,0.00001724799,0.00741584],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4315669,"threshold_uncertainty_score":0.7824582,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01114010563274158,"score_gpt":0.2817262020971429,"score_spread":0.2705860964644013,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}