{"id":"W4409587621","doi":"10.36227/techrxiv.174494992.25671081/v1","title":"Comparative analysis of LLMs, GPT-4 vs Gemini","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Reservoir Engineering and Simulation Methods","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Wilfrid Laurier University","funders":"","keywords":"Internal medicine; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004082413,0.001266884,0.0007683127,0.001611914,0.000536576,0.002554771,0.002264516,0.001192171,0.005609225],"category_scores_gemma":[0.02273949,0.0004014045,0.000882318,0.001568016,0.0007280092,0.004244377,0.001891291,0.001644691,0.002319495],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002131437,"about_ca_system_score_gemma":0.003337423,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01430161,"about_ca_topic_score_gemma":0.01432553,"domain_scores_codex":[0.9981245,0.0006861654,0.0001784687,0.0002884532,0.0005953168,0.000127093],"domain_scores_gemma":[0.9887638,0.007502204,0.0003770074,0.001385689,0.00154906,0.0004221526],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003013727,0.0006652139,0.0163628,0.002615782,0.0007189042,0.0005132455,0.001239775,0.4910508,0.009131147,0.03182551,0.05172514,0.391138],"study_design_scores_gemma":[0.0001432594,0.0005060608,0.002841031,0.0001111757,0.0001736245,0.0001949731,0.0004955362,0.9448843,0.01015013,0.01449039,0.02592489,0.0000847189],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6036642,0.00573079,0.220988,0.005380401,0.001037125,0.0007644825,0.01128365,0.08497953,0.06617188],"genre_scores_gemma":[0.8132017,0.002134943,0.1464955,0.0008940232,0.0001221553,0.0006197568,0.01846498,0.004055785,0.01401121],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01430161,"threshold_uncertainty_score":0.02843678,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04061953574607626,"score_gpt":0.3468940831269322,"score_spread":0.306274547380856,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}