{"id":"W7105599236","doi":"10.1109/access.2025.3632686","title":"Comparative Evaluation of Reasoning and Inference in LLM-Based and Diffusion-Based Approaches","year":2025,"lang":"en","type":"article","venue":"IEEE Access","topic":"Language and cultural evolution","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Inference; Generative grammar; Opportunistic reasoning; Model-based reasoning; Qualitative reasoning; Reasoning system; Process (computing)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005583205,0.00004527842,0.0001005255,0.00005661072,0.0001164525,0.00005188327,0.00008109991,0.00004342824,0.0000157491],"category_scores_gemma":[0.000245709,0.00003657626,0.000009504771,0.0002694924,0.0001280134,0.0002025007,0.00001377735,0.00004536157,2.07964e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005211968,"about_ca_system_score_gemma":0.0001801874,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003910362,"about_ca_topic_score_gemma":0.008848896,"domain_scores_codex":[0.9993046,0.0002123293,0.00009474429,0.0001152022,0.0001956786,0.00007747192],"domain_scores_gemma":[0.9995996,0.0001831645,0.0000535829,0.00004921809,0.00009434686,0.00002007084],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001629573,0.000423017,0.8147508,0.0001922091,0.00003225096,0.000001690572,0.06464098,0.009806194,0.005161978,0.01264061,0.0001942382,0.09199307],"study_design_scores_gemma":[0.001806958,0.00003850735,0.6687744,0.0003877751,0.00006865541,4.013881e-8,0.0109822,0.3063897,0.007688835,0.003601287,0.00007699812,0.0001846656],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9942276,0.0003666413,0.00067945,0.0002072082,0.00004142521,0.0002214147,0.000001427268,0.000009319109,0.004245494],"genre_scores_gemma":[0.9997863,0.000008170992,0.00008659167,0.00005531083,0.00001267756,0.00002247189,0.000003621521,8.883523e-7,0.00002392836],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2965835,"threshold_uncertainty_score":0.5911325,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1526966410294123,"score_gpt":0.4302979386398004,"score_spread":0.2776012976103881,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}