{"id":"W7160819649","doi":"10.66372/jger.v3i1.4","title":"Evaluating the Quality of Large Language Model-Generated Explanations in Recommendation Tasks: A Multi-Dimensional Comparative Analysis","year":2025,"lang":"","type":"article","venue":"Journal of Global Engineering Review","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Quality (philosophy); Annotation; Natural language; Recommender system; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007907164,0.0002611467,0.001150464,0.0003304569,0.0001182433,0.00008252749,0.0008001577,0.0000788588,0.00005113778],"category_scores_gemma":[0.001873298,0.0002065673,0.0004644696,0.005852523,0.00003656306,0.0005371322,0.0002013495,0.0004388655,0.000004192292],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005641965,"about_ca_system_score_gemma":0.0005514522,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001477171,"about_ca_topic_score_gemma":0.0003193833,"domain_scores_codex":[0.9953412,0.0008168209,0.002598099,0.0002914766,0.0006125495,0.0003397983],"domain_scores_gemma":[0.9963906,0.0005128229,0.001374081,0.0004369633,0.001196008,0.00008948147],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002055539,0.0002699413,0.00025615,0.0005855627,0.0005686917,0.000005032929,0.000728571,0.978587,0.001245023,0.01005429,0.0001762764,0.007502944],"study_design_scores_gemma":[0.0002991604,0.00007525106,0.001650292,0.003198751,0.0005066699,0.000006013388,0.0001982978,0.9932262,0.0004900126,0.00009101983,0.000100253,0.0001581337],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07106028,0.07356106,0.8516451,0.002659441,0.0004127714,0.0004976654,0.00007217099,0.00001320732,0.00007828501],"genre_scores_gemma":[0.9605001,0.003595296,0.03534105,0.0004810472,0.00002840335,0.00001300825,0.00001281459,0.000004467425,0.00002379014],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8894398,"threshold_uncertainty_score":0.8423569,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1448898721347533,"score_gpt":0.4794434752699744,"score_spread":0.3345536031352211,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}