{"id":"W7106791789","doi":"10.48448/k7pq-w583","title":"Beyond Pointwise Scores: Decomposed Criteria-Based Evaluation of LLM Responses","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Pointwise; Task (project management); Quality (philosophy); Benchmark (surveying); Ranking (information retrieval); Metric (unit); Correctness; Weighting","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02142495,0.001602481,0.001217859,0.005951611,0.0007023715,0.004431781,0.001655482,0.002647642,0.00808752],"category_scores_gemma":[0.1446557,0.0002810637,0.0009633872,0.002784952,0.001496179,0.005244861,0.00448811,0.001992432,0.004035428],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001392663,"about_ca_system_score_gemma":0.001620763,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002566277,"about_ca_topic_score_gemma":0.004068887,"domain_scores_codex":[0.9699187,0.01697603,0.002003033,0.002703849,0.007772998,0.0006252612],"domain_scores_gemma":[0.9183872,0.05062861,0.005497638,0.01024873,0.01328904,0.00194873],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002056557,0.0005931872,0.05087598,0.001968555,0.0006900201,0.000269679,0.00387408,0.05411263,0.0153307,0.03918817,0.03116191,0.7998785],"study_design_scores_gemma":[0.0002666361,0.001986928,0.0433354,0.001072736,0.0002670329,0.0007023088,0.002264318,0.7071701,0.02661264,0.1642222,0.05172725,0.0003725289],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.2380929,0.004046274,0.6756766,0.003030609,0.000432436,0.00110157,0.006442838,0.01596309,0.0552137],"genre_scores_gemma":[0.7981067,0.0002711499,0.1901109,0.0007160361,0.0001214314,0.0004504742,0.004511457,0.0009339941,0.004777838],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.02142495,"threshold_uncertainty_score":0.1133073,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08537339328573977,"score_gpt":0.4362928065944008,"score_spread":0.3509194133086611,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}