{"id":"W6929151390","doi":"10.48448/v2q6-0106","title":"UniSumEval: Towards Unified, Fine-grained, Multi-dimensional Summarization Evaluation for LLMs","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Automatic summarization; Benchmark (surveying); Annotation; Context (archaeology); Focus (optics); Quality (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009707894,0.003098263,0.00138602,0.006325741,0.001257737,0.004238696,0.002917328,0.002040537,0.00928735],"category_scores_gemma":[0.04160677,0.0005885403,0.001350198,0.003109601,0.0008224216,0.004857886,0.00395149,0.001819565,0.005508493],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001763562,"about_ca_system_score_gemma":0.001782099,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004661402,"about_ca_topic_score_gemma":0.009581732,"domain_scores_codex":[0.989552,0.004423866,0.00135578,0.001570837,0.002684791,0.0004129094],"domain_scores_gemma":[0.978045,0.009677716,0.00130875,0.004003034,0.006287727,0.0006777643],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001778474,0.0008108654,0.004800303,0.005520814,0.0007633141,0.0003882715,0.001499661,0.04106915,0.03077631,0.007927728,0.2427017,0.6619635],"study_design_scores_gemma":[0.001092603,0.002864185,0.0110777,0.001226237,0.0004731462,0.0007910437,0.00236711,0.6718414,0.1058901,0.02605433,0.1758855,0.0004366807],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1338761,0.01774868,0.5001542,0.002175951,0.002182211,0.00199584,0.05119332,0.2619257,0.02874804],"genre_scores_gemma":[0.2864208,0.001871773,0.5464402,0.0009554085,0.000396929,0.001851666,0.1408381,0.01017129,0.01105393],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009707894,"threshold_uncertainty_score":0.05134088,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08525053730682784,"score_gpt":0.3858531518827633,"score_spread":0.3006026145759355,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}