{"id":"W124170475","doi":"","title":"Discrepancy Between Automatic and Manual Evaluation of Summaries","year":2012,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Automatic summarization; Computer science; Metric (unit); ROUGE; Natural language processing; Information retrieval; Artificial intelligence; Coherence (philosophical gambling strategy); Evaluation methods; Significant difference; Data mining; Statistics; Mathematics; Reliability engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02983825,0.00150335,0.001195213,0.005340758,0.0009823336,0.003929817,0.001404218,0.001443913,0.001923481],"category_scores_gemma":[0.1477444,0.0005157541,0.0007126955,0.00238775,0.0007076157,0.002568204,0.001750264,0.000891365,0.002382924],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001151672,"about_ca_system_score_gemma":0.0007733212,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002912801,"about_ca_topic_score_gemma":0.004350428,"domain_scores_codex":[0.9308292,0.04069814,0.006636481,0.006227706,0.01462149,0.0009869706],"domain_scores_gemma":[0.7897689,0.1244864,0.01005747,0.02684912,0.04719587,0.001642292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002142732,0.0005299958,0.03709833,0.002954264,0.0009527408,0.0003980034,0.004618865,0.01000151,0.04831278,0.002637353,0.04752083,0.8428326],"study_design_scores_gemma":[0.0009114985,0.00500102,0.4058832,0.001339745,0.001419819,0.002783386,0.007247631,0.1891324,0.2151077,0.01259895,0.157344,0.001230778],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7617387,0.01094997,0.1568608,0.001421413,0.001424103,0.0009146893,0.00867538,0.03290716,0.02510778],"genre_scores_gemma":[0.9150818,0.0006800098,0.06719039,0.0004568598,0.000306413,0.0004046901,0.00980306,0.001797898,0.004278945],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02983825,"threshold_uncertainty_score":0.1578016,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06725810240843816,"score_gpt":0.3314334499184848,"score_spread":0.2641753475100466,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}