{"id":"W3174012329","doi":"10.48550/arxiv.2012.04056","title":"Semantics Altering Modifications for Evaluating Comprehension in Machine Reading","year":2020,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Computer science; Semantics (computer science); Artificial intelligence; Natural language processing; Process (computing); Sentence; Domain (mathematical analysis); Reading (process); Comprehension; Machine learning; Programming language; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005779427,0.001928576,0.0007945956,0.002009111,0.000465053,0.002390003,0.00180118,0.002652525,0.004057639],"category_scores_gemma":[0.03324886,0.0004639474,0.000806351,0.001275073,0.001069128,0.005201291,0.001882122,0.003038059,0.001531501],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001136061,"about_ca_system_score_gemma":0.0008860687,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002512331,"about_ca_topic_score_gemma":0.004179669,"domain_scores_codex":[0.996855,0.001577253,0.0002738783,0.0006597098,0.0005059408,0.0001280842],"domain_scores_gemma":[0.9822013,0.01300469,0.001102363,0.002027492,0.001305464,0.0003587772],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001569677,0.0008726994,0.03403879,0.002122798,0.000800289,0.0004107448,0.001915006,0.2069411,0.06504472,0.004651071,0.01008087,0.6715522],"study_design_scores_gemma":[0.0001063127,0.001489,0.02037479,0.0001482916,0.000228106,0.0003160243,0.0007143978,0.8946229,0.06310638,0.01358704,0.0051837,0.0001230303],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6486374,0.004437032,0.3123913,0.001316308,0.0003382668,0.0008281533,0.002853925,0.01428715,0.01491048],"genre_scores_gemma":[0.9117795,0.0004334153,0.08209822,0.0001732748,0.00006185325,0.00026625,0.00306677,0.0003268475,0.001793943],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.005779427,"threshold_uncertainty_score":0.0305649,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3421025677992835,"score_gpt":0.3722790278488095,"score_spread":0.03017646004952595,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}