{"id":"W1577122732","doi":"10.1007/3-540-45486-1_12","title":"The Power of the TSNLP: Lessons from a Diagnostic Evaluation of a Broad-Coverage Parser","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Suite; Parsing; Test suite; Natural language processing; Artificial intelligence; Test (biology); Programming language; Test case; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0193162,0.001320812,0.001041986,0.002954407,0.001575352,0.006705353,0.00435968,0.003599076,0.008141326],"category_scores_gemma":[0.1591336,0.001060069,0.0007482697,0.003188404,0.004790218,0.01728344,0.004436188,0.004480568,0.002743404],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003526861,"about_ca_system_score_gemma":0.004223197,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01345622,"about_ca_topic_score_gemma":0.011145,"domain_scores_codex":[0.9776398,0.01190474,0.001195153,0.001818044,0.006766654,0.0006755653],"domain_scores_gemma":[0.7917615,0.175848,0.002979961,0.01425682,0.01369117,0.001462569],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00292245,0.0008794367,0.0174068,0.001670683,0.0002298497,0.002855765,0.01264467,0.03202121,0.02759209,0.1458132,0.1079476,0.6480163],"study_design_scores_gemma":[0.0007950081,0.0008922504,0.006418685,0.0007683868,0.0005833143,0.003765245,0.006774054,0.4378822,0.1290253,0.2965952,0.1160587,0.0004419425],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3691562,0.003980821,0.4728022,0.03685443,0.001124436,0.0005382847,0.006170626,0.03746985,0.07190318],"genre_scores_gemma":[0.7659086,0.0008273887,0.2116324,0.003160503,0.0003374413,0.0001152438,0.002798934,0.00830852,0.00691086],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0193162,"threshold_uncertainty_score":0.1021551,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01886843362727648,"score_gpt":0.2853218196749292,"score_spread":0.2664533860476527,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}