{"id":"W1577122732","doi":"10.1007/3-540-45486-1_12","title":"The Power of the TSNLP: Lessons from a Diagnostic Evaluation of a Broad-Coverage Parser","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Suite; Parsing; Test suite; Natural language processing; Artificial intelligence; Test (biology); Programming language; Test case; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001506365,0.0003414995,0.0003849404,0.0002244928,0.0002510849,0.0002550889,0.005137091,0.0002449266,0.00003530716],"category_scores_gemma":[0.0008607028,0.0001994,0.0001664898,0.0006245777,0.0009371157,0.0003475916,0.001191706,0.0006741659,0.000004159644],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001839551,"about_ca_system_score_gemma":0.0008406786,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000117663,"about_ca_topic_score_gemma":0.0001327265,"domain_scores_codex":[0.996129,0.0001418752,0.0005488896,0.0008316554,0.001988457,0.0003601424],"domain_scores_gemma":[0.9947011,0.002206058,0.0005632909,0.001965224,0.0005137404,0.00005060454],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000006052855,0.00002541767,0.00004213383,0.0000119177,0.00001522017,0.000005651225,0.001249653,0.004036751,0.0002886761,0.01224609,0.00002802344,0.9820444],"study_design_scores_gemma":[0.000200952,0.00006887698,0.0006760067,0.0008698014,0.00003307723,0.00001207326,2.220411e-7,0.07779202,0.01579269,0.9038967,0.0003491819,0.0003084217],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001537096,0.007039648,0.9870243,0.001447928,0.0009226755,0.000674087,0.00002108442,0.00008620623,0.001246949],"genre_scores_gemma":[0.8966318,0.0001258745,0.1025902,0.0004651553,0.00009030267,0.00002243865,0.000001977589,0.00001972144,0.00005250234],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.981736,"threshold_uncertainty_score":0.9546078,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01886843362727648,"score_gpt":0.2853218196749292,"score_spread":0.2664533860476527,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}