{"id":"W2091014931","doi":"10.1145/1177055.1177057","title":"Confidence estimation for NLP applications","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Artificial intelligence; Natural language processing; Estimation; Confidence interval; Natural language; Machine learning; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0166974,0.001399912,0.001844465,0.005294047,0.0012853,0.005421385,0.00334534,0.003020306,0.007403893],"category_scores_gemma":[0.2205581,0.00100757,0.001283321,0.004816669,0.002120306,0.007606319,0.004191077,0.004928445,0.003392886],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001413335,"about_ca_system_score_gemma":0.001741208,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002366609,"about_ca_topic_score_gemma":0.001089834,"domain_scores_codex":[0.9736894,0.01161834,0.002234871,0.003222605,0.008601945,0.0006328522],"domain_scores_gemma":[0.8360274,0.1337048,0.00566833,0.01080912,0.01284312,0.0009473805],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004909163,0.0001440298,0.006720331,0.001248772,0.0002420924,0.0004877081,0.0005619212,0.1210634,0.005415148,0.2522609,0.02006124,0.5913035],"study_design_scores_gemma":[0.00004395898,0.00008241483,0.001354968,0.0002405878,0.0000555381,0.0004213444,0.0001046754,0.650025,0.006666863,0.3240908,0.01683585,0.00007799829],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002372802,0.001089597,0.9924812,0.0004443128,0.00007542683,0.00005790106,0.0002662093,0.001150586,0.002061976],"genre_scores_gemma":[0.3284656,0.002281546,0.6604008,0.0007382846,0.001019223,0.0006410613,0.002448832,0.001191807,0.002812807],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0166974,"threshold_uncertainty_score":0.08830529,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00987436474930131,"score_gpt":0.2842979056426198,"score_spread":0.2744235408933184,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}