{"id":"W4285310604","doi":"10.18653/v1/2022.nlp4convai-1.5","title":"Data Augmentation for Intent Classification with Off-the-shelf Large Language Models","year":2022,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Classifier (UML); Training set; Task (project management); Labeled data; Language model; Machine learning; Artificial intelligence; Scarcity; Data quality; Data mining; Metric (unit)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002770267,0.001952294,0.001186281,0.001253325,0.0005886601,0.001145752,0.002386648,0.001555637,0.004159496],"category_scores_gemma":[0.01508198,0.000699079,0.001499489,0.00109009,0.0007981821,0.002879291,0.002618524,0.00395413,0.00504744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006769065,"about_ca_system_score_gemma":0.001261623,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001941896,"about_ca_topic_score_gemma":0.004032193,"domain_scores_codex":[0.9982696,0.0007417953,0.0001185448,0.0005066764,0.0002547612,0.0001086962],"domain_scores_gemma":[0.9931576,0.004308321,0.0002721478,0.001240024,0.0008153748,0.0002065719],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001026471,0.0009298599,0.004844852,0.0007414691,0.0001572158,0.0003314647,0.0008365695,0.05887706,0.04955791,0.00353752,0.02334435,0.8558152],"study_design_scores_gemma":[0.0001298889,0.0004644985,0.001856428,0.00008563,0.0000701619,0.0001881462,0.0002857817,0.9473553,0.0280797,0.01137435,0.01003596,0.00007414851],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06596027,0.001338152,0.8993245,0.0007263013,0.0004832816,0.0005456324,0.002941994,0.02633978,0.002340097],"genre_scores_gemma":[0.4457392,0.0003958529,0.5334443,0.0007976695,0.0002660217,0.001599651,0.0126693,0.001108762,0.003979307],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004159496,"threshold_uncertainty_score":0.0146507,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1125840059713915,"score_gpt":0.3134050345317351,"score_spread":0.2008210285603436,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}