{"id":"W4381551981","doi":"10.48550/arxiv.2306.10414","title":"KEST: Kernel Distance Based Efficient Self-Training for Improving Controllable Text Generation","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Text generation; Fluency; Bottleneck; Generator (circuit theory); Natural language generation; Exploit; Kernel (algebra); Language model; Artificial intelligence; Process (computing); Machine learning; Natural language; Mathematics; Power (physics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006099574,0.0003266091,0.0003728491,0.0002335204,0.0003170757,0.0002729283,0.001342693,0.000246575,0.000003886987],"category_scores_gemma":[0.0001096345,0.0003946296,0.0002311475,0.0003984389,0.00003568163,0.0001989413,0.0007538709,0.0003361054,0.00002753315],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004166191,"about_ca_system_score_gemma":0.0005148437,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001121394,"about_ca_topic_score_gemma":0.00008406206,"domain_scores_codex":[0.9974275,0.00008921891,0.0002982358,0.001513419,0.000140757,0.0005308997],"domain_scores_gemma":[0.9978174,0.0002292788,0.0003310819,0.001221346,0.0002530557,0.0001478656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001378121,0.00003713276,0.0001664642,0.0001179742,0.00003170349,0.00003697945,0.000278891,0.9450849,0.00025101,0.05297023,0.00005822851,0.0009526847],"study_design_scores_gemma":[0.0009593293,0.00003271082,0.00006401674,0.00006133824,0.00005581439,6.621769e-7,0.00004620906,0.993671,0.0001932749,0.0042802,0.000227897,0.0004075595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05894711,0.00004638196,0.9377787,0.0001849171,0.00127351,0.0007356977,0.00002607748,0.0007324009,0.0002752205],"genre_scores_gemma":[0.9561136,0.000006468849,0.04242581,0.0001069076,0.0002355808,0.0000118036,0.00002679877,0.000034989,0.001038009],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8971666,"threshold_uncertainty_score":0.9998506,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1131849059198157,"score_gpt":0.2004997923629383,"score_spread":0.0873148864431226,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}