{"id":"W3089306998","doi":"10.48550/arxiv.2009.12452","title":"BET: A Backtranslation Approach for Easy Data Augmentation in Transformer-based Paraphrase Identification Context","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Paraphrase; Generalizability theory; Computer science; Transformer; Artificial intelligence; Natural language processing; Deep learning; Training set; Machine learning; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004437964,0.0002279359,0.000260496,0.0002598607,0.00008822765,0.0001367177,0.001899954,0.0002017275,0.000004602896],"category_scores_gemma":[0.00002481529,0.0002958608,0.0001157003,0.0004736648,0.00003889549,0.0008058244,0.0001885306,0.0002875317,0.000007731084],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001457445,"about_ca_system_score_gemma":0.0002475092,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000186911,"about_ca_topic_score_gemma":0.0001440042,"domain_scores_codex":[0.9977196,0.0001238552,0.000356216,0.001457151,0.0001148311,0.0002283623],"domain_scores_gemma":[0.9981664,0.00008739671,0.0002286464,0.001336291,0.00008507041,0.00009612755],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001678866,0.000226807,0.001466678,0.0005382142,0.00006752405,0.00001826426,0.001101944,0.9111775,0.0006004233,0.06793655,0.00007018218,0.01662809],"study_design_scores_gemma":[0.001298437,0.00002179799,0.0005410513,0.0000446545,0.00005505548,3.007218e-7,0.0001082824,0.9874614,0.0003802698,0.009706886,0.0001044997,0.0002773376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02562151,0.00004452624,0.9721329,0.0003749618,0.0002030983,0.001205626,0.0001310799,0.0001284444,0.0001579045],"genre_scores_gemma":[0.9672303,0.00002147517,0.03122473,0.0001169521,0.0000416224,0.00001063707,0.001294779,0.00001511276,0.00004436622],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9416088,"threshold_uncertainty_score":0.9999493,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2658474981201677,"score_gpt":0.2429260617664889,"score_spread":0.02292143635367885,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}