{"id":"W3212178294","doi":"10.1145/3480001.3480021","title":"DeepDup: Duplicate Question Detection in Community Question Answering","year":2021,"lang":"en","type":"article","venue":"","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Question answering; Computer science; Information retrieval; Task (project management); Transfer of learning; Identification (biology); Domain adaptation; Domain (mathematical analysis); Open domain; Named-entity recognition; Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006215066,0.001341771,0.001209293,0.002644222,0.00132583,0.001924758,0.003761298,0.002728975,0.004905079],"category_scores_gemma":[0.02195608,0.0005858433,0.001003456,0.00160725,0.001013221,0.006536115,0.005527296,0.003080863,0.002116334],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00158078,"about_ca_system_score_gemma":0.00226508,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008278391,"about_ca_topic_score_gemma":0.01430132,"domain_scores_codex":[0.9965153,0.001399934,0.0001813409,0.000873414,0.0007693546,0.0002606211],"domain_scores_gemma":[0.9912649,0.00480764,0.0004112422,0.001662522,0.001380271,0.0004734294],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001089689,0.001489551,0.01523143,0.001279702,0.0004708926,0.0006955023,0.002040192,0.03683252,0.0131009,0.01337191,0.08838195,0.8260158],"study_design_scores_gemma":[0.0001752769,0.0003010604,0.002930188,0.0001078534,0.00008343515,0.0004148473,0.0007380918,0.9144085,0.01648722,0.0345386,0.02974996,0.00006497676],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1972555,0.004242149,0.7192506,0.002993833,0.0007541516,0.001766333,0.009762198,0.05393732,0.01003797],"genre_scores_gemma":[0.5176338,0.0005664253,0.4535209,0.001094687,0.0002164817,0.0008760765,0.01608521,0.001017592,0.008988908],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008278391,"threshold_uncertainty_score":0.0328688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01902326341631096,"score_gpt":0.2744622188051346,"score_spread":0.2554389553888236,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}