{"id":"W3212178294","doi":"10.1145/3480001.3480021","title":"DeepDup: Duplicate Question Detection in Community Question Answering","year":2021,"lang":"en","type":"article","venue":"","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Question answering; Computer science; Information retrieval; Task (project management); Transfer of learning; Identification (biology); Domain adaptation; Domain (mathematical analysis); Open domain; Named-entity recognition; Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008362911,0.00008569237,0.0001096277,0.00009807346,0.0001815706,0.0001332648,0.000259865,0.0000771118,0.000003843399],"category_scores_gemma":[0.00008744228,0.00008668775,0.00003052917,0.0004672511,0.00001238724,0.0003908712,0.0001315753,0.0002582102,0.00004192727],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001213638,"about_ca_system_score_gemma":0.00002731575,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003058713,"about_ca_topic_score_gemma":0.003551252,"domain_scores_codex":[0.998785,0.0005084221,0.0002019187,0.0002049395,0.000130072,0.0001697098],"domain_scores_gemma":[0.9992693,0.00007409127,0.00004689657,0.0004905509,0.00007179943,0.00004736158],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0000159772,0.0005374713,0.02408404,0.0001508381,0.00002367792,0.00008732584,0.008745867,0.002024109,0.4921877,0.1253224,0.0002054987,0.3466151],"study_design_scores_gemma":[0.0007995717,0.0001888608,0.1102689,0.0004143885,0.000004906218,0.000307723,0.0008041221,0.33232,0.5437643,0.00687069,0.003615039,0.0006415339],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3851842,0.00008524009,0.6113002,0.0003222437,0.0003749228,0.00005704698,1.438723e-7,0.0002630848,0.002412943],"genre_scores_gemma":[0.9926389,0.0000167,0.006821002,0.00009892401,0.00004146348,0.00001857367,0.000002931358,0.000005554044,0.0003559783],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6074547,"threshold_uncertainty_score":0.4623879,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01902326341631096,"score_gpt":0.2744622188051346,"score_spread":0.2554389553888236,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}