{"id":"W2946233956","doi":"10.1109/tse.2019.2918520","title":"Characterizing Crowds to Better Optimize Worker Recommendation in Crowdsourced Testing","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"National Key Research and Development Program of China; China Scholarship Council; National Natural Science Foundation of China","keywords":"Computer science; Crowds; Crowdsourcing; Task (project management); Context (archaeology); Software bug; Relevance (law); Test (biology); Machine learning; Software; Data science; Computer security; World Wide Web; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003438842,0.002437201,0.002608715,0.001906399,0.001117592,0.001543499,0.00325681,0.002117791,0.002123591],"category_scores_gemma":[0.01630805,0.0009182571,0.001101017,0.001603833,0.0009977957,0.002236728,0.001915851,0.001252882,0.001075241],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001255236,"about_ca_system_score_gemma":0.002220772,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01623901,"about_ca_topic_score_gemma":0.01828503,"domain_scores_codex":[0.9970777,0.0009854118,0.0001563673,0.0009774049,0.0005435299,0.0002595072],"domain_scores_gemma":[0.9904853,0.005975623,0.0007992462,0.0009795956,0.001122556,0.0006377077],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","study_design_scores_codex":[0.001277537,0.001105514,0.02679914,0.001100153,0.0004080241,0.0005775124,0.001012994,0.4978125,0.0130024,0.003870812,0.01463417,0.4383993],"study_design_scores_gemma":[0.0001210102,0.000243844,0.002861445,0.00005864278,0.0000865187,0.0001040762,0.0002612292,0.984938,0.002077355,0.006138884,0.003065458,0.00004349176],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1674966,0.003216055,0.8161743,0.001316992,0.000272637,0.0008239317,0.000889823,0.004227981,0.005581686],"genre_scores_gemma":[0.8155667,0.0006038613,0.1769895,0.0006349104,0.0001790558,0.0005656869,0.001375643,0.0003160263,0.00376852],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01623901,"threshold_uncertainty_score":0.03228897,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01883341979320734,"score_gpt":0.2401437436804875,"score_spread":0.2213103238872801,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}