{"id":"W4327586534","doi":"10.21203/rs.3.rs-2573085/v1","title":"Design and Evaluation of Crowd-sourcing Platforms Based on Users’ Confidence Judgments","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Correctness; Computer science; Test (biology); Metacognition; Popularity; Crowd sourcing; Artificial intelligence; Human–computer interaction; Machine learning; Cognition; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.01598082,0.0002871697,0.0003946662,0.0009160546,0.0003693944,0.0005533218,0.00105533,0.0003038621,0.0000163749],"category_scores_gemma":[0.001316819,0.0002676781,0.0001024521,0.0007121883,0.0001971468,0.0001922657,0.001336945,0.001292843,0.00004198517],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003293252,"about_ca_system_score_gemma":0.001138697,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003022233,"about_ca_topic_score_gemma":0.00001257907,"domain_scores_codex":[0.9930341,0.0009560949,0.0004637459,0.001079979,0.003793252,0.0006727733],"domain_scores_gemma":[0.9947366,0.001886733,0.0002271898,0.001582863,0.001359506,0.0002071655],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000121854,0.0001960367,0.001670305,0.00170809,0.00008960633,0.00006889473,0.004268963,0.9334186,0.002440153,0.002682513,0.001694957,0.05164009],"study_design_scores_gemma":[0.0005553371,0.0002536776,0.00402844,0.003074243,0.00001775888,0.000002513372,0.0002224522,0.9716481,0.008657777,0.01126116,0.0000273637,0.0002511687],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3821561,0.0003389734,0.6127726,0.0006260725,0.0005291975,0.00265324,0.00001322633,0.0003019156,0.0006086674],"genre_scores_gemma":[0.9883994,0.00004883159,0.01110294,0.00003153699,0.0000650206,0.0001737343,0.00001629925,0.00004043835,0.0001218005],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6062433,"threshold_uncertainty_score":0.9999775,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2512312015292228,"score_gpt":0.4188936736434096,"score_spread":0.1676624721141868,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}