{"id":"W3041552504","doi":"10.24963/ijcai.2020/723","title":"Incentivizing Evaluation with Peer Prediction and Limited Access to Ground Truth (Extended Abstract)","year":2020,"lang":"en","type":"article","venue":"","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; University of Alberta; University of Waterloo","funders":"","keywords":"Truth telling; Ground truth; Computer science; Incentive; Grading (engineering); Aggregate (composite); Common ground; Set (abstract data type); Incentive compatibility; Nash equilibrium; Artificial intelligence; Mathematical optimization; Mathematics; Microeconomics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004569975,0.0001171628,0.0001113233,0.0000696768,0.0001537366,0.0008360858,0.0002464493,0.00003937832,0.00001717462],"category_scores_gemma":[0.0001400551,0.00009576976,0.00001705598,0.0004324993,0.0000205026,0.0009745292,0.0001635591,0.0001056437,0.00001092299],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000396109,"about_ca_system_score_gemma":0.00005401233,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004160252,"about_ca_topic_score_gemma":0.0000153177,"domain_scores_codex":[0.9985949,0.00004179302,0.0001622621,0.0004513578,0.0005629451,0.0001867589],"domain_scores_gemma":[0.9992176,0.00006742415,0.00005957159,0.0002230558,0.0002570152,0.0001753784],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002480896,0.000202487,0.01341687,0.0001448098,0.0001435568,0.00004656054,0.01870257,0.01192691,0.01989505,0.00825765,0.006906976,0.9201085],"study_design_scores_gemma":[0.001013606,0.0003116366,0.2871464,0.00008153675,0.00003953846,0.00002841035,0.0004146805,0.7031252,0.005745078,0.0004281915,0.001350914,0.00031483],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6897623,0.00001865115,0.3025943,0.004074696,0.0001326231,0.0002897292,0.000001069129,0.0002517662,0.002874842],"genre_scores_gemma":[0.9921513,0.000001679789,0.006722667,0.0009624983,0.00009074133,0.00001363878,0.000005203267,0.000008910624,0.00004339314],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9197937,"threshold_uncertainty_score":0.8062394,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05595112365464087,"score_gpt":0.2824146746786671,"score_spread":0.2264635510240262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}