{"id":"W3041552504","doi":"10.24963/ijcai.2020/723","title":"Incentivizing Evaluation with Peer Prediction and Limited Access to Ground Truth (Extended Abstract)","year":2020,"lang":"en","type":"article","venue":"","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; University of Alberta; University of Waterloo","funders":"","keywords":"Truth telling; Ground truth; Computer science; Incentive; Grading (engineering); Aggregate (composite); Common ground; Set (abstract data type); Incentive compatibility; Nash equilibrium; Artificial intelligence; Mathematical optimization; Mathematics; Microeconomics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01624557,0.001595721,0.001922657,0.0009593282,0.001494493,0.003248286,0.003748868,0.00342558,0.007769427],"category_scores_gemma":[0.08121964,0.0008170825,0.0007406161,0.001327001,0.003666993,0.005240703,0.005861917,0.002941404,0.001282445],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002567256,"about_ca_system_score_gemma":0.002615223,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003365465,"about_ca_topic_score_gemma":0.002120605,"domain_scores_codex":[0.9817773,0.01075261,0.0006901714,0.00330426,0.002456665,0.001019022],"domain_scores_gemma":[0.9111825,0.06015607,0.009618512,0.0124463,0.004499968,0.002096641],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001588035,0.0007267261,0.009062377,0.000641926,0.0003050636,0.0008452777,0.001792317,0.2363631,0.008958164,0.5767987,0.01379978,0.1491186],"study_design_scores_gemma":[0.0002323238,0.0002709024,0.001760314,0.00008125777,0.00005892845,0.0001656222,0.0001755474,0.5741985,0.003319666,0.4139051,0.005754252,0.00007749908],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1018864,0.0005515598,0.8672838,0.005646491,0.0002554859,0.0005619475,0.0004570854,0.001120714,0.02223659],"genre_scores_gemma":[0.9396229,0.0001520334,0.05482248,0.0004232097,0.0001709592,0.0002930325,0.0001196944,0.00009420749,0.004301353],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01624557,"threshold_uncertainty_score":0.0859158,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05595112365464087,"score_gpt":0.2824146746786671,"score_spread":0.2264635510240262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}