{"id":"W4398576675","doi":"10.7910/dvn/kpuoac","title":"Replication Data for: Public Leaderboard Feedback in Sampling Competition: An Experimental Investigation","year":2022,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Experimental Behavioral Economics Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Replication (statistics); Competition (biology); Sampling (signal processing); Computer science; Biology; Mathematics; Statistics; Telecommunications; Ecology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01839082,0.001908764,0.001696488,0.002863409,0.001531438,0.0032948,0.006687414,0.004437296,0.08289147],"category_scores_gemma":[0.08486166,0.00113324,0.001901516,0.005893362,0.001051475,0.002026026,0.00300589,0.002825644,0.07325861],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002056039,"about_ca_system_score_gemma":0.004901034,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02261926,"about_ca_topic_score_gemma":0.04862509,"domain_scores_codex":[0.992679,0.003320349,0.001010061,0.001159055,0.001410528,0.0004209862],"domain_scores_gemma":[0.942283,0.02206092,0.003635243,0.02282362,0.007798587,0.001398704],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002125559,0.00005935745,0.001967789,0.0005689418,0.00006882486,0.00002087864,0.00003198206,0.0003068251,0.00006404136,0.0008189298,0.9918914,0.003988621],"study_design_scores_gemma":[0.00356202,0.0001363063,0.01621749,0.0009244623,0.0002031723,0.0001884604,0.0001804208,0.001960707,0.0009423146,0.00847856,0.9670814,0.0001246225],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0005815675,0.0001432177,0.0006505075,0.0005127567,0.0001023499,0.0001031487,0.9963445,0.000655571,0.0009062955],"genre_scores_gemma":[0.00288128,0.00008637594,0.00231784,0.0002842394,0.0000442514,0.000902583,0.9912454,0.0002242247,0.002013796],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.08289147,"threshold_uncertainty_score":0.2772996,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2296859623064992,"score_gpt":0.3892608771146883,"score_spread":0.159574914808189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}