{"id":"W6906660697","doi":"10.17605/osf.io/sdpem","title":"10-month-old infants' social evaluation of true versus negligent accidents: A direct replication attempt","year":2017,"lang":"en","type":"other","venue":"Open Science Framework","topic":"Child and Animal Learning Development","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Replication (statistics); Evaluation methods; Process (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"reproducibility","study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.004281686,0.0002672561,0.0004745578,0.0002678364,0.0005279437,0.000387162,0.00336175,0.0004414755,0.03062831],"category_scores_gemma":[0.003345664,0.0002517182,0.00009437836,0.0004747207,0.0004586542,0.0002069324,0.0007109451,0.0003758213,0.001553221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002534834,"about_ca_system_score_gemma":0.0006303999,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009967628,"about_ca_topic_score_gemma":0.0001547045,"domain_scores_codex":[0.996376,0.0002173036,0.0003500211,0.001309723,0.001341691,0.000405301],"domain_scores_gemma":[0.9958445,0.0001119522,0.001117416,0.002558667,0.0002621292,0.0001052865],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004448288,0.0002170533,0.002160963,0.0000177011,0.0001726868,0.000003007069,0.002748518,0.000002992313,0.00007884738,0.005327317,0.5700486,0.4187775],"study_design_scores_gemma":[0.00107582,0.0001650562,0.1239419,0.0007728898,0.0001206173,0.000001172828,0.0001634868,0.00001454032,0.00007595566,0.0003349394,0.8728995,0.0004341832],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.003549198,0.0003861591,0.00003244531,0.0003677623,0.002247178,0.001261351,0.00002648956,0.00005865607,0.9920707],"genre_scores_gemma":[0.1828491,0.00004496216,0.001778911,0.0001581597,0.001005412,0.0003196162,0.00006104654,0.0001842506,0.8135985],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.4183433,"threshold_uncertainty_score":0.9999935,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08657738528796921,"score_gpt":0.4427051590937182,"score_spread":0.356127773805749,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}