{"id":"W4398612321","doi":"10.7910/dvn/k0oyqf/qso776","title":"accuracy.py","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Scientific Measurement and Uncertainty Evaluation","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0102234,0.0003569415,0.0005734768,0.0008689397,0.0002100393,0.001105352,0.003781197,0.0003371294,0.2143619],"category_scores_gemma":[0.02418926,0.0002688862,0.0002572471,0.001029264,0.0001123236,0.0007313053,0.000763124,0.0003837134,0.7157601],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001308958,"about_ca_system_score_gemma":0.0007273109,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001342502,"about_ca_topic_score_gemma":0.0002179968,"domain_scores_codex":[0.9917431,0.0005159164,0.00103209,0.001301768,0.004985325,0.00042179],"domain_scores_gemma":[0.9910558,0.001992335,0.0008361781,0.005065241,0.0008984887,0.0001518889],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004044361,0.00004627271,0.00001875172,0.00001829519,0.00002917386,0.000005809958,0.00001283241,0.00004416992,0.00002340383,0.00002666727,0.9956143,0.004119873],"study_design_scores_gemma":[0.0004435173,0.00002331023,0.00005087847,0.00006519125,0.00009844834,0.000001799868,0.0001007695,0.0004028633,0.000026491,0.0005673499,0.997915,0.0003043525],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.00001131,0.000002940343,0.0002376345,0.00003308639,0.008024022,0.0005936921,0.9880556,0.00002532352,0.00301635],"genre_scores_gemma":[0.00001202817,0.00006858702,0.0001104211,0.000685056,0.0002612424,0.00001824189,0.9777285,0.00000747265,0.02110841],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.5013981,"threshold_uncertainty_score":0.9999763,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2689786736356081,"score_gpt":0.4123754465781372,"score_spread":0.1433967729425291,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}