{"id":"W4398612321","doi":"10.7910/dvn/k0oyqf/qso776","title":"accuracy.py","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Scientific Measurement and Uncertainty Evaluation","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.005941776,0.002919347,0.002344019,0.008477543,0.001552372,0.007785106,0.004748169,0.003232248,0.3023647],"category_scores_gemma":[0.03776493,0.001283233,0.002023727,0.01291901,0.001226376,0.00361995,0.005492365,0.003089812,0.3872076],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002627853,"about_ca_system_score_gemma":0.003881149,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009574852,"about_ca_topic_score_gemma":0.01561764,"domain_scores_codex":[0.9936655,0.001520296,0.0009947561,0.001613936,0.001528469,0.0006771551],"domain_scores_gemma":[0.9763004,0.007948513,0.001949508,0.008935067,0.003634915,0.001231598],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000483483,0.00001349857,0.0004834982,0.001007313,0.00004703652,0.000007475142,0.00002550735,0.0001196995,0.00005913778,0.001108406,0.9934384,0.003641675],"study_design_scores_gemma":[0.0001578875,0.00001463209,0.001385031,0.0005218033,0.00003521748,0.00002910918,0.00003087138,0.000266303,0.0002957386,0.00297033,0.9942642,0.00002892938],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"software","genre_scores_codex":[0.00007832107,0.0001617168,0.0002046396,0.00021233,0.00007930757,0.00002102889,0.9939482,0.002087258,0.003207216],"genre_scores_gemma":[0.001024952,0.0003034193,0.0009477484,0.0003407235,0.00004767294,0.000183773,0.9939005,0.001151576,0.002099646],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.6976353,"threshold_uncertainty_score":0.9950921,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2689786736356081,"score_gpt":0.4123754465781372,"score_spread":0.1433967729425291,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}