{"id":"W6931827081","doi":"10.5281/zenodo.7790046","title":"UofT-EcoSystem/hotline: Final release for MLSys artifact evaluation","year":2023,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Multidisciplinary Science and Engineering Research","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Artifact (error); Feature (linguistics); Calibration; Process (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007717267,0.002677903,0.001333512,0.003462605,0.001248119,0.005425338,0.003277595,0.001566317,0.1186982],"category_scores_gemma":[0.025058,0.001828125,0.001781509,0.002315185,0.0007966995,0.004337571,0.004122565,0.002149227,0.1423962],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001098482,"about_ca_system_score_gemma":0.002772848,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01127175,"about_ca_topic_score_gemma":0.01448786,"domain_scores_codex":[0.9920171,0.00153109,0.000591982,0.0007143135,0.004565026,0.0005804969],"domain_scores_gemma":[0.9792813,0.002730452,0.0006563031,0.007752313,0.008502343,0.00107726],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007732232,0.0001781517,0.001747732,0.0006825731,0.0001029012,0.0001412204,0.0003654608,0.001700686,0.005027287,0.002701213,0.9183288,0.06825065],"study_design_scores_gemma":[0.0004409113,0.0002809512,0.004466723,0.0004939314,0.0001081925,0.000234141,0.0002844954,0.01375432,0.02406084,0.004380376,0.951268,0.0002271639],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"other","genre_scores_codex":[0.01035178,0.0003623425,0.1068019,0.0007690457,0.0008730068,0.001213021,0.141246,0.6712572,0.06712571],"genre_scores_gemma":[0.03533006,0.0002723034,0.1019526,0.000525158,0.0001440789,0.001462849,0.4639769,0.3235241,0.07281186],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.1186982,"threshold_uncertainty_score":0.3970852,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2546263431628282,"score_gpt":0.4002384161352245,"score_spread":0.1456120729723963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}