{"id":"W4386622318","doi":"10.5281/zenodo.8335990","title":"Raw Data for benchmarking structure based domain annotation","year":2023,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Benchmarking; Raw data; Annotation; Computer science; Domain (mathematical analysis); Information retrieval; Artificial intelligence; Business; Mathematics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005660024,0.003467418,0.001705645,0.00635008,0.002088872,0.00418754,0.002617171,0.00174159,0.09111042],"category_scores_gemma":[0.02917846,0.001184675,0.001808379,0.006377416,0.0008127444,0.003195025,0.004932281,0.003423377,0.1362826],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001616339,"about_ca_system_score_gemma":0.00349894,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006463168,"about_ca_topic_score_gemma":0.006999286,"domain_scores_codex":[0.9934936,0.001410914,0.000951595,0.001642284,0.002033023,0.0004685791],"domain_scores_gemma":[0.9853918,0.003925416,0.0005973101,0.00553234,0.003808813,0.0007443076],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006414639,0.0002877973,0.002961188,0.001994251,0.000121463,0.0002731369,0.0004913092,0.002813502,0.0077335,0.004011165,0.931163,0.04750818],"study_design_scores_gemma":[0.0003208115,0.000174852,0.01164632,0.0009621159,0.0001564532,0.000617222,0.0008440371,0.01856331,0.03342531,0.01839425,0.9146381,0.0002572534],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.005865628,0.0004023694,0.03673176,0.000321582,0.0004878285,0.0006396419,0.7640252,0.1741278,0.0173983],"genre_scores_gemma":[0.009052461,0.0001865081,0.02882325,0.000199596,0.00003501227,0.001080628,0.9360483,0.02045885,0.004115494],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.09111042,"threshold_uncertainty_score":0.3047947,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06301138541023635,"score_gpt":0.3067664919215459,"score_spread":0.2437551065113096,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}