{"id":"W2985974955","doi":"10.22148/16.050","title":"A Shared Task for the Digital Humanities Chapter 3: Description of Submitted Guidelines and Final Evaluation Results","year":2019,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Digital Humanities and Scholarship","field":"Arts and Humanities","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Task (project management); Annotation; Reflection (computer programming); Narrative; Computer science; Digital humanities; Mathematics education; Information retrieval; Data science; Artificial intelligence; Psychology; World Wide Web; Linguistics; Programming language; Engineering; Philosophy","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06616345,0.001742839,0.001418978,0.006245153,0.007908522,0.01014654,0.003210788,0.003592662,0.037553],"category_scores_gemma":[0.1586,0.001396477,0.001427465,0.005442129,0.00301132,0.006405545,0.01000162,0.004202381,0.03526781],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007636675,"about_ca_system_score_gemma":0.01890993,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0109248,"about_ca_topic_score_gemma":0.01718282,"domain_scores_codex":[0.917559,0.04724182,0.007825905,0.005443599,0.01807242,0.003857173],"domain_scores_gemma":[0.8749779,0.0297655,0.00284312,0.01408079,0.07264806,0.005684523],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001806225,0.001692886,0.00424937,0.009682487,0.000114783,0.0007732506,0.09002178,0.001964442,0.01932767,0.01759154,0.4398637,0.4129119],"study_design_scores_gemma":[0.0003891141,0.0008249132,0.005911415,0.005869525,0.000129742,0.0002466872,0.03556634,0.002354964,0.01781237,0.01148435,0.9190764,0.0003341483],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1331837,0.005519364,0.3706459,0.02138407,0.009855323,0.142858,0.04636386,0.0282648,0.241925],"genre_scores_gemma":[0.1504785,0.002422553,0.4202439,0.005575412,0.0008145286,0.2355574,0.04925694,0.01717796,0.1184729],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9338366,"threshold_uncertainty_score":0.34991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3084406244997505,"score_gpt":0.3152306524364105,"score_spread":0.006790027936659948,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}