{"id":"W2970844223","doi":"","title":"Team EP at TAC 2018: Automating data extraction in systematic reviews of environmental agents.","year":2018,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Mistake; Computer science; Regularization (linguistics); Task (project management); Word (group theory); Artificial intelligence; Natural language processing; Set (abstract data type); Training set; Data extraction; Sequence (biology); Layer (electronics); Information retrieval; Machine learning; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04668481,0.002275943,0.002417359,0.01480437,0.001838758,0.00459247,0.002719084,0.002202043,0.03215055],"category_scores_gemma":[0.1724051,0.001496233,0.003620861,0.007890671,0.0004387391,0.005229298,0.006389199,0.002252154,0.02326467],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002478607,"about_ca_system_score_gemma":0.01510804,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006031378,"about_ca_topic_score_gemma":0.02649158,"domain_scores_codex":[0.9684398,0.01664015,0.005460854,0.003372029,0.005543426,0.0005437753],"domain_scores_gemma":[0.8704247,0.0759238,0.01436335,0.01542397,0.0202598,0.003604472],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004893586,0.0001240922,0.003784607,0.03464905,0.001404034,0.0002872116,0.001223101,0.001097145,0.00540067,0.002299,0.6311459,0.3180958],"study_design_scores_gemma":[0.0009552963,0.000617995,0.0138368,0.01337529,0.001936347,0.0006251334,0.00113559,0.01751899,0.01515433,0.0161079,0.9184076,0.0003288655],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.01290053,0.0283628,0.32324,0.01239516,0.004633669,0.01737026,0.4634514,0.1168353,0.02081081],"genre_scores_gemma":[0.02510386,0.004168709,0.7350243,0.002378224,0.0006224337,0.01334104,0.2043453,0.004707641,0.01030846],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9533152,"threshold_uncertainty_score":0.2468958,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03917844733812462,"score_gpt":0.3005423099718216,"score_spread":0.261363862633697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}