{"id":"W2601473681","doi":"","title":"Enriching the legacy literature with OCR corrections and text-mined semantic metadata","year":2014,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Species Distribution and Climate Change","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University; Open Text (Canada)","funders":"","keywords":"Computer science; Metadata; Information retrieval; Context (archaeology); Encyclopedia; Optical character recognition; Ranking (information retrieval); World Wide Web; Word (group theory); Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.004195324,0.001397352,0.001046711,0.04791131,0.001183284,0.004130152,0.001642985,0.001043975,0.00648019],"category_scores_gemma":[0.02315555,0.0005757216,0.001186564,0.02450574,0.001062592,0.006493005,0.003422358,0.0009857077,0.00962744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007059964,"about_ca_system_score_gemma":0.002800172,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002458227,"about_ca_topic_score_gemma":0.005828223,"domain_scores_codex":[0.9967327,0.0004943097,0.0006694369,0.0007228361,0.001251246,0.0001294797],"domain_scores_gemma":[0.9808657,0.006216231,0.002253796,0.00341814,0.006849034,0.0003969729],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002163036,0.0001585949,0.007641957,0.005795554,0.0001339718,0.001537686,0.002635262,0.001487846,0.04558616,0.004741922,0.02301089,0.9070538],"study_design_scores_gemma":[0.00009525505,0.0003348137,0.05409312,0.002965387,0.0008679935,0.004627998,0.006245204,0.03583238,0.1223117,0.0214054,0.7508196,0.0004012556],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1462614,0.01499165,0.6543891,0.003833265,0.001809669,0.002181215,0.0884978,0.04861125,0.03942471],"genre_scores_gemma":[0.09125009,0.003695208,0.8345249,0.0003335404,0.000567923,0.0004823293,0.05832285,0.003373947,0.007449148],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9958699,"threshold_uncertainty_score":0.02218723,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04346305473385294,"score_gpt":0.2529606173105499,"score_spread":0.2094975625766969,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}