{"id":"W2806319975","doi":"10.63317/2zjrem3tbunr","title":"Transforming Wikipedia into a Large-Scale Fine-Grained Entity Type Corpus","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Scale (ratio); Information retrieval; Artificial intelligence; Type (biology); World Wide Web; Geology; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002809803,0.0001303496,0.0001332711,0.00009813905,0.000170574,0.0001482238,0.0008280752,0.00008540802,0.00009442475],"category_scores_gemma":[0.00007071737,0.0001046671,0.00004707375,0.0005838674,0.00006686037,0.0006438595,0.0001939362,0.0001395981,0.0001014753],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003560965,"about_ca_system_score_gemma":0.00006910607,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001189991,"about_ca_topic_score_gemma":0.0008453531,"domain_scores_codex":[0.9989531,0.00002380091,0.0001687817,0.0003309913,0.0002200307,0.0003033015],"domain_scores_gemma":[0.9992166,0.00002992893,0.00004659586,0.0004181497,0.000205013,0.00008366875],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00007842341,0.0003326274,0.002117362,0.0001587148,0.00004475562,0.00007875086,0.01755144,3.209136e-7,0.1358585,0.3676895,0.02273604,0.4533536],"study_design_scores_gemma":[0.0008403716,0.0006681014,0.0001836356,0.00009760164,0.00002106249,0.00006210964,0.00007406337,0.02264347,0.6902801,0.2068291,0.0774204,0.0008800642],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01775686,0.0004707855,0.9753309,0.0007453398,0.0004018641,0.0001473815,0.000001046453,0.001265927,0.003879863],"genre_scores_gemma":[0.4483109,0.00000493551,0.5500649,0.0004602103,0.0001360938,0.000004952826,0.000002276692,0.00000771133,0.001008127],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.5544215,"threshold_uncertainty_score":0.4268198,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009159296413010809,"score_gpt":0.2687434385094682,"score_spread":0.2595841420964574,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}