{"id":"W2970808735","doi":"10.18653/v1/d19-1069","title":"KnowledgeNet: A Benchmark Dataset for Knowledge Base Population","year":2019,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Benchmark (surveying); Computer science; Natural language processing; Base (topology); Population; Artificial intelligence; Knowledge base; Joint (building); Geography; Cartography; Engineering; Demography; Mathematics; Sociology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003127623,0.002370232,0.001235351,0.009242417,0.001632974,0.003045402,0.004772726,0.003991804,0.01013722],"category_scores_gemma":[0.01831118,0.0006811331,0.001570534,0.008099057,0.0007078666,0.004478029,0.002462706,0.002105229,0.01031368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002772368,"about_ca_system_score_gemma":0.004647187,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03733343,"about_ca_topic_score_gemma":0.04731642,"domain_scores_codex":[0.9974782,0.0005104754,0.0003953855,0.0006418903,0.0007930235,0.000181078],"domain_scores_gemma":[0.9933954,0.002713638,0.0003853505,0.001303994,0.001630999,0.0005705837],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004479191,0.0005572226,0.004978555,0.002443889,0.0003023562,0.0003693867,0.0001832602,0.01147794,0.0009982014,0.005501972,0.8925943,0.08014516],"study_design_scores_gemma":[0.001048026,0.0004001786,0.01416313,0.001317893,0.0004641108,0.001003967,0.0008311131,0.08962002,0.007894871,0.0212822,0.8617993,0.0001752703],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.02616194,0.004707415,0.01209477,0.00221164,0.0005903557,0.0006799619,0.9221658,0.0151582,0.0162299],"genre_scores_gemma":[0.01479785,0.0008629343,0.01704562,0.0002667562,0.00004450156,0.0004729675,0.9639366,0.0003004121,0.002272352],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.03733343,"threshold_uncertainty_score":0.07423222,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03238429445618977,"score_gpt":0.2885714976084831,"score_spread":0.2561872031522933,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}