{"id":"W2429937759","doi":"10.1145/2932194.2932203","title":"Accurate fact harvesting from natural language text in wikipedia with Lector","year":2016,"lang":"en","type":"article","venue":"","topic":"Wikis in Education and Collaboration","field":"Social Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Information retrieval; Information extraction; Domain (mathematical analysis); Relationship extraction; Natural language; Natural language processing; Fraction (chemistry); Question answering; Artificial intelligence; World Wide Web; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003133983,0.001140206,0.0009564862,0.009516038,0.0008853189,0.002754879,0.001261489,0.001261392,0.001936521],"category_scores_gemma":[0.02138527,0.00064694,0.00103442,0.00494455,0.0006261083,0.007028233,0.002473375,0.001058385,0.003280689],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005699524,"about_ca_system_score_gemma":0.001163184,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003990527,"about_ca_topic_score_gemma":0.008344213,"domain_scores_codex":[0.9964665,0.0008605603,0.0003891695,0.0008164558,0.001263637,0.0002036925],"domain_scores_gemma":[0.9808739,0.0096286,0.001590084,0.004991885,0.002708942,0.0002066689],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005778808,0.0003605052,0.03472456,0.002811986,0.0006323066,0.001545976,0.001895798,0.02368832,0.03904927,0.01003203,0.04667899,0.8380023],"study_design_scores_gemma":[0.0001481895,0.0004804778,0.06360575,0.001232522,0.001008096,0.002719363,0.002244897,0.415455,0.2060241,0.06046829,0.246319,0.0002944125],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1694135,0.007090525,0.7227872,0.001728317,0.0006855648,0.0007316869,0.03897067,0.04518435,0.01340831],"genre_scores_gemma":[0.2879622,0.002238661,0.6377879,0.0003776618,0.0002329691,0.0003156518,0.06656098,0.001002011,0.003521957],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009516038,"threshold_uncertainty_score":0.01657426,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01279322473618379,"score_gpt":0.3151136089758648,"score_spread":0.3023203842396811,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}