{"id":"W2890045624","doi":"10.18653/v1/w19-4002","title":"WiRe57 : A Fine-Grained Benchmark for Open Information Extraction","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Granularity; Task (project management); Benchmark (surveying); Tuple; Inference; Information extraction; Annotation; Information retrieval; Data mining; Artificial intelligence; Programming language; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006962998,0.001750413,0.001039682,0.007564172,0.002201628,0.004032326,0.003156134,0.00221413,0.0116916],"category_scores_gemma":[0.03440406,0.000784937,0.001196338,0.009032816,0.001121994,0.006248159,0.003728761,0.001617584,0.007034814],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001594928,"about_ca_system_score_gemma":0.002347861,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008511967,"about_ca_topic_score_gemma":0.01029396,"domain_scores_codex":[0.9909884,0.002278953,0.001695105,0.001328521,0.003068176,0.0006407347],"domain_scores_gemma":[0.9756631,0.01022553,0.00103745,0.007564204,0.004931328,0.0005783432],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001444598,0.001339696,0.0134969,0.005085054,0.0004864424,0.0008745419,0.001636535,0.0294151,0.02114381,0.03341568,0.4027698,0.4888918],"study_design_scores_gemma":[0.0007184108,0.001381999,0.02315363,0.001173559,0.0003352362,0.00206364,0.002611093,0.1843782,0.116958,0.05192712,0.6149265,0.0003726487],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2327565,0.0109426,0.3383929,0.004136631,0.001543199,0.002161338,0.1702471,0.1224635,0.1173562],"genre_scores_gemma":[0.2262936,0.002368724,0.3151437,0.0006516641,0.0002417254,0.001204951,0.4215998,0.01262172,0.01987403],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0116916,"threshold_uncertainty_score":0.03911227,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0208554406895695,"score_gpt":0.3239822860998366,"score_spread":0.3031268454102671,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}