{"id":"W6891792210","doi":"10.48448/4nmw-gv06","title":"BenchIE^FL: A Manually Re-Annotated Fact-Based Open Information Extraction Benchmark","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Benchmark (surveying); Field (mathematics); Information extraction; Natural language; Open source; Natural (archaeology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005119844,0.001711468,0.0006435185,0.0106651,0.001610618,0.003913857,0.002999742,0.002176128,0.01611045],"category_scores_gemma":[0.03305057,0.0005548671,0.001082024,0.005660453,0.001070744,0.005484235,0.003840714,0.001518908,0.009818403],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00130745,"about_ca_system_score_gemma":0.002745163,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01039586,"about_ca_topic_score_gemma":0.01598575,"domain_scores_codex":[0.9944134,0.001177728,0.0008139015,0.001071775,0.00222021,0.0003029631],"domain_scores_gemma":[0.9792987,0.009962563,0.0009822659,0.004288061,0.004857679,0.0006107136],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009163553,0.0005468145,0.01178632,0.005411012,0.0003843679,0.001692782,0.001389844,0.01263132,0.01184065,0.02576145,0.5655841,0.362055],"study_design_scores_gemma":[0.0002856066,0.0002650835,0.01400633,0.0009598521,0.0001794613,0.001489651,0.001072072,0.05668492,0.03539788,0.02162092,0.8678049,0.0002332056],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.1027488,0.007966748,0.2679389,0.006136678,0.002171332,0.001568464,0.3101962,0.1761962,0.1250767],"genre_scores_gemma":[0.1079732,0.00113937,0.2474375,0.001031645,0.0002020319,0.0005664668,0.6147949,0.007964652,0.01889018],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01611045,"threshold_uncertainty_score":0.05389482,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02999519996806134,"score_gpt":0.3591961353819997,"score_spread":0.3292009354139384,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}