{"id":"W6890357168","doi":"10.35111/wxrn-qr14","title":"Benchmarks for Open Relation Extraction","year":2014,"lang":"en","type":"other","venue":"Americanae (AECID Library)","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Relationship extraction; Relation (database); Scripting language; Task (project management); Set (abstract data type); Binary relation; Benchmark (surveying); Sentence; Training set; Information extraction","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0001994268,0.0006107736,0.0008508872,0.0007304078,0.0001374194,0.0001625748,0.001316917,0.0004719558,0.03867862],"category_scores_gemma":[0.0001219116,0.0006353815,0.0002365219,0.0007519081,0.000229262,0.001434107,0.000316801,0.0003785192,0.005298231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001006733,"about_ca_system_score_gemma":0.0002417699,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02228645,"about_ca_topic_score_gemma":0.000163431,"domain_scores_codex":[0.9973147,0.0002235111,0.0005091275,0.001021701,0.0003647397,0.0005662547],"domain_scores_gemma":[0.9968961,0.0002478301,0.00143151,0.001144414,0.00002608436,0.0002540415],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008869233,0.00006554928,0.005818995,0.00003828631,0.000157826,0.000003719304,0.00001470583,0.000008822767,0.00005911824,0.002027142,0.9788643,0.01285289],"study_design_scores_gemma":[0.0007333406,0.0002039736,0.01883892,0.0001718811,0.00013668,0.000004048259,0.0000270018,0.0002173799,0.00007941042,0.00056721,0.9782302,0.0007899902],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0000847413,0.000476194,0.003650552,0.0005848579,0.0009420436,0.002711678,0.0008620791,0.001003927,0.9896839],"genre_scores_gemma":[0.002265161,0.0001384586,0.03617116,0.001001883,0.001778267,0.0007290641,0.008764902,0.004497933,0.9446532],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.04503076,"threshold_uncertainty_score":0.9996098,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01360695840352202,"score_gpt":0.2791115131780302,"score_spread":0.2655045547745082,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}