{"id":"W6929236097","doi":"10.48448/x501-ff87","title":"It Takes Two to Tango: Navigating Conceptualizations of NLP Tasks and Measurements of Performance","year":2022,"lang":"en","type":"other","venue":"Open MIND","topic":"Enzyme Structure and Function","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Merck Canada Inc. (Canada)","funders":"","keywords":"Coreference; Operationalization; Taxonomy (biology); Benchmark (surveying); Resolution (logic)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.000248796,0.0001066156,0.0002305846,0.00004443014,0.00006267212,0.00003114418,0.0002692109,0.00004797939,0.04676053],"category_scores_gemma":[0.00004001697,0.00009865198,0.00001615133,0.0001337336,0.00006823507,0.00007040568,0.0001368505,0.00006290913,0.00002975859],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001754456,"about_ca_system_score_gemma":0.00004994831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000215701,"about_ca_topic_score_gemma":0.0003197155,"domain_scores_codex":[0.9991621,0.00004638786,0.0002220132,0.000222766,0.0002475937,0.0000991219],"domain_scores_gemma":[0.9994361,0.00001453706,0.0002730775,0.0002054967,0.00003625199,0.00003450142],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00007803918,0.00008033191,0.008564838,0.0002527304,0.00009878317,0.000002027348,0.01252098,0.0002126495,0.7958243,0.00007412177,0.1354563,0.04683485],"study_design_scores_gemma":[0.0006793519,0.000240714,0.0001932707,0.0006179609,0.00008511348,0.000004971086,0.001462617,0.00001824948,0.2829956,0.000007206831,0.7133855,0.0003095116],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.06275878,0.0002705247,0.0001682517,0.00004210673,0.0004868387,0.0007133784,0.0003286923,0.000003797714,0.9352276],"genre_scores_gemma":[0.6029398,0.0000330077,0.05381107,0.0003456669,0.0003819241,0.0001325058,0.000369569,0.0003311473,0.3416553],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5935723,"threshold_uncertainty_score":0.9541109,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06449320705796197,"score_gpt":0.3354499216564781,"score_spread":0.2709567145985161,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}