{"id":"W6929236097","doi":"10.48448/x501-ff87","title":"It Takes Two to Tango: Navigating Conceptualizations of NLP Tasks and Measurements of Performance","year":2022,"lang":"en","type":"other","venue":"Open MIND","topic":"Enzyme Structure and Function","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Merck Canada Inc. (Canada)","funders":"","keywords":"Coreference; Operationalization; Taxonomy (biology); Benchmark (surveying); Resolution (logic)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2887681,0.002068032,0.002661588,0.02093294,0.00480216,0.02488804,0.006149123,0.004252868,0.004980225],"category_scores_gemma":[0.5283484,0.001641328,0.003164585,0.02422399,0.02237304,0.05546333,0.01393452,0.0124779,0.001668458],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01647679,"about_ca_system_score_gemma":0.01771446,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01846975,"about_ca_topic_score_gemma":0.0128495,"domain_scores_codex":[0.7296119,0.2107708,0.01719823,0.01386211,0.02686821,0.001688821],"domain_scores_gemma":[0.3358456,0.5249046,0.04113347,0.06158792,0.0339027,0.002625768],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000302756,0.0001170069,0.02097827,0.008915282,0.001088723,0.00009464085,0.05992363,0.002462528,0.0007559082,0.5799444,0.02982461,0.2955924],"study_design_scores_gemma":[0.0001074135,0.0002296908,0.01605541,0.02169745,0.0006352292,0.0001741207,0.03061491,0.004007723,0.001621992,0.7970309,0.1274106,0.0004145857],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06732693,0.09242431,0.5207141,0.2301559,0.003814848,0.001051227,0.005505073,0.002463678,0.07654402],"genre_scores_gemma":[0.6590906,0.02198006,0.2697654,0.03288773,0.001017541,0.00574977,0.004664775,0.001743027,0.003100992],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2887681,"threshold_uncertainty_score":0.8770756,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06449320705796197,"score_gpt":0.3354499216564781,"score_spread":0.2709567145985161,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}