{"id":"W4385570394","doi":"10.18653/v1/2023.findings-acl.45","title":"Can Language Models Be Specific? How?","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Office of the Vice Chancellor for Research, University of Illinois at Chicago; University of Illinois at Urbana-Champaign; Center for Cognitive Computing Systems Research; National Science Foundation","keywords":"Computer science; Benchmark (surveying); Language model; Machine learning; Artificial intelligence; Preference; Natural language processing; Measure (data warehouse); Test (biology); Data mining","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008122038,0.001446065,0.0007907113,0.001012075,0.0005165713,0.003655695,0.001200473,0.001920927,0.002733026],"category_scores_gemma":[0.05102683,0.0007543011,0.001016307,0.0007361061,0.001592561,0.01068737,0.001877619,0.004024312,0.002461368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008692229,"about_ca_system_score_gemma":0.001321883,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002730452,"about_ca_topic_score_gemma":0.004437945,"domain_scores_codex":[0.9920623,0.004885933,0.0003090973,0.001716806,0.0006525789,0.0003732921],"domain_scores_gemma":[0.9594488,0.031343,0.001846672,0.00492198,0.001629514,0.0008101268],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002252941,0.0006078814,0.1825662,0.001616827,0.0009098547,0.001492955,0.007137109,0.166135,0.02934824,0.03582111,0.02526473,0.5468472],"study_design_scores_gemma":[0.000118607,0.000350579,0.02993627,0.0003965041,0.0004028634,0.001783022,0.003243772,0.7221918,0.02070301,0.2020081,0.01864765,0.0002178977],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5070424,0.001609004,0.4547672,0.01038129,0.0002954531,0.0002371695,0.003067994,0.004517252,0.01808227],"genre_scores_gemma":[0.962219,0.0002888897,0.0327046,0.0009717306,0.00004721241,0.00006683636,0.002072194,0.0002728795,0.001356598],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008122038,"threshold_uncertainty_score":0.04295397,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0321223790638185,"score_gpt":0.2750230125668324,"score_spread":0.2429006335030139,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}