{"id":"W3119519251","doi":"10.1609/aaai.v35i16.17681","title":"What's the Best Place for an AI Conference, Vancouver or _______: Why Completing Comparative Questions is Difficult","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Competence (human resources); Artificial intelligence; sort; Natural language processing; Benchmark (surveying); Set (abstract data type); Task (project management); Language model; Machine learning; Cognitive science; Psychology; Information retrieval","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0006279231,0.0003753137,0.00047748,0.00008983783,0.0007642473,0.001582213,0.002741844,0.0001339534,0.000117295],"category_scores_gemma":[0.0005306515,0.0002381089,0.0001665029,0.0008573991,0.0004113898,0.00161683,0.0005898537,0.0005192282,0.00004105818],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007338214,"about_ca_system_score_gemma":0.0004570909,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001332753,"about_ca_topic_score_gemma":0.001004773,"domain_scores_codex":[0.9970023,0.00007508096,0.000804289,0.0009179534,0.0006524864,0.0005478896],"domain_scores_gemma":[0.9948307,0.0004512658,0.0005060356,0.0006950467,0.003365288,0.0001517125],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001436876,0.0004376681,0.00005307218,0.00007296237,0.00005974331,0.000001131675,0.01343431,0.0008416687,0.01285446,0.945792,0.004730644,0.02157866],"study_design_scores_gemma":[0.0001118273,0.0004914767,0.00002711821,0.0007764627,0.00005741406,0.00001286065,0.02374031,0.6707511,0.1974535,0.1026309,0.003428069,0.0005190196],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1589772,0.0001385438,0.7809706,0.04471461,0.00431093,0.002649602,0.00006365814,0.0003330032,0.007841812],"genre_scores_gemma":[0.984163,0.00007141113,0.01136894,0.002643366,0.0001897316,0.000136037,0.000002496785,0.00001898657,0.001406061],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8431611,"threshold_uncertainty_score":0.9994543,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2076591633305094,"score_gpt":0.3647780794122421,"score_spread":0.1571189160817327,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}