{"id":"W4287240206","doi":"10.48550/arxiv.2104.01940","title":"What's the best place for an AI conference, Vancouver or ______: Why\\n completing comparative questions is difficult","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Competence (human resources); Artificial intelligence; sort; Natural language processing; Set (abstract data type); Benchmark (surveying); Task (project management); Language model; Machine learning; Cognitive science; Information retrieval; Psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001213776,0.0006275238,0.0004599396,0.0009090464,0.005202269,0.005831003,0.0008869492,0.001947135,0.1090986],"category_scores_gemma":[0.005072177,0.0003129808,0.0003096687,0.001548047,0.001257503,0.003745839,0.001155905,0.002569962,0.03230689],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004923104,"about_ca_system_score_gemma":0.005168947,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.2045688,"about_ca_topic_score_gemma":0.5441618,"domain_scores_codex":[0.9993719,0.0001675239,0.00002984807,0.0001455148,0.0001548092,0.0001303342],"domain_scores_gemma":[0.9981905,0.0003140458,0.00005719401,0.0001195555,0.0006777131,0.000641022],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001843507,0.00009560121,0.004505753,0.0004164152,0.00004211059,0.0003261049,0.001483914,0.000918353,0.001851625,0.015962,0.7516475,0.2225663],"study_design_scores_gemma":[0.00002388574,0.00003630284,0.008327434,0.0003508819,0.00002280448,0.0001893071,0.004559695,0.001260754,0.001363249,0.008251397,0.9755625,0.00005178455],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.03478134,0.01146686,0.007539747,0.1142778,0.008191057,0.0001990651,0.005129084,0.002080483,0.8163346],"genre_scores_gemma":[0.2679535,0.01061613,0.0202053,0.008594787,0.001486415,0.0001672401,0.008091654,0.001004587,0.6818804],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2045688,"threshold_uncertainty_score":0.4067562,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2179718214704391,"score_gpt":0.2610461724021927,"score_spread":0.04307435093175357,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}