{"meta":{"query_hash":"2d56ba59d3cf","filters":{"venue":"Findings of the Association for Computational Linguistics: NAACL 2022"},"cohort_total":2,"direct_labels_cover":0,"predictions_cover":2,"exported":2,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/2d56ba59d3cf","api":"https://metacan.xera.ac/api/v1/cohort?venue=Findings+of+the+Association+for+Computational+Linguistics%3A+NAACL+2022"},"results":[{"id":"W3206996280","doi":"10.18653/v1/2022.findings-naacl.184","title":"CCQA: A New Web-Scale Question Answering Dataset for Model Pre-Training","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: NAACL 2022","topic":"Topic Modeling","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Question answering; Computer science; Open domain; Task (project management); Language model; Domain (mathematical analysis); Artificial intelligence; Natural language processing; Scale (ratio); Training set; Information retrieval; Resource (disambiguation); Natural language; Natural language understanding","score_opus":0.024946483409200453,"score_gpt":0.2847366473702726,"score_spread":0.25979016396107213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3206996280","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12543306,0.0044658626,0.12004773,0.0033744194,0.0012015299,0.0028411355,0.6442641,0.08274676,0.015625397],"genre_scores_gemma":[0.0571815,0.00034489608,0.073013596,0.0008548268,0.00016924723,0.0017994597,0.86189425,0.0010798576,0.0036624174],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9962774,0.001265941,0.00036322634,0.0011188239,0.0007297485,0.00024496793],"domain_scores_gemma":[0.9919264,0.0030654683,0.00037140463,0.0020403767,0.0019216799,0.00067472784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032401495,0.0027537001,0.0013396178,0.0042916546,0.001685424,0.0019022243,0.0041624773,0.0035417746,0.0070302594],"category_scores_gemma":[0.014567032,0.0007344283,0.0021314938,0.0034177843,0.0009826336,0.0036684566,0.0038213255,0.0036911604,0.010096237],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067070464,0.001996464,0.011246784,0.0027336332,0.00033279418,0.0005369476,0.0010447161,0.012562797,0.012868493,0.0050311196,0.8091158,0.14185971],"study_design_scores_gemma":[0.0012488556,0.0010007857,0.034746327,0.00045643596,0.00034814785,0.0016330238,0.0014981083,0.28053483,0.030064585,0.021643553,0.6264046,0.0004207354],"about_ca_topic_score_codex":0.021281388,"about_ca_topic_score_gemma":0.040030234,"teacher_disagreement_score":0.021281388,"about_ca_system_score_codex":0.0019118458,"about_ca_system_score_gemma":0.003238328,"threshold_uncertainty_score":0.042315066},"labels":[],"label_agreement":null},{"id":"W4287889735","doi":"10.18653/v1/2022.findings-naacl.151","title":"Great Power, Great Responsibility: Recommendations for Reducing Energy for Training Language Models","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: NAACL 2022","topic":"Topic Modeling","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Massachusetts Green High Performance Computing Center; Leukemia and Lymphoma Society of Canada; National Science Foundation","keywords":"Computer science; Energy consumption; Inference; Pace; Cloud computing; Transformer; Efficient energy use; Machine learning; Artificial intelligence; Risk analysis (engineering); Engineering","score_opus":0.03769841022498763,"score_gpt":0.29149186124092274,"score_spread":0.2537934510159351,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287889735","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044264193,0.045963854,0.2954997,0.576948,0.0075787874,0.00056601677,0.002152587,0.013649608,0.053215016],"genre_scores_gemma":[0.08480859,0.06035386,0.67628574,0.10067007,0.004559816,0.0022521634,0.004844624,0.007806075,0.05841904],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99483323,0.0023193494,0.00035415197,0.00052261807,0.0015862206,0.0003844307],"domain_scores_gemma":[0.96031874,0.02444462,0.00082213397,0.003493205,0.0083966255,0.0025246877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00952204,0.0020443585,0.0011778746,0.0029080098,0.0018693338,0.0058419546,0.00773802,0.006517267,0.0493919],"category_scores_gemma":[0.05165201,0.001655877,0.0013954372,0.0043852483,0.0033481666,0.020110983,0.003375087,0.009638011,0.024651064],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018452817,0.0003196915,0.0011321657,0.0010314243,0.000064507316,0.000108614535,0.00029755654,0.0071336864,0.0015853194,0.05173662,0.60894597,0.32746005],"study_design_scores_gemma":[0.0003150504,0.00012334644,0.0014413437,0.0016347109,0.000086418535,0.00023168103,0.0010205891,0.028409865,0.0029720473,0.15742996,0.8061858,0.00014924223],"about_ca_topic_score_codex":0.01761981,"about_ca_topic_score_gemma":0.03374566,"teacher_disagreement_score":0.0493919,"about_ca_system_score_codex":0.002918367,"about_ca_system_score_gemma":0.006886862,"threshold_uncertainty_score":0.16523236},"labels":[],"label_agreement":null}]}