{"meta":{"query_hash":"3a76d6254d98","filters":{"venue":"ACM Transactions on Speech and Language Processing"},"cohort_total":4,"direct_labels_cover":0,"predictions_cover":4,"exported":4,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/3a76d6254d98","api":"https://metacan.xera.ac/api/v1/cohort?venue=ACM+Transactions+on+Speech+and+Language+Processing"},"results":[{"id":"W2052227241","doi":"10.1145/1289600.1289602","title":"Adaptive text correction with Web-crawled domain-dependent dictionaries","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Trigram; Vocabulary; Natural language processing; Domain (mathematical analysis); Information retrieval; Word (group theory); Artificial intelligence; Web page; Crawling; Punctuation; Linguistics; World Wide Web","score_opus":0.008003122257323006,"score_gpt":0.25074320704803005,"score_spread":0.24274008479070705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052227241","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2529732,0.0011843689,0.65903944,0.0003326262,0.000270491,0.0008032728,0.0020172053,0.07972155,0.003657857],"genre_scores_gemma":[0.30058587,0.0005381117,0.6807449,0.00019249314,0.000079642625,0.000465995,0.0064838016,0.0036890793,0.007220119],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99799275,0.0006028052,0.0002432243,0.00052234647,0.00055748207,0.000081434046],"domain_scores_gemma":[0.9814299,0.008723803,0.0011015758,0.0050277743,0.003455435,0.00026141197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001528184,0.0012171477,0.0012574226,0.0031464489,0.00065421703,0.0012170492,0.0013626122,0.0008773308,0.0017663983],"category_scores_gemma":[0.021536484,0.00070223695,0.000598648,0.002912188,0.0005772642,0.0025585066,0.00148824,0.0010218711,0.0028891729],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007466272,0.00073962036,0.013504569,0.0009826246,0.00026950525,0.0007490371,0.001371365,0.028967073,0.09859189,0.0018542091,0.012497717,0.8397259],"study_design_scores_gemma":[0.000277876,0.0006226463,0.012742062,0.00012757239,0.0003289746,0.0016742971,0.0008417614,0.68009555,0.26705912,0.003618448,0.032420315,0.00019149201],"about_ca_topic_score_codex":0.0047981422,"about_ca_topic_score_gemma":0.0073138997,"teacher_disagreement_score":0.0047981422,"about_ca_system_score_codex":0.0004148117,"about_ca_system_score_gemma":0.0011409895,"threshold_uncertainty_score":0.009540439},"labels":[],"label_agreement":null},{"id":"W2091014931","doi":"10.1145/1177055.1177057","title":"Confidence estimation for NLP applications","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Artificial intelligence; Natural language processing; Estimation; Confidence interval; Natural language; Machine learning; Statistics; Mathematics","score_opus":0.00987436474930131,"score_gpt":0.28429790564261975,"score_spread":0.27442354089331844,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2091014931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023728015,0.0010895975,0.99248123,0.0004443128,0.000075426826,0.000057901063,0.0002662093,0.0011505857,0.002061976],"genre_scores_gemma":[0.32846558,0.002281546,0.6604008,0.0007382846,0.0010192234,0.0006410613,0.0024488322,0.0011918071,0.002812807],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97368944,0.011618336,0.0022348706,0.0032226054,0.008601945,0.0006328522],"domain_scores_gemma":[0.83602744,0.13370478,0.0056683305,0.010809118,0.012843124,0.0009473805],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016697397,0.0013999122,0.0018444652,0.0052940473,0.0012853,0.0054213847,0.00334534,0.003020306,0.0074038934],"category_scores_gemma":[0.22055812,0.00100757,0.0012833207,0.0048166686,0.002120306,0.007606319,0.004191077,0.0049284445,0.003392886],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049091625,0.00014402978,0.006720331,0.0012487725,0.00024209241,0.00048770805,0.00056192116,0.12106341,0.005415148,0.25226086,0.020061241,0.59130347],"study_design_scores_gemma":[0.00004395898,0.00008241483,0.0013549677,0.00024058777,0.000055538098,0.00042134442,0.000104675375,0.65002495,0.0066668633,0.32409078,0.016835846,0.00007799829],"about_ca_topic_score_codex":0.0023666092,"about_ca_topic_score_gemma":0.0010898344,"teacher_disagreement_score":0.016697397,"about_ca_system_score_codex":0.0014133352,"about_ca_system_score_gemma":0.0017412083,"threshold_uncertainty_score":0.088305295},"labels":[],"label_agreement":null},{"id":"W2109163800","doi":"10.1145/1187415.1187417","title":"A statistical model for near-synonym choice","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Synonym (taxonomy); Artificial intelligence; Natural language processing; Task (project management); Context (archaeology); Machine translation; Thesaurus; Mutual information; Information retrieval; Machine learning","score_opus":0.01971804255985616,"score_gpt":0.31857975632009145,"score_spread":0.2988617137602353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109163800","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016976943,0.00021293604,0.9791524,0.000506069,0.00006433646,0.00009699557,0.00049630535,0.0005595031,0.001934551],"genre_scores_gemma":[0.59344393,0.0005410455,0.39135447,0.00053533196,0.0004672578,0.0010174061,0.0028528816,0.00052279193,0.00926476],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942364,0.0027181036,0.0003914164,0.0012748787,0.0010980079,0.00028132155],"domain_scores_gemma":[0.9796816,0.015149493,0.0015252143,0.0016194872,0.0016586988,0.0003654337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077102073,0.00088695186,0.001535097,0.0046470505,0.001216191,0.0023656366,0.0037324168,0.0020973834,0.006610387],"category_scores_gemma":[0.02908325,0.0009377501,0.0018020209,0.0047803647,0.0021323692,0.0065434733,0.0022221326,0.0030292806,0.0029168688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005407126,0.00053368835,0.013181713,0.0003272263,0.0006339837,0.0007943607,0.00097515294,0.3685607,0.006033978,0.3656934,0.012716758,0.23000842],"study_design_scores_gemma":[0.000026786825,0.00004102195,0.00084774324,0.000014006671,0.000025491036,0.0001286284,0.000027773705,0.8600659,0.00049306755,0.13690296,0.001387926,0.000038727547],"about_ca_topic_score_codex":0.0051065097,"about_ca_topic_score_gemma":0.008466461,"teacher_disagreement_score":0.0077102073,"about_ca_system_score_codex":0.0014510335,"about_ca_system_score_gemma":0.0019570682,"threshold_uncertainty_score":0.040775955},"labels":[],"label_agreement":null},{"id":"W2130998770","doi":"10.1145/1596517.1596518","title":"Extrinsic summarization evaluation","year":2009,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Topic Modeling","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Sixth Framework Programme","keywords":"Automatic summarization; Computer science; Task (project management); Information retrieval; Audit; Human–computer interaction; Natural language processing; Artificial intelligence","score_opus":0.024218930316466627,"score_gpt":0.29291025333555487,"score_spread":0.26869132301908827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2130998770","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28006172,0.005843942,0.6303888,0.0012019642,0.0014384667,0.0037494241,0.013576097,0.02590509,0.03783436],"genre_scores_gemma":[0.66572064,0.0013493291,0.2719752,0.0006516043,0.000780337,0.0020881465,0.034933437,0.0029828935,0.019518495],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97795564,0.011013883,0.002280524,0.0018795349,0.0064198915,0.000450543],"domain_scores_gemma":[0.9409888,0.027940696,0.0033251138,0.005820352,0.021047236,0.0008779025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012293993,0.0021904244,0.0012839051,0.0030159126,0.00083377556,0.00304517,0.0016740845,0.0014544902,0.010098683],"category_scores_gemma":[0.06587509,0.0002529407,0.00080758554,0.0016037388,0.0005683884,0.0030679258,0.0018361142,0.0011024199,0.0041175596],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037885576,0.0008779663,0.009817999,0.003564328,0.000605443,0.00043589008,0.0009977736,0.014730721,0.06241746,0.0032411434,0.039600782,0.859922],"study_design_scores_gemma":[0.0010370007,0.009488849,0.0650547,0.0007347289,0.0016352694,0.0030045542,0.0029590465,0.42778963,0.289561,0.009641589,0.18836515,0.0007285698],"about_ca_topic_score_codex":0.00080244784,"about_ca_topic_score_gemma":0.0011147424,"teacher_disagreement_score":0.012293993,"about_ca_system_score_codex":0.00080430816,"about_ca_system_score_gemma":0.0008347191,"threshold_uncertainty_score":0.06501764},"labels":[],"label_agreement":null}]}