{"meta":{"query_hash":"918e7084900e","filters":{"venue":"Natural language processing"},"cohort_total":2,"direct_labels_cover":0,"predictions_cover":2,"exported":2,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/918e7084900e","api":"https://metacan.xera.ac/api/v1/cohort?venue=Natural+language+processing"},"results":[{"id":"W2475426007","doi":"10.1075/nlp.9.05hab","title":"Arabic preprocessing for Statistical Machine Translation","year":2012,"lang":"en","type":"book-chapter","venue":"Natural language processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Arabic; Preprocessor; Computer science; Natural language processing; Translation (biology); Machine translation; Artificial intelligence; Linguistics; Philosophy; Chemistry","score_opus":0.01989698859571343,"score_gpt":0.29778559573148655,"score_spread":0.2778886071357731,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2475426007","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032104745,0.041367218,0.68170106,0.00276356,0.004012957,0.00031321688,0.001595624,0.01123346,0.25380233],"genre_scores_gemma":[0.030514793,0.03346782,0.71410185,0.001585932,0.0022318328,0.00053095986,0.0044124536,0.004429098,0.2087253],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992717,0.00015896565,0.00005679534,0.0001300971,0.00035835183,0.000024169782],"domain_scores_gemma":[0.99902785,0.00039515,0.000052389725,0.00021642969,0.00027973382,0.000028476385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070880493,0.0019391692,0.00069329754,0.0020852885,0.0008792074,0.0024801015,0.00090902305,0.0009274623,0.056097288],"category_scores_gemma":[0.0025095053,0.0005148371,0.00063174183,0.0040768175,0.0005829395,0.0020397108,0.0010717175,0.0020889402,0.058444213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005313639,0.000041191663,0.00016354368,0.00083541684,0.000029262543,0.00023148324,0.00020551469,0.002020884,0.012822002,0.055457477,0.13941687,0.78872323],"study_design_scores_gemma":[0.000008660308,0.000044878787,0.0004376755,0.0002818006,0.000016965228,0.0008856605,0.00006760712,0.008333332,0.011935171,0.040272687,0.9376802,0.000035262434],"about_ca_topic_score_codex":0.000580272,"about_ca_topic_score_gemma":0.0009912975,"teacher_disagreement_score":0.056097288,"about_ca_system_score_codex":0.00074615376,"about_ca_system_score_gemma":0.0008219837,"threshold_uncertainty_score":0.18766409},"labels":[],"label_agreement":null},{"id":"W2548230849","doi":"10.1075/nlp.2.15mey","title":"Extracting knowledge-rich contexts for terminography","year":2001,"lang":"en","type":"book-chapter","venue":"Natural language processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":215,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Construct (python library); Computer science; Paralanguage; Domain knowledge; Domain (mathematical analysis); Knowledge extraction; Field (mathematics); Context (archaeology); Knowledge management; Natural language processing; Data science; Artificial intelligence; Psychology; Communication; Geography; Mathematics","score_opus":0.015630032468229135,"score_gpt":0.30286852901542444,"score_spread":0.2872384965471953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2548230849","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006694648,0.0056135827,0.9614586,0.00064546097,0.00021878842,0.00016266714,0.00046669674,0.0012028265,0.023536794],"genre_scores_gemma":[0.039570004,0.0051958705,0.94439495,0.00018241013,0.0001314663,0.00022337977,0.0015250286,0.00049381435,0.008283064],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990559,0.00035996974,0.00009406748,0.00015774896,0.00029602257,0.00003618616],"domain_scores_gemma":[0.9973213,0.0018525376,0.000098133016,0.00042182,0.0002590098,0.000047177462],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013881346,0.0006908762,0.00059115223,0.0030333297,0.0011315914,0.002879164,0.0011629503,0.00079222466,0.0047595976],"category_scores_gemma":[0.0059986236,0.00072562293,0.00091358204,0.0033122667,0.0014118414,0.0072092996,0.0023557968,0.0018680994,0.0036112852],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061908984,0.00003753115,0.0010432815,0.0013162857,0.00004172568,0.0006070177,0.003521673,0.0035422232,0.010266702,0.4065999,0.020075407,0.55288637],"study_design_scores_gemma":[0.000021082735,0.000032369415,0.0012840141,0.0011981163,0.000070265465,0.0013682056,0.00097531005,0.026884936,0.015587053,0.56500834,0.3875002,0.000070042675],"about_ca_topic_score_codex":0.00058003643,"about_ca_topic_score_gemma":0.0017609375,"teacher_disagreement_score":0.0047595976,"about_ca_system_score_codex":0.00078638503,"about_ca_system_score_gemma":0.0010683622,"threshold_uncertainty_score":0.015922487},"labels":[],"label_agreement":null}]}