{"meta":{"query_hash":"4c2cc3b5088f","filters":{"venue":"Corpora"},"cohort_total":4,"direct_labels_cover":0,"predictions_cover":4,"exported":4,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/4c2cc3b5088f","api":"https://metacan.xera.ac/api/v1/cohort?venue=Corpora"},"results":[{"id":"W1970536639","doi":"10.3366/cor.2007.2.1.97","title":"The Wenzhou Spoken Corpus","year":2007,"lang":"en","type":"article","venue":"Corpora","topic":"China's Ethnic Minorities and Relations","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"XPath; Markup language; XML; Computer science; Transcription (linguistics); Natural language processing; Information retrieval; Linguistics; Artificial intelligence; World Wide Web; XML database","score_opus":0.026061586547180323,"score_gpt":0.3246620743798674,"score_spread":0.29860048783268706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1970536639","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37717,0.003249293,0.032357104,0.0033896095,0.00084951875,0.0030626573,0.41324377,0.0023867611,0.16429129],"genre_scores_gemma":[0.5359937,0.0017342088,0.038182043,0.00058527006,0.00022171276,0.0053921794,0.35467327,0.00083045656,0.062387235],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9988416,0.00033046765,0.00024026929,0.00019549913,0.00029179832,0.00010038329],"domain_scores_gemma":[0.99752194,0.0008438833,0.00011451879,0.00059179007,0.0007633068,0.000164598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018422629,0.0003843027,0.00044826834,0.0030694618,0.0020224834,0.0014241759,0.00075838814,0.00040014868,0.031351525],"category_scores_gemma":[0.004508547,0.00026149928,0.00015897494,0.006211312,0.000864343,0.0010382734,0.0020297572,0.00047397104,0.005734429],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062596775,0.00016744186,0.023432136,0.003815094,0.00012771544,0.004588209,0.04336408,0.0020004374,0.024172619,0.1066647,0.41016716,0.38087443],"study_design_scores_gemma":[0.000073246345,0.0000525585,0.047655005,0.00018124253,0.000046251615,0.00059494906,0.0073187454,0.0013325477,0.0054762852,0.0055073486,0.9316925,0.00006937113],"about_ca_topic_score_codex":0.030136725,"about_ca_topic_score_gemma":0.036556516,"teacher_disagreement_score":0.031351525,"about_ca_system_score_codex":0.0016768737,"about_ca_system_score_gemma":0.0045146355,"threshold_uncertainty_score":0.104881346},"labels":[],"label_agreement":null},{"id":"W2077469553","doi":"10.3366/cor.2013.0032","title":"Challenges in cross-linguistic corpus-assisted discourse studies","year":2013,"lang":"en","type":"article","venue":"Corpora","topic":"Discourse Analysis in Language Studies","field":"Arts and Humanities","cited_by":60,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Corpus linguistics; Linguistics; Focus (optics); Population; Contrastive linguistics; Sociology; Applied linguistics; Computer science; Philosophy","score_opus":0.16594680154034003,"score_gpt":0.36325899686749796,"score_spread":0.19731219532715794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077469553","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10593766,0.05395466,0.5663074,0.1050988,0.0055740126,0.009988272,0.0063322424,0.0018372635,0.14496966],"genre_scores_gemma":[0.4293774,0.009896004,0.5085256,0.008797354,0.0017166068,0.024897384,0.0052161952,0.0016295902,0.009943843],"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.44093674,0.4801955,0.023232391,0.0190866,0.03397805,0.0025707802],"domain_scores_gemma":[0.17792977,0.6698792,0.014900867,0.06952594,0.06475298,0.003011228],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.41170993,0.001975871,0.0035504622,0.031124717,0.020236215,0.03623689,0.018898217,0.009619286,0.010395822],"category_scores_gemma":[0.5745197,0.0032123285,0.0012597673,0.031757433,0.025747662,0.04698267,0.049333572,0.009424245,0.0037865744],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027807275,0.0004877788,0.009779578,0.0075945137,0.00057875115,0.0013818798,0.38813522,0.0024695985,0.002771949,0.27348617,0.023887703,0.28914893],"study_design_scores_gemma":[0.000192296,0.00014681183,0.008235915,0.012916704,0.00016681905,0.0013073253,0.30785862,0.008631598,0.0041460763,0.22974716,0.42630345,0.00034725663],"about_ca_topic_score_codex":0.016513735,"about_ca_topic_score_gemma":0.016388906,"teacher_disagreement_score":0.41170993,"about_ca_system_score_codex":0.013839782,"about_ca_system_score_gemma":0.02250857,"threshold_uncertainty_score":0.7254664},"labels":[],"label_agreement":null},{"id":"W2516495486","doi":"10.3366/cor.2016.0091","title":"Discourse relations and evaluation","year":2016,"lang":"en","type":"article","venue":"Corpora","topic":"Discourse Analysis in Language Studies","field":"Arts and Humanities","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Simon Fraser University","keywords":"Adjective; Polarity (international relations); Linguistics; Noun; Adverb; Verb; Rhetorical question; Appraisal theory; Relation (database); Interpretation (philosophy); Negation; Psychology; Discourse marker; Social psychology; Computer science; Philosophy","score_opus":0.05206585222861549,"score_gpt":0.3048947820211111,"score_spread":0.2528289297924956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2516495486","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2976316,0.013917371,0.26878813,0.0063213375,0.00080837496,0.001398642,0.0017077047,0.0009902433,0.4084365],"genre_scores_gemma":[0.9444078,0.0013166491,0.044260122,0.00028856433,0.00021202494,0.0005091831,0.0006911645,0.00029783082,0.008016704],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9762127,0.015344963,0.0014669464,0.0021709907,0.004268325,0.0005361264],"domain_scores_gemma":[0.9444158,0.04063332,0.00465037,0.002324031,0.007383031,0.0005935457],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016991058,0.0007890756,0.0005751084,0.004854095,0.0015998678,0.007201481,0.000766841,0.00090071914,0.0072202054],"category_scores_gemma":[0.076845035,0.00035810584,0.00041507176,0.0037131312,0.0038843711,0.0077111274,0.0023720735,0.0010178796,0.0010794685],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007320447,0.00020380209,0.02476609,0.0019824815,0.00015925326,0.00044966876,0.04211986,0.0028493798,0.010646311,0.4210929,0.008308962,0.48668924],"study_design_scores_gemma":[0.00018288882,0.0007011754,0.07961673,0.001996944,0.00034832105,0.00082911056,0.039197303,0.025842942,0.020895263,0.5486378,0.28145424,0.00029725308],"about_ca_topic_score_codex":0.002294485,"about_ca_topic_score_gemma":0.0015153165,"teacher_disagreement_score":0.016991058,"about_ca_system_score_codex":0.0034600399,"about_ca_system_score_gemma":0.0015940837,"threshold_uncertainty_score":0.08985841},"labels":[],"label_agreement":null},{"id":"W2606816378","doi":"10.3366/cor.2017.0108","title":"Subtopic annotation and automatic segmentation for news texts in Brazilian Portuguese","year":2017,"lang":"en","type":"article","venue":"Corpora","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Annotation; Computer science; Segmentation; Natural language processing; Portuguese; Artificial intelligence; Process (computing); Rhetorical question; Computational linguistics; Brazilian Portuguese; Linguistics","score_opus":0.02239593382287337,"score_gpt":0.309573640430716,"score_spread":0.28717770660784264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2606816378","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7801127,0.005109657,0.16390307,0.0018367576,0.0005291325,0.0012323706,0.015193081,0.0027001319,0.02938308],"genre_scores_gemma":[0.7440562,0.0019855928,0.2173753,0.00010957827,0.00021106019,0.0013847542,0.026054641,0.00083758624,0.00798521],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99722916,0.0012566132,0.00029571838,0.000646111,0.00045992466,0.00011236027],"domain_scores_gemma":[0.98455876,0.010280893,0.0012940168,0.0012617189,0.002327778,0.00027682792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022968363,0.0006252937,0.00053838344,0.0054756952,0.0022733158,0.0015201734,0.0006314571,0.00061918655,0.0029320228],"category_scores_gemma":[0.01973695,0.00048497829,0.00037893114,0.004679099,0.0011383047,0.0016100968,0.0016121254,0.0007802087,0.0008519304],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016707552,0.0004039287,0.014219993,0.0083947815,0.00008434681,0.0039730947,0.101888664,0.006932846,0.17297664,0.023700004,0.02706179,0.6386932],"study_design_scores_gemma":[0.000275418,0.00060384325,0.15781331,0.0024877165,0.00031453816,0.0038463797,0.046317406,0.09339608,0.14588109,0.019460965,0.5291846,0.0004186535],"about_ca_topic_score_codex":0.010285049,"about_ca_topic_score_gemma":0.01685056,"teacher_disagreement_score":0.010285049,"about_ca_system_score_codex":0.001327701,"about_ca_system_score_gemma":0.0017195885,"threshold_uncertainty_score":0.020450354},"labels":[],"label_agreement":null}]}