{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":4,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":4,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"4c2cc3b5088f","filters":{"venue":"Corpora"}},"results":[{"id":"W2077469553","doi":"10.3366/cor.2013.0032","title":"Challenges in cross-linguistic corpus-assisted discourse studies","year":2013,"lang":"en","type":"article","venue":"Corpora","topic":"Discourse Analysis in Language Studies","field":"Arts and Humanities","cited_by":60,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Corpus linguistics; Linguistics; Focus (optics); Population; Contrastive linguistics; Sociology; Applied linguistics; Computer science; Philosophy","authors":[{"name":"Rachelle Vessey","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.16594680154034,"gpt":0.363258996867498,"spread":0.1973121953271579,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.4117099,0.001975871,0.003550462,0.03112472,0.02023621,0.03623689,0.01889822,0.009619286,0.01039582],"category_scores_gemma":[0.5745197,0.003212329,0.001259767,0.03175743,0.02574766,0.04698267,0.04933357,0.009424245,0.003786574],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01383978,"about_ca_system_score_gemma":0.02250857,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01651374,"about_ca_topic_score_gemma":0.01638891,"domain_scores_codex":[0.4409367,0.4801955,0.02323239,0.0190866,0.03397805,0.00257078],"domain_scores_gemma":[0.1779298,0.6698792,0.01490087,0.06952594,0.06475298,0.003011228],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002780728,0.0004877788,0.009779578,0.007594514,0.0005787511,0.00138188,0.3881352,0.002469599,0.002771949,0.2734862,0.0238877,0.2891489],"study_design_scores_gemma":[0.000192296,0.0001468118,0.008235915,0.0129167,0.0001668191,0.001307325,0.3078586,0.008631598,0.004146076,0.2297472,0.4263034,0.0003472566],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1059377,0.05395466,0.5663074,0.1050988,0.005574013,0.009988272,0.006332242,0.001837263,0.1449697],"genre_scores_gemma":[0.4293774,0.009896004,0.5085256,0.008797354,0.001716607,0.02489738,0.005216195,0.00162959,0.009943843],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4117099,"threshold_uncertainty_score":0.7254664,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2516495486","doi":"10.3366/cor.2016.0091","title":"Discourse relations and evaluation","year":2016,"lang":"en","type":"article","venue":"Corpora","topic":"Discourse Analysis in Language Studies","field":"Arts and Humanities","cited_by":14,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Simon Fraser University","keywords":"Adjective; Polarity (international relations); Linguistics; Noun; Adverb; Verb; Rhetorical question; Appraisal theory; Relation (database); Interpretation (philosophy); Negation; Psychology; Discourse marker; Social psychology; Computer science; Philosophy","authors":[{"name":"Radoslava Trnavac","is_ca":false},{"name":"Debopam Das","is_ca":false},{"name":"Maite Taboada","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05206585222861549,"gpt":0.3048947820211111,"spread":0.2528289297924956,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01699106,0.0007890756,0.0005751084,0.004854095,0.001599868,0.007201481,0.000766841,0.0009007191,0.007220205],"category_scores_gemma":[0.07684503,0.0003581058,0.0004150718,0.003713131,0.003884371,0.007711127,0.002372073,0.00101788,0.001079469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00346004,"about_ca_system_score_gemma":0.001594084,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002294485,"about_ca_topic_score_gemma":0.001515317,"domain_scores_codex":[0.9762127,0.01534496,0.001466946,0.002170991,0.004268325,0.0005361264],"domain_scores_gemma":[0.9444158,0.04063332,0.00465037,0.002324031,0.007383031,0.0005935457],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007320447,0.0002038021,0.02476609,0.001982481,0.0001592533,0.0004496688,0.04211986,0.00284938,0.01064631,0.4210929,0.008308962,0.4866892],"study_design_scores_gemma":[0.0001828888,0.0007011754,0.07961673,0.001996944,0.000348321,0.0008291106,0.0391973,0.02584294,0.02089526,0.5486378,0.2814542,0.0002972531],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.2976316,0.01391737,0.2687881,0.006321338,0.000808375,0.001398642,0.001707705,0.0009902433,0.4084365],"genre_scores_gemma":[0.9444078,0.001316649,0.04426012,0.0002885643,0.0002120249,0.0005091831,0.0006911645,0.0002978308,0.008016704],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01699106,"threshold_uncertainty_score":0.08985841,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1970536639","doi":"10.3366/cor.2007.2.1.97","title":"The Wenzhou Spoken Corpus","year":2007,"lang":"en","type":"article","venue":"Corpora","topic":"China's Ethnic Minorities and Relations","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"XPath; Markup language; XML; Computer science; Transcription (linguistics); Natural language processing; Information retrieval; Linguistics; Artificial intelligence; World Wide Web; XML database","authors":[{"name":"John Newman","is_ca":true},{"name":"Jingxia Lin","is_ca":false},{"name":"Terry Butler","is_ca":false},{"name":"Eric Zhang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02606158654718032,"gpt":0.3246620743798674,"spread":0.2986004878326871,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001842263,0.0003843027,0.0004482683,0.003069462,0.002022483,0.001424176,0.0007583881,0.0004001487,0.03135153],"category_scores_gemma":[0.004508547,0.0002614993,0.0001589749,0.006211312,0.000864343,0.001038273,0.002029757,0.000473971,0.005734429],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001676874,"about_ca_system_score_gemma":0.004514636,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03013672,"about_ca_topic_score_gemma":0.03655652,"domain_scores_codex":[0.9988416,0.0003304677,0.0002402693,0.0001954991,0.0002917983,0.0001003833],"domain_scores_gemma":[0.9975219,0.0008438833,0.0001145188,0.0005917901,0.0007633068,0.000164598],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.0006259677,0.0001674419,0.02343214,0.003815094,0.0001277154,0.004588209,0.04336408,0.002000437,0.02417262,0.1066647,0.4101672,0.3808744],"study_design_scores_gemma":[0.00007324634,0.0000525585,0.04765501,0.0001812425,0.00004625161,0.0005949491,0.007318745,0.001332548,0.005476285,0.005507349,0.9316925,0.00006937113],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.37717,0.003249293,0.0323571,0.00338961,0.0008495188,0.003062657,0.4132438,0.002386761,0.1642913],"genre_scores_gemma":[0.5359937,0.001734209,0.03818204,0.0005852701,0.0002217128,0.005392179,0.3546733,0.0008304566,0.06238724],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03135153,"threshold_uncertainty_score":0.1048813,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2606816378","doi":"10.3366/cor.2017.0108","title":"Subtopic annotation and automatic segmentation for news texts in Brazilian Portuguese","year":2017,"lang":"en","type":"article","venue":"Corpora","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Annotation; Computer science; Segmentation; Natural language processing; Portuguese; Artificial intelligence; Process (computing); Rhetorical question; Computational linguistics; Brazilian Portuguese; Linguistics","authors":[{"name":"Paula Christina Figueira Cardoso","is_ca":false},{"name":"Thiago Alexandre Salgueiro Pardo","is_ca":false},{"name":"Maite Taboada","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02239593382287337,"gpt":0.309573640430716,"spread":0.2871777066078426,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002296836,0.0006252937,0.0005383834,0.005475695,0.002273316,0.001520173,0.0006314571,0.0006191866,0.002932023],"category_scores_gemma":[0.01973695,0.0004849783,0.0003789311,0.004679099,0.001138305,0.001610097,0.001612125,0.0007802087,0.0008519304],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001327701,"about_ca_system_score_gemma":0.001719589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01028505,"about_ca_topic_score_gemma":0.01685056,"domain_scores_codex":[0.9972292,0.001256613,0.0002957184,0.000646111,0.0004599247,0.0001123603],"domain_scores_gemma":[0.9845588,0.01028089,0.001294017,0.001261719,0.002327778,0.0002768279],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001670755,0.0004039287,0.01421999,0.008394782,0.00008434681,0.003973095,0.1018887,0.006932846,0.1729766,0.0237,0.02706179,0.6386932],"study_design_scores_gemma":[0.000275418,0.0006038432,0.1578133,0.002487716,0.0003145382,0.00384638,0.04631741,0.09339608,0.1458811,0.01946096,0.5291846,0.0004186535],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7801127,0.005109657,0.1639031,0.001836758,0.0005291325,0.001232371,0.01519308,0.002700132,0.02938308],"genre_scores_gemma":[0.7440562,0.001985593,0.2173753,0.0001095783,0.0002110602,0.001384754,0.02605464,0.0008375862,0.00798521],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01028505,"threshold_uncertainty_score":0.02045035,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}