{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":60,"total_is_capped":false,"direct_labels_cover":1,"predictions_cover":60,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"24b6424d1b26","filters":{"venue":"American Journal of Evaluation"}},"results":[{"id":"W2043887666","doi":"10.1177/1098214005278752","title":"Is Sustainability Possible? A Review and Commentary on Empirical Studies of Program Sustainability","year":2005,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":715,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Sustainability; Champion; Sustainability organizations; Empirical research; Program evaluation; Sustainability science; Social sustainability; Business; Environmental resource management; Political science; Economics; Public administration","authors":[{"name":"Mary Ann Scheirer","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.712236943529094,"gpt":0.7821963698676211,"spread":0.06995942633852714,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04766485,0.001465676,0.005013492,0.009062669,0.00166231,0.004907587,0.005507357,0.00648641,0.004180246],"category_scores_gemma":[0.2053299,0.0008413101,0.002834513,0.01742792,0.007057196,0.007098801,0.002422823,0.006529585,0.001124083],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01258433,"about_ca_system_score_gemma":0.03787336,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02767613,"about_ca_topic_score_gemma":0.04067196,"domain_scores_codex":[0.9571732,0.02216136,0.006279396,0.002545046,0.01097141,0.000869579],"domain_scores_gemma":[0.6365977,0.3168034,0.01247436,0.003082047,0.0297127,0.001329777],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002663283,0.00004963698,0.0006842408,0.216451,0.0008997972,0.0004773735,0.002961692,0.0004237046,0.0001948467,0.02724699,0.4412705,0.3090739],"study_design_scores_gemma":[0.00008455742,0.00008503868,0.001698026,0.3436534,0.0009639572,0.0004045386,0.003290785,0.0001369973,0.0002126601,0.008895287,0.6405097,0.00006501991],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.000135166,0.9410856,0.0001585876,0.0532549,0.004318345,0.00001790555,0.00005517929,0.000006901454,0.000967271],"genre_scores_gemma":[0.005885133,0.9514987,0.0004953026,0.03712201,0.004337843,0.0001430696,0.00006464641,0.00001824022,0.0004350924],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.04766485,"threshold_uncertainty_score":0.2520788,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4244203368","doi":"10.1177/109821400302400106","title":"A Comparison of Three Retrospective Self-reporting Methods of Measuring Change in Instructional Practice","year":2003,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Behavioral and Psychological Studies","field":"Psychology","cited_by":221,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Satisficing; Psychology; Recall; Retrospective cohort study; Attitude change; Cognition; Behavior change; Intervention (counseling); Social psychology; Applied psychology; Cognitive psychology; Medicine; Computer science","authors":[{"name":"Tony C. M. Lam","is_ca":true},{"name":"Priscilla Bengo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.5729309612507211,"gpt":0.5646686413581473,"spread":0.008262319892573755,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02802865,0.001277545,0.001153638,0.003189627,0.000501429,0.00131265,0.001244841,0.001056902,0.001731878],"category_scores_gemma":[0.08032431,0.001057114,0.001521587,0.002469829,0.0009751674,0.002034587,0.001660536,0.001183681,0.0005177194],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007115036,"about_ca_system_score_gemma":0.0008452845,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001217355,"about_ca_topic_score_gemma":0.003080636,"domain_scores_codex":[0.9545159,0.02703703,0.005561003,0.002593681,0.00974712,0.0005453029],"domain_scores_gemma":[0.8564786,0.08354625,0.02734023,0.01391352,0.01690233,0.001819117],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01863281,0.004502828,0.5850407,0.002154378,0.003941045,0.00007125129,0.006409683,0.001460249,0.01181587,0.001519483,0.001751398,0.3627003],"study_design_scores_gemma":[0.001417893,0.01847988,0.9590155,0.0003065994,0.001017427,0.0003446185,0.002431995,0.004467707,0.007668966,0.001016444,0.003572537,0.0002604061],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9136469,0.001846849,0.06516715,0.0002377452,0.0003783423,0.006758137,0.001518969,0.0004456336,0.01000017],"genre_scores_gemma":[0.83996,0.001662574,0.1271869,0.0002822531,0.0002616239,0.02364661,0.002512571,0.0001418473,0.004345561],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9719713,"threshold_uncertainty_score":0.1482314,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1965824013","doi":"10.1177/1098214013478142","title":"The Case for Participatory Evaluation in an Era of Accountability","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":165,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Accountability; Technocracy; Citizen journalism; Participatory evaluation; Context (archaeology); Participatory GIS; Public sector; Sociology; Politics; Public administration; Public relations; Government (linguistics); Social accounting; Political science; Business; Accounting; Law","authors":[{"name":"Jill Anne Chouinard","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3658990701241563,"gpt":0.5819220676204166,"spread":0.2160229974962604,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3841093,0.001580279,0.002539853,0.005424679,0.02037202,0.03746073,0.005565479,0.02068214,0.00607849],"category_scores_gemma":[0.2871268,0.001473973,0.002193286,0.004498021,0.1477503,0.05836201,0.02999454,0.03143195,0.001131737],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02487101,"about_ca_system_score_gemma":0.0538993,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007304897,"about_ca_topic_score_gemma":0.005667371,"domain_scores_codex":[0.4363875,0.48815,0.007526622,0.01924803,0.03947902,0.009208781],"domain_scores_gemma":[0.5521003,0.3530542,0.01426022,0.04126886,0.02840918,0.01090716],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002612646,0.00002672882,0.0003052581,0.0001858557,0.00002505957,0.0001079094,0.01126229,0.0003112229,0.00005548607,0.9717579,0.005535211,0.01040084],"study_design_scores_gemma":[0.00003957012,0.00004198078,0.0002275569,0.0009409554,0.00001286087,0.0001209584,0.005961976,0.0006141496,0.0001331155,0.8933764,0.09848613,0.00004430714],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.006311462,0.00987296,0.1439248,0.630372,0.003483114,0.0007440262,0.00006235111,0.0001874456,0.2050418],"genre_scores_gemma":[0.7358721,0.006076299,0.1403603,0.08818028,0.003883894,0.004163501,0.00006061887,0.0004070577,0.02099602],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3841093,"threshold_uncertainty_score":0.7595029,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2169058265","doi":"10.1177/1098214009354774","title":"Bibliometrics as a Performance Measurement Tool for Research Evaluation: The Case of Research Funded by the National Cancer Institute of Canada","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":157,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Université du Québec à Montréal; Canadian Partnership Against Cancer; Canadian Cancer Society","funders":"National Cancer Institute","keywords":"Bibliometrics; Documentation; Management science; Medical education; Medicine; Library science; Computer science; Engineering","authors":[{"name":"David Campbell","is_ca":false},{"name":"Michelle Picard-Aitken","is_ca":false},{"name":"Grégoire Côté","is_ca":false},{"name":"Julie Caruso","is_ca":false},{"name":"Rodolfo Valentim","is_ca":true},{"name":"Stuart Edmonds","is_ca":true},{"name":"Gregory Thomas Williams","is_ca":true},{"name":"Benoît Macaluso","is_ca":true},{"name":"Jean-Pierre Robitaille","is_ca":true},{"name":"Nicolas Bastien","is_ca":true},{"name":"Marie-Claude Laframboise","is_ca":true},{"name":"Louis-Michel Lebeau","is_ca":true},{"name":"Philippe Mirabel","is_ca":true},{"name":"Vincent Larivière","is_ca":true},{"name":"Éric Archambault","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8412037386951862,"gpt":0.6852424986880619,"spread":0.1559612400071243,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.1013011,0.001009329,0.001686211,0.0227184,0.01286031,0.01932848,0.00289484,0.002698591,0.001260833],"category_scores_gemma":[0.2066818,0.000571372,0.0009884005,0.06673732,0.01086778,0.007500842,0.006294741,0.002863382,0.0002996283],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.1002914,"about_ca_system_score_gemma":0.1212239,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.6881527,"about_ca_topic_score_gemma":0.7024083,"domain_scores_codex":[0.8053959,0.1143318,0.007320051,0.00263669,0.06422172,0.006093888],"domain_scores_gemma":[0.7692756,0.1481205,0.01035147,0.006832714,0.06129992,0.004119823],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.0003033713,0.0002791192,0.06473158,0.003607124,0.0003252834,0.002933453,0.06032768,0.01892596,0.001262194,0.4977456,0.02436499,0.3251937],"study_design_scores_gemma":[0.0002921967,0.0006480977,0.1408474,0.007246402,0.0007299358,0.002112721,0.1480428,0.1127572,0.006216255,0.208373,0.3718486,0.0008854733],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.2862603,0.01945312,0.1436784,0.09588692,0.0007570921,0.003099218,0.001905036,0.0009810205,0.447979],"genre_scores_gemma":[0.8915482,0.004842841,0.09794287,0.0006138059,0.0001308593,0.0005940086,0.0002954428,0.0001154715,0.003916567],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9772816,"threshold_uncertainty_score":0.7276683,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2147607381","doi":"10.1177/1098214007309280","title":"The Evaluation of Large Research Initiatives","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":144,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Program evaluation; Agency (philosophy); Government (linguistics); Work (physics); Political science; Management science; Public relations; Public administration; Sociology; Engineering; Social science","authors":[{"name":"William M. K. Trochim","is_ca":false},{"name":"Stephen E. Marcus","is_ca":false},{"name":"Louise C. Mâsse","is_ca":true},{"name":"Richard P. Moser","is_ca":false},{"name":"Patrick C. Weld","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8816684527440397,"gpt":0.727468180018336,"spread":0.1542002727257037,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3613473,0.001785546,0.002429208,0.02055239,0.005477903,0.01671479,0.003627355,0.002248733,0.00552063],"category_scores_gemma":[0.5048438,0.0007501174,0.00111175,0.02251466,0.005659251,0.01208037,0.01310085,0.002091287,0.001008721],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02276064,"about_ca_system_score_gemma":0.03949312,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005389399,"about_ca_topic_score_gemma":0.004339722,"domain_scores_codex":[0.4545095,0.4034488,0.02540201,0.01255778,0.09817842,0.005903495],"domain_scores_gemma":[0.2543141,0.5156031,0.04259395,0.04531682,0.1319751,0.010197],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002146187,0.002632642,0.06772961,0.00753504,0.00152378,0.0004025057,0.02768094,0.004817496,0.003534869,0.05350277,0.01617216,0.8123221],"study_design_scores_gemma":[0.00408297,0.02306516,0.3304878,0.01458847,0.003534067,0.0006938296,0.1513661,0.02874144,0.0262826,0.1216868,0.2944879,0.0009828226],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7063445,0.01478335,0.05792497,0.01953088,0.001384108,0.0283512,0.002609863,0.00178208,0.167289],"genre_scores_gemma":[0.900063,0.002928243,0.07829303,0.001400703,0.0005232908,0.01166955,0.001299659,0.0002109764,0.003611446],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9794476,"threshold_uncertainty_score":0.7875726,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2120292430","doi":"10.1177/1098214012440030","title":"A New Realistic Evaluation Analysis Method","year":2012,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":144,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Public Health Ontario; University of Toronto","funders":"Health Research Board","keywords":"Coding (social sciences); Computer science; Qualitative research; Qualitative analysis; Narrative; Management science; Context (archaeology); Evaluation methods; Data science; Psychology; Sociology; Social science; Engineering","authors":[{"name":"Suzanne F. Jackson","is_ca":true},{"name":"Gillian Kolla","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2974231815816114,"gpt":0.6135208376358036,"spread":0.3160976560541922,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06957363,0.00149236,0.001222669,0.008029932,0.003907089,0.01031143,0.002786169,0.001781475,0.02319369],"category_scores_gemma":[0.1507358,0.001019224,0.001508378,0.006802655,0.004171227,0.0080764,0.00729907,0.00341167,0.003351172],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008372096,"about_ca_system_score_gemma":0.01042158,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002799827,"about_ca_topic_score_gemma":0.003397235,"domain_scores_codex":[0.8707089,0.09604144,0.008519234,0.009061929,0.01434924,0.001319286],"domain_scores_gemma":[0.8829531,0.06641935,0.004977448,0.01471382,0.0296535,0.001282723],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003847771,0.0002514973,0.002914126,0.001655118,0.0001459418,0.0002103673,0.0267929,0.004322817,0.003387316,0.4314783,0.0157929,0.5126641],"study_design_scores_gemma":[0.0004987163,0.0008429159,0.005444862,0.002442941,0.0002198995,0.000828204,0.02484058,0.07299381,0.009699829,0.3805006,0.5013437,0.0003438703],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00711156,0.0002032155,0.9453391,0.000963445,0.0003117517,0.005982948,0.0006328205,0.0005342365,0.03892093],"genre_scores_gemma":[0.06165802,0.0001707607,0.9128368,0.0002512664,0.00006331362,0.01680908,0.0004433508,0.0003225198,0.007444848],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.06957363,"threshold_uncertainty_score":0.3679449,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2800401326","doi":"10.1177/1098214018765698","title":"Outcomes and Impacts of Development Interventions","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":120,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Royal Roads University","funders":"International Fund for Agricultural Development; Consortium of International Agricultural Research Centers; Centre for International Forestry Research; Canada Research Chairs; United Nations Development Programme","keywords":"CLARITY; Psychological intervention; Consistency (knowledge bases); Outcome (game theory); Accountability; Confusion; Management science; Psychology; Computer science; Sociology; Political science; Economics","authors":[{"name":"B. Belcher","is_ca":true},{"name":"Markus Palenberg","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2737634216383244,"gpt":0.5787112819025034,"spread":0.304947860264179,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05953061,0.001586378,0.001187356,0.009203728,0.002047702,0.01125762,0.001523222,0.002451308,0.009539082],"category_scores_gemma":[0.1630226,0.0004374599,0.001886572,0.005841544,0.01087653,0.009573197,0.008454396,0.003445449,0.0006577182],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01037113,"about_ca_system_score_gemma":0.01021324,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002679595,"about_ca_topic_score_gemma":0.002092816,"domain_scores_codex":[0.8732661,0.09049764,0.006798836,0.004151207,0.02218847,0.003097723],"domain_scores_gemma":[0.8833846,0.08496131,0.01302681,0.004987116,0.01189518,0.001745],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.0005076659,0.0005079065,0.02470242,0.006027138,0.0007413821,0.0001765454,0.007871564,0.007817643,0.0005657935,0.7739328,0.006635491,0.1705136],"study_design_scores_gemma":[0.0002985804,0.001696319,0.07662067,0.01293994,0.001529463,0.0003840656,0.01583317,0.007518196,0.006786332,0.7692204,0.1069365,0.0002364523],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1559768,0.02988899,0.1602249,0.04048444,0.002599476,0.004017968,0.004878709,0.0004923518,0.6014364],"genre_scores_gemma":[0.9541497,0.007212939,0.02836558,0.001574936,0.0002959243,0.00309228,0.0007414857,0.0001169194,0.004450218],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.05953061,"threshold_uncertainty_score":0.3148317,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1975564755","doi":"10.1177/1098214013477235","title":"Understanding Dimensions of Organizational Evaluation Capacity","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":118,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Ottawa; École Nationale d'Administration Publique","funders":"Australian Government","keywords":"Dimension (graph theory); Capacity building; Government (linguistics); Knowledge management; Organization development; Business; Organizational learning; Process management; Computer science; Economics; Economic growth","authors":[{"name":"Isabelle Bourgeois","is_ca":true},{"name":"J. Bradley Cousins","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.5129125946126178,"gpt":0.4844682783398096,"spread":0.02844431627280825,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01176049,0.0003310084,0.000251636,0.005433454,0.003147458,0.007667558,0.001051915,0.001078287,0.001955654],"category_scores_gemma":[0.03238937,0.0002676264,0.0003565613,0.002898118,0.01383663,0.008487763,0.00617096,0.00137838,0.0001039918],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01782574,"about_ca_system_score_gemma":0.01767699,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.07625224,"about_ca_topic_score_gemma":0.05396079,"domain_scores_codex":[0.9903617,0.004234136,0.0005569273,0.0005160632,0.002220267,0.002111042],"domain_scores_gemma":[0.9601122,0.02281154,0.00385824,0.002195387,0.007701904,0.003320779],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008537759,0.0001661269,0.1693332,0.0004228289,0.00006819455,0.0002845415,0.1581808,0.005302267,0.001541962,0.5604243,0.002524302,0.1016661],"study_design_scores_gemma":[0.00002826247,0.0001241918,0.2839769,0.001123231,0.00005599604,0.0003299126,0.3372654,0.01309324,0.001657591,0.2994421,0.06274616,0.000157015],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8318784,0.001303942,0.01888683,0.007468014,0.00002771453,0.0001521189,0.0001293365,0.00005393857,0.1400998],"genre_scores_gemma":[0.9981254,0.0001265398,0.001340364,0.00005144017,0.000003250886,0.00002515092,0.00002500413,0.000002824963,0.0003001361],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07625224,"threshold_uncertainty_score":0.1516168,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2102960304","doi":"10.1177/1098214012464037","title":"Arguments for a Common Set of Principles for Collaborative Inquiry in Evaluation","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":113,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University; Carleton University; University of Ottawa","funders":"","keywords":"Logic model; Set (abstract data type); Context (archaeology); Field (mathematics); Stakeholder; Program evaluation; Management science; Engineering ethics; Sociology; Computer science; Epistemology; Knowledge management; Public relations; Political science; Social science; Public administration; Engineering","authors":[{"name":"J. Bradley Cousins","is_ca":true},{"name":"Elizabeth Whitmore","is_ca":true},{"name":"Lyn M. Shulha","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3925397717083799,"gpt":0.5662179449091042,"spread":0.1736781732007242,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3024054,0.002648088,0.005341647,0.009153823,0.01660413,0.03869847,0.0134351,0.02960807,0.007384952],"category_scores_gemma":[0.2380435,0.002645942,0.006060009,0.007560718,0.1416953,0.05481188,0.028115,0.03547414,0.003185324],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01995035,"about_ca_system_score_gemma":0.02732571,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004482527,"about_ca_topic_score_gemma":0.002870991,"domain_scores_codex":[0.653598,0.2366872,0.02336997,0.01930781,0.06143515,0.005601895],"domain_scores_gemma":[0.6968179,0.2143449,0.01047326,0.04115148,0.03127029,0.005942136],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000009428871,0.00002103519,0.00005940397,0.0001038433,0.00001253874,0.00001996107,0.002033072,0.0002359699,0.00002057937,0.9940954,0.0009130957,0.002475625],"study_design_scores_gemma":[0.00005020612,0.00002049205,0.00005358454,0.0002900191,0.00000860828,0.00004477476,0.0007774134,0.000804695,0.00008450032,0.98588,0.01196606,0.00001973255],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002909262,0.002639645,0.7886564,0.1185515,0.0009607204,0.001240571,0.00009607198,0.0003470665,0.08459885],"genre_scores_gemma":[0.2712866,0.002130514,0.6895554,0.02131657,0.001014092,0.008652979,0.0001595166,0.0003917677,0.005492499],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3024054,"threshold_uncertainty_score":0.8602583,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2253092986","doi":"10.1177/1098214015615230","title":"Introducing Evidence-Based Principles to Guide Collaborative Approaches to Evaluation","year":2015,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa; Carleton University; Queen's University","funders":"","keywords":"Variety (cybernetics); Set (abstract data type); Context (archaeology); Computer science; Management science; Knowledge management; Psychology; Data science; Engineering; Artificial intelligence","authors":[{"name":"Lyn M. Shulha","is_ca":true},{"name":"Elizabeth Whitmore","is_ca":true},{"name":"J. Bradley Cousins","is_ca":true},{"name":"Nathalie Gilbert","is_ca":true},{"name":"Hind Al Hudib","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.7348226152021603,"gpt":0.5424577550404095,"spread":0.1923648601617508,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5746235,0.003932948,0.005515346,0.02863439,0.01159662,0.03210038,0.01628013,0.01781291,0.002347921],"category_scores_gemma":[0.5653546,0.003991971,0.005193956,0.01107727,0.04504406,0.02587137,0.0253346,0.0352226,0.001771353],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02609422,"about_ca_system_score_gemma":0.07054382,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008942683,"about_ca_topic_score_gemma":0.01406893,"domain_scores_codex":[0.3337961,0.5167508,0.07010379,0.01164808,0.06392045,0.003780874],"domain_scores_gemma":[0.2769093,0.5771062,0.02489837,0.0300023,0.08548024,0.005603571],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000144385,0.0005861977,0.004545994,0.01498578,0.000823706,0.001113038,0.04207227,0.009353304,0.0008644863,0.603222,0.02698696,0.295302],"study_design_scores_gemma":[0.0002742091,0.0003344859,0.00174794,0.0389306,0.0003067542,0.0006584228,0.01460076,0.007764653,0.001661731,0.7709996,0.1623579,0.0003629504],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003016136,0.0138688,0.8404174,0.1102368,0.002277929,0.009031439,0.0001746539,0.0004125329,0.02056438],"genre_scores_gemma":[0.02938825,0.003163582,0.9569552,0.004267695,0.0002445022,0.005324292,0.00008387613,0.00005862253,0.0005140285],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4253765,"threshold_uncertainty_score":0.524565,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1989835141","doi":"10.1177/1098214009349865","title":"A Review and Synthesis of Current Research on Cross-Cultural Evaluation","year":2009,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":91,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Construct (python library); Context (archaeology); Indigenous; Sociology; Empirical research; Management science; Knowledge management; Engineering ethics; Epistemology; Computer science; Ecology","authors":[{"name":"Jill Anne Chouinard","is_ca":true},{"name":"J. Bradley Cousins","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.7361957222054291,"gpt":0.7346471200265722,"spread":0.001548602178856884,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01879073,0.0009637714,0.003291463,0.01330095,0.0011179,0.005517026,0.00156006,0.002156009,0.006884042],"category_scores_gemma":[0.05930339,0.0005939085,0.001193103,0.02032941,0.001774354,0.006183416,0.001922584,0.001443491,0.001542376],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004478382,"about_ca_system_score_gemma":0.01179444,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004271362,"about_ca_topic_score_gemma":0.009102548,"domain_scores_codex":[0.9868796,0.006160551,0.002669234,0.0007319265,0.003268118,0.0002905985],"domain_scores_gemma":[0.9126095,0.07110912,0.003833268,0.001738885,0.01003186,0.0006773634],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.00007192457,0.00009402713,0.00070553,0.06813958,0.0001957389,0.0001532295,0.001493093,0.000387296,0.000370915,0.005670491,0.01280526,0.9099128],"study_design_scores_gemma":[0.00004042992,0.000365259,0.009699894,0.3098919,0.001048197,0.00129495,0.00651931,0.0005175025,0.001058655,0.01321838,0.6562402,0.0001053173],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0009562321,0.9896212,0.002019227,0.001966982,0.0003983581,0.00006842449,0.00005513178,0.00001986231,0.00489454],"genre_scores_gemma":[0.008049755,0.9878169,0.002691254,0.0005531487,0.0002057028,0.0001216722,0.00007281925,0.00001187608,0.0004769341],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01879073,"threshold_uncertainty_score":0.09937608,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2000821194","doi":"10.1177/1098214009340580","title":"Toward Accurate Measurement of Participation","year":2009,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":86,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Conceptualization; Operationalization; Citizen journalism; Stakeholder; Field (mathematics); Participatory evaluation; Psychology; Sociology; Management science; Computer science; Epistemology; Political science; Public relations; Social science; Artificial intelligence; Engineering; Mathematics","authors":[{"name":"Pierre‐Marc Daigneault","is_ca":true},{"name":"Steve Jacob","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4707113104953488,"gpt":0.5611236037354239,"spread":0.0904122932400751,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.129045,0.001050197,0.001380554,0.005572424,0.002463244,0.007968769,0.002286744,0.003742525,0.001950204],"category_scores_gemma":[0.2553814,0.0006870954,0.0006551579,0.005905854,0.006546078,0.01691736,0.01324718,0.005529729,0.0008727475],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004102475,"about_ca_system_score_gemma":0.007649344,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002586099,"about_ca_topic_score_gemma":0.002070055,"domain_scores_codex":[0.7992246,0.1395029,0.01081099,0.01179876,0.03593687,0.002725986],"domain_scores_gemma":[0.7364173,0.1390915,0.02466551,0.03202649,0.06439362,0.003405645],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001664041,0.0004270651,0.05150686,0.001732354,0.0001274381,0.00008471746,0.02802654,0.003280602,0.004387979,0.4711637,0.01231708,0.4267793],"study_design_scores_gemma":[0.00008981324,0.0008898553,0.05800097,0.0046851,0.0001365181,0.0003383751,0.02752696,0.0188682,0.01059101,0.7301467,0.1484498,0.0002766705],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06412876,0.002990213,0.8594982,0.01758174,0.0007717481,0.00148996,0.0005701922,0.0005056194,0.05246357],"genre_scores_gemma":[0.4866986,0.00227063,0.4999613,0.00274034,0.0002746374,0.004670847,0.0005224606,0.0001176954,0.002743457],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.129045,"threshold_uncertainty_score":0.682463,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3035271094","doi":"10.1177/1098214019899164","title":"Talking Circles: A Culturally Responsive Evaluation Practice","year":2020,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Stollery Children's Hospital","funders":"","keywords":"Privilege (computing); Indigenous; Invisibility; Sociology; Power (physics); Stakeholder; Culturally appropriate; Power structure; Psychology; Pedagogy; Social psychology; Public relations; Computer science; Ethnography; Political science; Medicine; Computer security; Artificial intelligence","authors":[{"name":"Martha A. Brown","is_ca":false},{"name":"Sherri Di Lallo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2213240597741148,"gpt":0.5437861279564394,"spread":0.3224620681823247,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1800964,0.001232699,0.001068914,0.005214349,0.01412304,0.01782911,0.005078589,0.00442523,0.009659098],"category_scores_gemma":[0.1616835,0.001097373,0.001172671,0.002745451,0.02196689,0.0148943,0.02143405,0.006730668,0.003297844],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009125361,"about_ca_system_score_gemma":0.02456633,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002239535,"about_ca_topic_score_gemma":0.005314847,"domain_scores_codex":[0.6821614,0.2902946,0.004625014,0.00588567,0.01415479,0.002878602],"domain_scores_gemma":[0.8242204,0.1102158,0.006964464,0.01845596,0.02567669,0.01446669],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0002914284,0.001045832,0.00346065,0.001514735,0.000132406,0.001277257,0.3945871,0.001624755,0.003281649,0.150869,0.05233152,0.3895836],"study_design_scores_gemma":[0.0002275622,0.0008695464,0.002005951,0.004351746,0.0001440245,0.001673736,0.2779998,0.004314963,0.006430396,0.1779484,0.5236403,0.0003936325],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07662977,0.003420535,0.5459233,0.08687682,0.002406705,0.007118287,0.0001371365,0.003579787,0.2739076],"genre_scores_gemma":[0.5257627,0.002403901,0.4270812,0.01218535,0.0005464227,0.006373995,0.000101794,0.001435321,0.02410936],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1800964,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[{"model":"gpt","categories":[],"domain":null,"study_design":"qualitative","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"},{"model":"grok","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"},{"model":"opus","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"split"},{"id":"W2149719915","doi":"10.1177/1098214008325023","title":"An Assessment of the Theoretical Underpinnings of Practical Participatory Evaluation","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":63,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"","keywords":"Context (archaeology); Process (computing); Citizen journalism; Key (lock); Management science; Participatory action research; Knowledge management; Empirical research; Action (physics); Computer science; Epistemology; Psychology; Sociology; Engineering","authors":[{"name":"Pernelle Smits","is_ca":true},{"name":"François Champagne","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3684489379509288,"gpt":0.6248471033074683,"spread":0.2563981653565395,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1114058,0.001079733,0.001234708,0.005749513,0.004565151,0.01159483,0.00305133,0.004607562,0.00618435],"category_scores_gemma":[0.1568365,0.001003374,0.001125159,0.004205174,0.03136447,0.01567058,0.008094897,0.005019501,0.000549591],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009636865,"about_ca_system_score_gemma":0.01025011,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002196924,"about_ca_topic_score_gemma":0.001841043,"domain_scores_codex":[0.9201732,0.06050497,0.002390088,0.002347904,0.01305143,0.001532433],"domain_scores_gemma":[0.7557033,0.2180339,0.006109952,0.009140763,0.00985541,0.001156658],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002414764,0.00004398696,0.0006900472,0.0004410531,0.00001063044,0.00005229392,0.002433086,0.001739941,0.00006450642,0.9654332,0.0003606168,0.02870655],"study_design_scores_gemma":[0.00003694344,0.0001059586,0.001163969,0.00207822,0.00001839342,0.000182316,0.004145402,0.009500822,0.0003360428,0.9570967,0.02529908,0.00003609853],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03939401,0.02084627,0.6056702,0.069545,0.0005850194,0.002180231,0.0001131427,0.0001750416,0.2614912],"genre_scores_gemma":[0.8407024,0.008396662,0.1425539,0.002192306,0.0002854294,0.002949103,0.00008831076,0.00006198672,0.002769724],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8885942,"threshold_uncertainty_score":0.5891774,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2073141549","doi":"10.1177/1098214007312630","title":"Using Self-Assessments to Detect Workshop Success","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Communication in Education and Healthcare","field":"Psychology","cited_by":49,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Alberta Health Services; University of British Columbia; University of Calgary; University of Saskatchewan","funders":"","keywords":"Psychology; Reliability (semiconductor); Applied psychology; Program evaluation; Scale (ratio); Gold standard (test); Medical education; Medicine; Statistics","authors":[{"name":"Marcel D’Eon","is_ca":true},{"name":"Leslie Sadownik","is_ca":true},{"name":"Alexandra Harrison","is_ca":true},{"name":"Jill Nation","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2077316628935172,"gpt":0.5490251839928058,"spread":0.3412935210992886,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01243692,0.0006488789,0.0005104873,0.002369971,0.0003709705,0.0009779796,0.0006853293,0.0004906459,0.00169484],"category_scores_gemma":[0.03933124,0.0002533481,0.0005235733,0.0008335364,0.0003641522,0.001006723,0.0009689453,0.0007218465,0.001211422],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002430842,"about_ca_system_score_gemma":0.0004481417,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005822537,"about_ca_topic_score_gemma":0.00105712,"domain_scores_codex":[0.9891512,0.004466258,0.001648005,0.0008204754,0.003627996,0.0002859251],"domain_scores_gemma":[0.9420525,0.02360508,0.01301971,0.003894336,0.01571169,0.001716791],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009081695,0.001385695,0.8139325,0.0004976636,0.000277138,0.000144865,0.003640969,0.0008537095,0.00731396,0.0002635134,0.002812169,0.1679697],"study_design_scores_gemma":[0.00009216472,0.004309395,0.9711244,0.0001777281,0.00010121,0.0003660049,0.002599628,0.005837168,0.01020288,0.000503251,0.004589289,0.00009699782],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9656932,0.000363255,0.0216183,0.0001561879,0.00009625824,0.001319117,0.001050705,0.000398112,0.009304901],"genre_scores_gemma":[0.9675133,0.0003315375,0.02511181,0.0001079842,0.00007575933,0.001927357,0.001530332,0.00005711971,0.003344715],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01243692,"threshold_uncertainty_score":0.06577349,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4313250641","doi":"10.1177/10982140221106991","title":"Laying a Solid Foundation for the Next Generation of Evaluation Capacity Building: Findings from an Integrative Review","year":2022,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":46,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University; University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Scholarship; Foundation (evidence); Capacity building; Engineering ethics; Political science; Management science; Sociology; Public relations; Economics; Engineering; Law","authors":[{"name":"Isabelle Bourgeois","is_ca":true},{"name":"Sebastian Lemire","is_ca":false},{"name":"Leslie A. Fierro","is_ca":true},{"name":"Ann Marie Castleman","is_ca":false},{"name":"Minji Cho","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.5281446215139854,"gpt":0.552257622583783,"spread":0.02411300106979752,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04974335,0.0006450059,0.002389895,0.01338162,0.001626432,0.0088867,0.001497236,0.001947407,0.002536958],"category_scores_gemma":[0.1270067,0.0006312651,0.001927978,0.01662015,0.002782702,0.01144428,0.004515173,0.002815939,0.0003331511],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005010719,"about_ca_system_score_gemma":0.02949065,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003850973,"about_ca_topic_score_gemma":0.01064767,"domain_scores_codex":[0.9819375,0.009753063,0.003560815,0.0008757217,0.003340823,0.0005320158],"domain_scores_gemma":[0.7148072,0.2480959,0.01314967,0.003389573,0.01881794,0.001739705],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0001426743,0.0001746376,0.005737985,0.2220565,0.001669685,0.0003281016,0.01472504,0.001001466,0.0007297506,0.0421762,0.01191716,0.6993408],"study_design_scores_gemma":[0.00005888685,0.0002745704,0.01657292,0.6067102,0.005644298,0.0007676105,0.03432775,0.001045154,0.001040695,0.03230424,0.3010931,0.0001605572],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.006280578,0.9670913,0.004769043,0.01560075,0.0004985606,0.0002821476,0.0001418705,0.00002144469,0.005314355],"genre_scores_gemma":[0.06987146,0.9132697,0.01147981,0.003787894,0.0004968773,0.0005810567,0.0001709838,0.00002549614,0.0003167644],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9502566,"threshold_uncertainty_score":0.2630711,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2318178033","doi":"10.1177/1098214013503698","title":"Managing Tensions Between Evaluation and Research","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":44,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Université de Sherbrooke","funders":"","keywords":"Temporality; Psychological intervention; Software deployment; Process (computing); Management science; Psychology; Sociology; Engineering ethics; Computer science; Epistemology","authors":[{"name":"Lynda Rey","is_ca":true},{"name":"Marie‐Claude Tremblay","is_ca":true},{"name":"Astrid Brousselle","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4777265607402056,"gpt":0.6157514516411698,"spread":0.1380248909009642,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7301786,0.001874394,0.006284974,0.01587693,0.01548614,0.04943062,0.007474562,0.01323625,0.003959014],"category_scores_gemma":[0.6980662,0.003130226,0.001673081,0.009548042,0.1006884,0.05113991,0.04177064,0.02155967,0.00113256],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.03255483,"about_ca_system_score_gemma":0.05285377,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003163849,"about_ca_topic_score_gemma":0.003088337,"domain_scores_codex":[0.1488277,0.7442114,0.02984769,0.01487995,0.05616722,0.006065983],"domain_scores_gemma":[0.1042697,0.791567,0.01907684,0.03203415,0.04314361,0.009908788],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0005061971,0.0003378964,0.006145997,0.004396831,0.0003454243,0.0009904495,0.1585471,0.001214275,0.0008063689,0.5191184,0.01231226,0.2952788],"study_design_scores_gemma":[0.0004672955,0.0008581026,0.003915611,0.01289076,0.0001843925,0.001429369,0.1032585,0.004412172,0.001216245,0.7627423,0.1082053,0.000420031],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.04276428,0.04784472,0.275167,0.554916,0.005339212,0.002194999,0.00006547049,0.0005686366,0.07113967],"genre_scores_gemma":[0.7766484,0.01271348,0.1421269,0.05121756,0.004600528,0.006546795,0.00005497956,0.0004960606,0.005595319],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2698214,"threshold_uncertainty_score":0.3327379,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2884500380","doi":"10.1177/1098214018778809","title":"The Need for Analysts in Social Impact Measurement","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":40,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Psychology; Program evaluation; Management science; Actuarial science; Applied psychology; Business; Political science; Economics; Public administration","authors":[{"name":"Katherine Ruff","is_ca":true},{"name":"Sara Olsen","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3492754177554118,"gpt":0.5984297125613699,"spread":0.2491542948059581,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3016407,0.002085191,0.002832839,0.01460758,0.008830369,0.02508241,0.008349001,0.01812244,0.006904041],"category_scores_gemma":[0.5140064,0.002417527,0.001768144,0.005616866,0.02354062,0.05653181,0.01838304,0.03252623,0.003340876],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01013755,"about_ca_system_score_gemma":0.05349706,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01319212,"about_ca_topic_score_gemma":0.01522988,"domain_scores_codex":[0.7712388,0.1554177,0.01233666,0.01010602,0.04659386,0.004306987],"domain_scores_gemma":[0.2280047,0.5587533,0.02266036,0.0393338,0.1264849,0.02476299],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006470824,0.0009874406,0.02984484,0.002298061,0.0002895838,0.0004785366,0.01967375,0.004037233,0.002831406,0.3698583,0.1394507,0.4296031],"study_design_scores_gemma":[0.0002298537,0.0002240045,0.006729942,0.003516034,0.0001315057,0.0007589989,0.01920074,0.01341732,0.001701891,0.7252929,0.2283875,0.0004093143],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.01323866,0.01188901,0.1494852,0.7838632,0.005095208,0.0003513724,0.000183571,0.00158009,0.03431368],"genre_scores_gemma":[0.4878427,0.007557959,0.377449,0.1084107,0.007427937,0.001622633,0.0003760162,0.00085082,0.008462204],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3016407,"threshold_uncertainty_score":0.8612013,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2319861733","doi":"10.1177/1098214014542100","title":"Insights on Using Developmental Evaluation for Innovating","year":2014,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":38,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Conceptualization; Process (computing); Knowledge management; Process management; Rendering (computer graphics); Computer science; Psychology; Management science; Business; Engineering; Artificial intelligence","authors":[{"name":"Chi Yan Lam","is_ca":true},{"name":"Lyn M. Shulha","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3573519414879998,"gpt":0.5543062644225002,"spread":0.1969543229345004,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1616608,0.001231213,0.001076411,0.008369868,0.005066621,0.0171725,0.00323444,0.00354057,0.004348835],"category_scores_gemma":[0.2225381,0.000773814,0.001081967,0.003842633,0.03223392,0.02818067,0.0129343,0.004043126,0.0004947126],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01410508,"about_ca_system_score_gemma":0.01600442,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004851668,"about_ca_topic_score_gemma":0.004775603,"domain_scores_codex":[0.7423955,0.227111,0.006280934,0.003314413,0.01664665,0.004251513],"domain_scores_gemma":[0.5737016,0.3778665,0.008866041,0.01609004,0.02032818,0.003147609],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001005008,0.0001786807,0.005268728,0.0007192925,0.00002883185,0.00032536,0.03744263,0.001581952,0.0003999371,0.8094504,0.00212687,0.1423767],"study_design_scores_gemma":[0.0001314359,0.0005166459,0.005249611,0.003994573,0.00007348069,0.00111893,0.04566437,0.008933743,0.00421297,0.7700124,0.1598874,0.0002044635],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07395881,0.01056835,0.435897,0.06277229,0.0004243325,0.001682017,0.00009947737,0.0005686982,0.4140291],"genre_scores_gemma":[0.8674265,0.002954636,0.1226356,0.002084006,0.0000916018,0.001016592,0.000039332,0.0001445667,0.003607173],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1616608,"threshold_uncertainty_score":0.8549542,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2014357589","doi":"10.1177/1098214011405311","title":"Legislator Uses of Public Performance Reports","year":2011,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Public Policy and Administration Research","field":"Social Sciences","cited_by":35,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Victoria","funders":"","keywords":"Accountability; Legislator; Government (linguistics); Performance measurement; Dual (grammatical number); Key (lock); Public relations; Business; Public administration; Accounting; Political science; Computer science; Computer security; Marketing; Law; Legislation","authors":[{"name":"James C. McDavid","is_ca":true},{"name":"Irene Huse","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2353065547670905,"gpt":0.4390290283376902,"spread":0.2037224735705997,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03951629,0.0002118771,0.0003623622,0.003598166,0.001385448,0.004399227,0.001156947,0.0005860323,0.001872882],"category_scores_gemma":[0.1762383,0.0003400374,0.0003519677,0.004147693,0.00152767,0.001796668,0.001652239,0.001194082,0.0007520065],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003076752,"about_ca_system_score_gemma":0.004575052,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01704972,"about_ca_topic_score_gemma":0.01546361,"domain_scores_codex":[0.9213498,0.04487934,0.005270212,0.003529714,0.0227018,0.002269114],"domain_scores_gemma":[0.7174219,0.1319987,0.07020061,0.02819922,0.04871816,0.003461439],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002755534,0.000241391,0.6699903,0.00044049,0.0001942282,0.0002537285,0.05266971,0.001525686,0.002409222,0.01283254,0.0143886,0.2447785],"study_design_scores_gemma":[0.00002756161,0.0005873028,0.8613518,0.0005436758,0.0001152178,0.0003421344,0.02799653,0.003662165,0.006310034,0.00263267,0.0962138,0.0002171824],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9307249,0.0007973321,0.006205139,0.005783147,0.0001054995,0.0001555638,0.001734565,0.0004608577,0.05403308],"genre_scores_gemma":[0.996416,0.0002382437,0.0009987619,0.0002193887,0.00003380552,0.00006093483,0.000366943,0.00002533612,0.00164051],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03951629,"threshold_uncertainty_score":0.2089846,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2520519417","doi":"10.1177/1098214016668401","title":"Introducing Reflexivity to Evaluation Practice","year":2016,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":34,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Reflexivity; Action (physics); Psychology; Thematic analysis; Action research; Engineering ethics; Action plan; Competence (human resources); Sociology; Qualitative research; Pedagogy; Social psychology; Social science","authors":[{"name":"Jenna van Draanen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2742311431016408,"gpt":0.5920985262212245,"spread":0.3178673831195837,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6363066,0.002392385,0.004010295,0.01174564,0.01073996,0.03516992,0.008170935,0.01226435,0.005727237],"category_scores_gemma":[0.6249049,0.002773824,0.003329147,0.004971263,0.1384601,0.04205873,0.03220939,0.02306559,0.001526392],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02091165,"about_ca_system_score_gemma":0.04442669,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00243647,"about_ca_topic_score_gemma":0.001892803,"domain_scores_codex":[0.1760871,0.7614061,0.02004773,0.01483796,0.02403522,0.003585878],"domain_scores_gemma":[0.1790014,0.7025886,0.02003249,0.05233655,0.04006831,0.005972773],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002051395,0.0001843021,0.001518221,0.004297161,0.0002661089,0.0004661024,0.2069985,0.001406432,0.0007360814,0.678782,0.008746704,0.09639315],"study_design_scores_gemma":[0.0002774509,0.0002732338,0.0004969805,0.009680139,0.00008800164,0.0003592958,0.04204626,0.002315884,0.001628637,0.8379896,0.1046766,0.0001678459],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01511174,0.0107748,0.7004799,0.185629,0.005149069,0.004476747,0.0000986639,0.0009400742,0.07733988],"genre_scores_gemma":[0.4465027,0.004806163,0.5060679,0.02285674,0.001972299,0.01169689,0.00007680689,0.0006485169,0.005371956],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3636934,"threshold_uncertainty_score":0.4484988,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2032438539","doi":"10.1177/1098214009349792","title":"Exploring the Intervention— Context Interface","year":2009,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":33,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Centre Hospitalier de l’Université de Montréal; Université de Montréal","funders":"","keywords":"Sociotechnical system; Context (archaeology); Adaptation (eye); Workaround; Process (computing); Knowledge management; Computer science; Social network analysis; Psychology; Social media; World Wide Web","authors":[{"name":"Sherri Bisset","is_ca":true},{"name":"Mark Daniel","is_ca":true},{"name":"Louise Potvin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8024925403175549,"gpt":0.7139293653214512,"spread":0.08856317499610367,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01495927,0.0005044288,0.0005610331,0.001179409,0.003175795,0.005801539,0.001471693,0.002217174,0.01247086],"category_scores_gemma":[0.02235223,0.0004189714,0.0005658684,0.001019822,0.005290743,0.006044458,0.007243447,0.001854709,0.000485407],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004509752,"about_ca_system_score_gemma":0.005658432,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00313137,"about_ca_topic_score_gemma":0.003401865,"domain_scores_codex":[0.9768754,0.02022953,0.0002985532,0.0009760637,0.0008126847,0.0008077003],"domain_scores_gemma":[0.9803348,0.01763151,0.0005010969,0.0004023576,0.0005010281,0.0006292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009916791,0.002344217,0.02479144,0.003862918,0.0001824795,0.002130498,0.3081358,0.003714994,0.009706903,0.393081,0.003464719,0.2475933],"study_design_scores_gemma":[0.001302701,0.004054852,0.03910056,0.006916181,0.001007996,0.001603734,0.3784125,0.02042976,0.01398432,0.262689,0.270297,0.0002014441],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5277456,0.002236929,0.1766034,0.02083708,0.000343953,0.004159653,0.0003942384,0.0004954281,0.2671838],"genre_scores_gemma":[0.9540952,0.0003876158,0.04019307,0.0009179674,0.00001585887,0.002079258,0.00005240169,0.00003519754,0.00222349],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01495927,"threshold_uncertainty_score":0.07911319,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2156859842","doi":"10.1177/1098214014535658","title":"The Ethical Tipping Points of Evaluators in Conflict Zones","year":2014,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":32,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"International Development Research Centre","funders":"","keywords":"Affect (linguistics); Section (typography); Ethical issues; Face (sociological concept); Psychology; Engineering ethics; Conflict of interest; Sociology; Social psychology; Public relations; Political science; Law; Social science; Computer science","authors":[{"name":"Colleen Duggan","is_ca":true},{"name":"Kenneth Bush","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1709391473582322,"gpt":0.5305063413621447,"spread":0.3595671940039125,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.211502,0.0009122252,0.001324825,0.00341084,0.01974556,0.02995326,0.003719114,0.008961262,0.003517037],"category_scores_gemma":[0.4363956,0.001342861,0.001161612,0.002049129,0.06641145,0.02330599,0.02270532,0.01534582,0.0009654629],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01105745,"about_ca_system_score_gemma":0.0117102,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002231163,"about_ca_topic_score_gemma":0.00279901,"domain_scores_codex":[0.5120551,0.4341094,0.01058659,0.007723509,0.0245439,0.01098143],"domain_scores_gemma":[0.5772742,0.3266524,0.03125395,0.01733588,0.03298106,0.01450251],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0002713741,0.0001270105,0.01140265,0.0005372809,0.0001098995,0.00123081,0.6273545,0.001075916,0.001323991,0.2971383,0.009341587,0.05008663],"study_design_scores_gemma":[0.00009169229,0.0001875981,0.00707584,0.002242244,0.00006014991,0.00126571,0.4669904,0.002352132,0.002909597,0.4288045,0.08770216,0.0003179334],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4116566,0.006306849,0.1695249,0.2357533,0.002036275,0.000883845,0.00006600906,0.000342691,0.1734296],"genre_scores_gemma":[0.9711338,0.0006986902,0.01483471,0.009725428,0.0002140143,0.0004170503,0.00001360006,0.0001250421,0.002837609],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.211502,"threshold_uncertainty_score":0.9723585,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2034409422","doi":"10.1177/1098214008316655","title":"Cross-Disciplinarization: A New Talisman for Evaluation?","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":32,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Cross disciplinary; Discipline; Field (mathematics); Engineering ethics; Face (sociological concept); Management science; Sociology; Interdisciplinarity; Computer science; Data science; Social science; Engineering","authors":[{"name":"Steve Jacob","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2784832306773482,"gpt":0.5746443419319662,"spread":0.296161111254618,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3271042,0.001755195,0.004173931,0.009414955,0.0140621,0.04709668,0.006636934,0.01527587,0.004600798],"category_scores_gemma":[0.2859482,0.001071125,0.002177661,0.009106935,0.1333352,0.07546074,0.03263156,0.03000859,0.0009263376],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02986205,"about_ca_system_score_gemma":0.03743245,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005559109,"about_ca_topic_score_gemma":0.004887207,"domain_scores_codex":[0.5601331,0.388498,0.01077412,0.008191075,0.02789446,0.004509308],"domain_scores_gemma":[0.5856812,0.3355227,0.0089923,0.03403752,0.02771072,0.008055558],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00005893689,0.00006824936,0.0005406922,0.000556172,0.00006417742,0.00008577467,0.01876247,0.0004504121,0.00005919618,0.9022142,0.01334412,0.0637955],"study_design_scores_gemma":[0.00004456188,0.00007636595,0.000268363,0.002675295,0.00003369644,0.0001574807,0.01585641,0.001122381,0.0002070387,0.8963106,0.08319224,0.00005546259],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.004114204,0.05598763,0.143845,0.7184181,0.005956477,0.0005496954,0.00003103897,0.0002318998,0.07086591],"genre_scores_gemma":[0.5897056,0.04191535,0.1941486,0.1469611,0.008940903,0.004406097,0.00007467626,0.0006638358,0.01318379],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6728958,"threshold_uncertainty_score":0.8298003,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4238820305","doi":"10.1177/109821400402500311","title":"Commentary: Minimizing Evaluation Misuse as Principled Practice","year":2004,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":30,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Psychology; Management science; Computer science; Sociology; Economics","authors":[{"name":"J. Bradley Cousins","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.191202468548675,"gpt":0.5583508310794473,"spread":0.3671483625307723,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03790354,0.002234606,0.003515667,0.003667878,0.01276333,0.01133105,0.0126226,0.1853193,0.01013695],"category_scores_gemma":[0.2939438,0.002473647,0.003623226,0.004974994,0.02024094,0.01017099,0.007591005,0.1492979,0.007828983],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02098926,"about_ca_system_score_gemma":0.03259691,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0318085,"about_ca_topic_score_gemma":0.03968847,"domain_scores_codex":[0.9461485,0.01738978,0.006801644,0.007658439,0.01671663,0.005284987],"domain_scores_gemma":[0.6635491,0.2468704,0.01373428,0.00622974,0.05752358,0.01209294],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004509703,0.00001404916,0.00008137106,0.0002219574,0.00004423498,0.0001268492,0.000530478,0.00005834092,0.00008013381,0.006620167,0.9893264,0.002850954],"study_design_scores_gemma":[0.00046891,0.00006976515,0.0009664276,0.004228417,0.0003893784,0.0005769232,0.002093716,0.0007245809,0.000616981,0.03829546,0.9513704,0.0001991676],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0001118353,0.0006893316,0.0001192338,0.967313,0.03087731,0.00001419072,0.00003801144,0.00001433221,0.0008227932],"genre_scores_gemma":[0.00175677,0.0003322605,0.0003294583,0.9667979,0.02959647,0.00005792287,0.00001221417,0.00001952653,0.001097433],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.9620965,"threshold_uncertainty_score":0.2004555,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4308916752","doi":"10.1177/10982140211008978","title":"A Comparison of Fidelity Implementation Frameworks Used in the Field of Early Intervention","year":2022,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Family and Disability Support Research","field":"Psychology","cited_by":30,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Fidelity; Conceptualization; Conceptual framework; Intervention (counseling); Field (mathematics); Management science; Computer science; Quality (philosophy); Knowledge management; Psychology; Sociology; Engineering; Social science","authors":[{"name":"Colombe Lemire","is_ca":true},{"name":"Michel Rousseau","is_ca":true},{"name":"Carmen Dionne","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1101399459900095,"gpt":0.5610765366306308,"spread":0.4509365906406213,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3012662,0.0009396049,0.001924761,0.01302424,0.003959098,0.008835308,0.002830603,0.002377978,0.001828981],"category_scores_gemma":[0.4581412,0.001160765,0.00449123,0.00826134,0.007220123,0.009585715,0.007741642,0.004913808,0.0002032709],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02412668,"about_ca_system_score_gemma":0.02539397,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01219006,"about_ca_topic_score_gemma":0.01362647,"domain_scores_codex":[0.7011972,0.2073304,0.03067744,0.00515989,0.0517824,0.003852737],"domain_scores_gemma":[0.4440159,0.445656,0.02745346,0.01644228,0.06396163,0.002470761],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.001063251,0.0004894634,0.1018808,0.01119501,0.001345636,0.0001760335,0.08471248,0.00195681,0.0007513129,0.1103118,0.00328494,0.6828325],"study_design_scores_gemma":[0.0008435901,0.0105301,0.4940529,0.0822896,0.003892805,0.002680188,0.1608617,0.01898393,0.006874815,0.106036,0.1114365,0.001517902],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4622589,0.08683386,0.3417361,0.0245392,0.00187218,0.01515112,0.001086756,0.0006751155,0.06584686],"genre_scores_gemma":[0.8187833,0.01418708,0.1542214,0.001663043,0.0001652723,0.009269229,0.0005582258,0.000187578,0.0009648705],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6987338,"threshold_uncertainty_score":0.8616632,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3178131540","doi":"10.1177/1098214020940409","title":"Reviewing Health Service and Program Evaluations in Indigenous Contexts: A Systematic Review","year":2021,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Indigenous Health, Education, and Rights","field":"Social Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Centre for Addiction and Mental Health; Public Health Ontario; University of Toronto; St. Michael's Hospital","funders":"Ontario Ministry of Health and Long-Term Care","keywords":"Indigenous; Reciprocity (cultural anthropology); Public relations; Sociology; Service (business); Management science; Psychology; Political science; Business; Social science; Engineering; Marketing; Ecology","authors":[{"name":"Raglan Maddox","is_ca":true},{"name":"Genevieve Blais","is_ca":true},{"name":"Angela Mashford‐Pringle","is_ca":false},{"name":"Renée Monchalin","is_ca":true},{"name":"Michelle Firestone","is_ca":true},{"name":"Carolyn Ziegler","is_ca":true},{"name":"Melody E. Morton Ninomiya","is_ca":true},{"name":"Janet Smylie","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07921618690179015,"gpt":0.4988806023287977,"spread":0.4196644154270076,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06274541,0.002065806,0.008989451,0.02426232,0.001993431,0.004786278,0.003377312,0.002701777,0.003880836],"category_scores_gemma":[0.2347347,0.001906876,0.005148672,0.02077808,0.002440463,0.005109769,0.003877495,0.002322346,0.0004861105],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01085946,"about_ca_system_score_gemma":0.04949241,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01376219,"about_ca_topic_score_gemma":0.04623939,"domain_scores_codex":[0.9321017,0.03164806,0.0202156,0.002426815,0.01266751,0.0009403411],"domain_scores_gemma":[0.7730811,0.1721668,0.02221715,0.004975047,0.02584073,0.001719239],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0001129366,0.00002829616,0.000635713,0.9187303,0.002998941,0.00009731534,0.0009000894,0.0001198572,0.0001569195,0.0005820302,0.003035915,0.0726018],"study_design_scores_gemma":[0.00006778248,0.00008453326,0.001253876,0.9652172,0.01029294,0.00008673058,0.0008031089,0.00005854688,0.0001445891,0.0003320281,0.02163217,0.00002652123],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.001867572,0.9911841,0.001080208,0.001029083,0.0005562024,0.002486479,0.0008057826,0.00003580688,0.0009547367],"genre_scores_gemma":[0.02320752,0.967424,0.003458171,0.001062655,0.0001748519,0.003721595,0.0006537779,0.00002697524,0.0002705087],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9372546,"threshold_uncertainty_score":0.3318334,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3183879452","doi":"10.1177/10982140211007573","title":"Understanding Evaluation Policy and Organizational Capacity for Evaluation: An Interview Study","year":2021,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Accountability; Context (archaeology); Thematic analysis; Organizational performance; Organizational learning; Capacity building; Conceptual framework; Knowledge management; Management science; Sociology; Psychology; Public relations; Political science; Qualitative research; Computer science; Social science; Economics","authors":[{"name":"Hind Al Hudib","is_ca":true},{"name":"J. Bradley Cousins","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.6422351798637937,"gpt":0.5788854004605903,"spread":0.06334977940320341,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07192578,0.000349987,0.0006125908,0.002783895,0.01306042,0.008076048,0.001636312,0.003629299,0.002005186],"category_scores_gemma":[0.09034042,0.0008942203,0.0003083175,0.002884045,0.01097222,0.009771001,0.006285984,0.005802101,0.0002410537],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01234882,"about_ca_system_score_gemma":0.01357585,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004700779,"about_ca_topic_score_gemma":0.005780962,"domain_scores_codex":[0.9435546,0.04726794,0.001784834,0.001327527,0.002261138,0.003804044],"domain_scores_gemma":[0.836495,0.1413202,0.006470966,0.002420934,0.008016272,0.005276552],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00002592125,0.0001453202,0.006112574,0.00007516429,0.0000032613,0.0002982199,0.9777541,0.0001612511,0.0005187251,0.008486974,0.0006047361,0.005813755],"study_design_scores_gemma":[0.000006568277,0.0000477602,0.001393371,0.0001468876,0.000002952186,0.00008193329,0.980648,0.000427126,0.0002717076,0.00141717,0.01554286,0.00001369914],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9635603,0.0007420038,0.009309752,0.01282132,0.00007179799,0.000350905,0.00005261507,0.00002022613,0.01307114],"genre_scores_gemma":[0.9926243,0.0006371503,0.003355223,0.001506804,0.00003791825,0.0003015699,0.00002493326,0.00001404515,0.001498082],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9280742,"threshold_uncertainty_score":0.3803844,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2936658129","doi":"10.1177/1098214019835821","title":"Research and Evaluation With Community-Based Projects: Approaches, Considerations, and Strategies","year":2019,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":26,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"Public Health Agency of Canada","keywords":"Context (archaeology); Management science; Program evaluation; Engineering ethics; Psychology; Process management; Political science; Business; Engineering","authors":[{"name":"Naomi C. Z. Andrews","is_ca":true},{"name":"Debra Pepler","is_ca":true},{"name":"Mary Motz","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8556213530259436,"gpt":0.7057655941049885,"spread":0.149855758920955,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.6562068,0.002948402,0.003294567,0.009485471,0.01763879,0.03077736,0.009855736,0.01276916,0.004826874],"category_scores_gemma":[0.467567,0.003152824,0.00204433,0.007776286,0.0281919,0.03273946,0.0303736,0.01108603,0.00121259],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02407199,"about_ca_system_score_gemma":0.09932799,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006919044,"about_ca_topic_score_gemma":0.0129744,"domain_scores_codex":[0.174156,0.7807003,0.01953742,0.004529105,0.01762966,0.00344746],"domain_scores_gemma":[0.3340166,0.5492615,0.02241931,0.03193484,0.04759129,0.01477639],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007696518,0.004045384,0.01468589,0.02129656,0.0006086809,0.00104105,0.1550503,0.004117558,0.001180416,0.2675629,0.0137772,0.5158644],"study_design_scores_gemma":[0.002184915,0.004332617,0.009628681,0.05782796,0.0006409262,0.001676482,0.2866313,0.01400492,0.00414627,0.4885527,0.1296111,0.0007621759],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03824028,0.02489352,0.5219778,0.2522244,0.001789805,0.1118764,0.0002587144,0.0007255223,0.04801356],"genre_scores_gemma":[0.2200388,0.005999919,0.6008642,0.01029125,0.0003313238,0.1602333,0.00007913318,0.0001530324,0.002008955],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6562068,"threshold_uncertainty_score":0.4239582,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2060998986","doi":"10.1177/1098214006287990","title":"Developing a Stakeholder-Driven Anticipated Timeline of Impact for Evaluation of Social Programs","year":2006,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Timeline; Stakeholder; Stakeholder engagement; Stakeholder analysis; Program evaluation; Process (computing); Process management; Computer science; Management science; Public relations; Business; Political science; Engineering; Geography","authors":[{"name":"Sanjeev Sridharan","is_ca":false},{"name":"Bernadette Campbell","is_ca":true},{"name":"Heidi M. Zinzow","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4786542061575202,"gpt":0.5785712379293496,"spread":0.09991703177182937,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03480466,0.001423932,0.0006762443,0.005121,0.001433458,0.003617323,0.001559629,0.001140809,0.00505093],"category_scores_gemma":[0.07378218,0.0007407424,0.0008319034,0.00324536,0.001048645,0.004985571,0.002228393,0.002219961,0.000969056],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004905745,"about_ca_system_score_gemma":0.006900115,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004698495,"about_ca_topic_score_gemma":0.008434633,"domain_scores_codex":[0.9725398,0.01868354,0.001598029,0.001022178,0.005680006,0.0004763287],"domain_scores_gemma":[0.9228941,0.04043746,0.007051401,0.004365209,0.02399438,0.001257465],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006521189,0.0005331055,0.01418176,0.001996655,0.0001611783,0.0003135066,0.01155484,0.05609674,0.01348288,0.1453764,0.01279384,0.7428569],"study_design_scores_gemma":[0.0007920281,0.004814674,0.03422949,0.003392046,0.000315396,0.0007775935,0.01935754,0.4146462,0.0698073,0.2311999,0.219561,0.001106924],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01898997,0.0001570164,0.9613044,0.0006874569,0.00008330184,0.001915792,0.0005085217,0.0008184149,0.01553499],"genre_scores_gemma":[0.09838486,0.0001075209,0.8974547,0.00008419059,0.00001137923,0.002569891,0.0003115428,0.0001221944,0.0009537378],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9651954,"threshold_uncertainty_score":0.1840668,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2061653962","doi":"10.1177/1098214007307942","title":"Evaluations That Consider the Cost of Educational Programs","year":2007,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"School Choice and Performance","field":"Social Sciences","cited_by":23,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Program evaluation; Cost effectiveness; Set (abstract data type); Cost–benefit analysis; Computer science; Cost estimate; Macro; Program Design Language; Actuarial science; Cost contingency; Management science; Relevant cost; Risk analysis (engineering); Economics; Business","authors":[{"name":"John A. Ross","is_ca":true},{"name":"Khaled Barkaoui","is_ca":true},{"name":"Garth Scott","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.103493342416852,"gpt":0.4664295884873363,"spread":0.3629362460704843,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07197724,0.001717571,0.003151518,0.01160341,0.0007077509,0.006687449,0.001622842,0.002791207,0.009782809],"category_scores_gemma":[0.3914034,0.0006505743,0.004877076,0.01117462,0.001664989,0.007130791,0.002206753,0.003136018,0.0006482194],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00687873,"about_ca_system_score_gemma":0.005385838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003599308,"about_ca_topic_score_gemma":0.004596537,"domain_scores_codex":[0.8281466,0.1241284,0.01242943,0.001994535,0.03175754,0.001543563],"domain_scores_gemma":[0.5234043,0.4207718,0.02287336,0.006427344,0.0243072,0.002216029],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.01223656,0.001899402,0.03439082,0.05075409,0.02755006,0.0003521041,0.0009761,0.04116457,0.0007659221,0.08799132,0.03188718,0.7100319],"study_design_scores_gemma":[0.01127491,0.02204353,0.1248009,0.1031226,0.1018035,0.001799945,0.00440219,0.05584646,0.01108413,0.2141664,0.3482964,0.001359172],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.189585,0.2818342,0.1109257,0.03064813,0.009780872,0.0276605,0.0218306,0.0007646361,0.3269702],"genre_scores_gemma":[0.8864071,0.04200588,0.04991212,0.004180188,0.001244816,0.008365535,0.002456895,0.0001923579,0.005235195],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.07197724,"threshold_uncertainty_score":0.3806566,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3035076565","doi":"10.1177/1098214020908211","title":"The Role of Intuition in Evaluative Judgment and Decision","year":2020,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":20,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Intuition; Psychology; Epistemology; Social psychology; Management science; Cognitive science","authors":[{"name":"Marthe Hurteau","is_ca":true},{"name":"Jeiran Rahmanian","is_ca":true},{"name":"Sylvain Houle","is_ca":true},{"name":"Marie-Pier Marchand","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.117224265564639,"gpt":0.4903323073651327,"spread":0.3731080418004937,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06891401,0.0006702294,0.000775956,0.003933454,0.003491103,0.0101061,0.001312378,0.002237361,0.001628857],"category_scores_gemma":[0.1794416,0.0006475599,0.0008793494,0.001682092,0.02371384,0.009955036,0.005774757,0.003782434,0.0003497686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003865645,"about_ca_system_score_gemma":0.007019525,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002331206,"about_ca_topic_score_gemma":0.002122372,"domain_scores_codex":[0.9071913,0.07056125,0.003431717,0.003731327,0.01199214,0.003092175],"domain_scores_gemma":[0.712106,0.249019,0.01439947,0.009949345,0.01123721,0.003288969],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004013953,0.0003894521,0.03223934,0.001314264,0.0002618623,0.001105988,0.2239335,0.008508652,0.00439742,0.4956857,0.003374209,0.2283882],"study_design_scores_gemma":[0.0000861354,0.00046902,0.0264247,0.001823611,0.00009275157,0.0008841656,0.04097775,0.01513104,0.003413989,0.870823,0.03945461,0.0004193339],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4656816,0.006278141,0.3289952,0.01732504,0.0004458459,0.0005672115,0.00008502105,0.0003204506,0.1803016],"genre_scores_gemma":[0.9621582,0.0007257574,0.03500022,0.0007133237,0.00005724546,0.0001175146,0.00002074138,0.00004285533,0.001164225],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06891401,"threshold_uncertainty_score":0.3644565,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2802781340","doi":"10.1177/1098214018763553","title":"Evaluating Social Innovations","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":18,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Variety (cybernetics); Psychological intervention; Management science; Conceptual framework; Computer science; Knowledge management; Sociology; Engineering ethics; Psychology; Social science; Economics; Engineering","authors":[{"name":"Kate Svensson","is_ca":true},{"name":"Barbara Szijarto","is_ca":true},{"name":"Peter Milley","is_ca":true},{"name":"J. Bradley Cousins","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4567176324915049,"gpt":0.6367359267377268,"spread":0.180018294246222,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1713789,0.001647837,0.002029701,0.006742428,0.001579147,0.006889093,0.001799242,0.001922302,0.006919329],"category_scores_gemma":[0.3296553,0.0004090881,0.001917049,0.004527533,0.003589458,0.005487453,0.003952455,0.001264989,0.0004191913],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00747007,"about_ca_system_score_gemma":0.01028104,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001468389,"about_ca_topic_score_gemma":0.003383617,"domain_scores_codex":[0.7264995,0.2377928,0.01287694,0.003284745,0.01809296,0.001453153],"domain_scores_gemma":[0.5541863,0.3911916,0.01737155,0.007212576,0.02765054,0.002387309],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0015708,0.001737356,0.02219985,0.03847276,0.003730101,0.000193853,0.01293815,0.008654835,0.001579754,0.07033233,0.006294911,0.8322953],"study_design_scores_gemma":[0.004553753,0.04761819,0.08520885,0.1558338,0.01829831,0.0007481138,0.08245593,0.04097948,0.03118846,0.3165852,0.2156029,0.0009270485],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5218914,0.08301069,0.1863139,0.01401749,0.002895264,0.03659081,0.001848713,0.0004633503,0.1529683],"genre_scores_gemma":[0.8224733,0.01465473,0.1493143,0.001099307,0.0002876918,0.009540528,0.0004463416,0.00007339763,0.002110443],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.1713789,"threshold_uncertainty_score":0.9063493,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2093411079","doi":"10.1177/1098214010379038","title":"Evaluating the Science of Discovery in Complex Health Systems","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Interdisciplinary Research and Collaboration","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia; Nutrasource; University of Toronto","funders":"","keywords":"Process (computing); Discipline; Plan (archaeology); Data science; Work (physics); Engineering ethics; Computer science; Scientific discovery; Management science; Logic model; Translational science; Health science; Sociology; Psychology; Engineering; Medicine; Social science; Medical education","authors":[{"name":"Cameron D. Norman","is_ca":true},{"name":"Allan Best","is_ca":true},{"name":"Sharon T. Mortimer","is_ca":true},{"name":"Timothy R. Huerta","is_ca":false},{"name":"A.M.J. Buchan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2891501171514995,"gpt":0.5838757997116935,"spread":0.2947256825601939,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.174943,0.00121159,0.002318626,0.006500632,0.003260616,0.0113854,0.001767978,0.002378021,0.002902748],"category_scores_gemma":[0.3850008,0.0005507854,0.001560012,0.006591642,0.009825239,0.009418487,0.007536601,0.002618602,0.000188594],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0142338,"about_ca_system_score_gemma":0.01528035,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005463939,"about_ca_topic_score_gemma":0.004859953,"domain_scores_codex":[0.657289,0.3118983,0.006916304,0.003412491,0.01846792,0.002015992],"domain_scores_gemma":[0.3711487,0.5841909,0.01735689,0.009757331,0.01411246,0.003433753],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.002456707,0.002226433,0.08196265,0.005831198,0.00289207,0.0003843595,0.006033222,0.2780213,0.001054388,0.3566656,0.003736009,0.258736],"study_design_scores_gemma":[0.0006542319,0.008235364,0.02310676,0.002091086,0.001125777,0.0002390251,0.01088767,0.3692299,0.003323398,0.5690762,0.01176083,0.0002698804],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6285566,0.0144631,0.2477834,0.02431371,0.000682074,0.007189738,0.0009884919,0.0002538754,0.07576893],"genre_scores_gemma":[0.9082904,0.002227718,0.08634306,0.0005487455,0.0001139258,0.001750961,0.0001525313,0.00002146143,0.0005511542],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.825057,"threshold_uncertainty_score":0.9251979,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3208939115","doi":"10.1177/1098214020936769","title":"The Use of Evaluability Assessments in Improving Future Evaluations: A Scoping Review of 10 Years of Literature (2008–2018)","year":2021,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":14,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; University of Guelph","funders":"","keywords":"Ambiguity; Psychology; Engineering ethics; Equity (law); Relevance (law); Management science; Political science; Engineering; Computer science","authors":[{"name":"Steven Lâm","is_ca":true},{"name":"Kelly Skinner","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3614297121157451,"gpt":0.6078438813411042,"spread":0.2464141692253591,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1707472,0.001848716,0.004163675,0.0356332,0.002375867,0.007143104,0.002581965,0.00285594,0.003184413],"category_scores_gemma":[0.3521426,0.001652154,0.005458835,0.02698022,0.002985786,0.009955379,0.004614105,0.003138896,0.0005955464],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009703105,"about_ca_system_score_gemma":0.04526919,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01069714,"about_ca_topic_score_gemma":0.02489615,"domain_scores_codex":[0.8831443,0.06618196,0.02825104,0.003273722,0.01793993,0.001209023],"domain_scores_gemma":[0.5665203,0.3434977,0.03032964,0.007311644,0.05108504,0.001255592],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0001393811,0.00005632607,0.001601269,0.4223826,0.001519809,0.0001433965,0.003350172,0.0005047263,0.000303745,0.005759147,0.008148799,0.5560906],"study_design_scores_gemma":[0.00003085478,0.00007082749,0.001955319,0.918065,0.003246182,0.0001461546,0.001468,0.0002054281,0.0003137795,0.002319762,0.07213477,0.00004383843],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.001118342,0.9895198,0.002940457,0.002761074,0.0004146266,0.0009197474,0.0002521065,0.00002339922,0.002050423],"genre_scores_gemma":[0.01601227,0.9725578,0.007591161,0.001032633,0.0001946185,0.002025092,0.0003054569,0.00002745366,0.0002533914],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8292528,"threshold_uncertainty_score":0.9030084,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2063922189","doi":"10.1177/1098214007304536","title":"Analysis of Thin Online Interview Data","year":2007,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Focus Groups and Qualitative Methods","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"","keywords":"Meaning (existential); Computer science; Context (archaeology); Qualitative property; Perception; Data collection; Semantics (computer science); Data science; Qualitative analysis; Qualitative research; Psychology; Sociology; Machine learning","authors":[{"name":"Richard J. Kitto","is_ca":true},{"name":"John Barnett","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3466871645174802,"gpt":0.5830954045773991,"spread":0.2364082400599189,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04955923,0.0008848597,0.001041415,0.006723556,0.003145188,0.00248134,0.00190618,0.0009034008,0.01213612],"category_scores_gemma":[0.1678944,0.0007579456,0.0004685851,0.007097464,0.00245394,0.002794508,0.004952084,0.001812057,0.00272618],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003178503,"about_ca_system_score_gemma":0.005330491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002268156,"about_ca_topic_score_gemma":0.003623849,"domain_scores_codex":[0.9434441,0.03549924,0.006111395,0.003196554,0.01033713,0.001411562],"domain_scores_gemma":[0.7354707,0.1641988,0.01729336,0.02499888,0.05621768,0.001820549],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.00122405,0.0007026765,0.03511804,0.007702193,0.0001407955,0.002539048,0.3773907,0.002264703,0.02557844,0.04228316,0.03246457,0.4725916],"study_design_scores_gemma":[0.0002290245,0.001099718,0.0654721,0.007631163,0.0001331897,0.001759307,0.4127765,0.01550921,0.02864535,0.06992067,0.3965373,0.0002865147],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.4821796,0.0008030158,0.3841838,0.003513519,0.0005784166,0.04037525,0.03037688,0.001295604,0.05669388],"genre_scores_gemma":[0.5066553,0.0008213947,0.3746781,0.001695111,0.0001606567,0.08352979,0.01595898,0.0008277086,0.01567291],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9504408,"threshold_uncertainty_score":0.2620974,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2070663696","doi":"10.1177/1098214010371817","title":"A Realist Evaluation Approach to Unpacking the Impacts of the Sentencing Guidelines","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Criminal Justice and Corrections Analysis","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; St. Michael's Hospital","funders":"","keywords":"Unpacking; Prison; Sentencing guidelines; Work (physics); Psychological intervention; Resource (disambiguation); Impact evaluation; Key (lock); Control (management); Linkage (software); Public economics; Sociology; Management science; Criminology; Political science; Psychology; Economics; Computer science; Computer security","authors":[{"name":"Kim Hunt","is_ca":false},{"name":"Sanjeev Sridharan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08324908097625391,"gpt":0.4245087693437239,"spread":0.34125968836747,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1246737,0.00189923,0.002452385,0.0059081,0.002642897,0.009041074,0.002972827,0.00279809,0.008703293],"category_scores_gemma":[0.1841532,0.001014385,0.001833803,0.002946395,0.01277389,0.01137047,0.005328909,0.00477653,0.0003237036],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01362843,"about_ca_system_score_gemma":0.01684675,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005303892,"about_ca_topic_score_gemma":0.007551192,"domain_scores_codex":[0.7963806,0.182933,0.003656978,0.00389117,0.01134557,0.001792722],"domain_scores_gemma":[0.7888367,0.1839984,0.009783783,0.008729085,0.007505709,0.00114635],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005399717,0.0007182433,0.005145205,0.00307768,0.0009410692,0.0002164459,0.004439966,0.02994672,0.0006411594,0.8609806,0.002806009,0.09054694],"study_design_scores_gemma":[0.0007411971,0.002934581,0.007649017,0.002253386,0.0008524348,0.0001415997,0.005744477,0.07195447,0.002184285,0.8711768,0.03414981,0.0002178986],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07558848,0.004276105,0.7514526,0.02923648,0.001053528,0.01008252,0.0008339921,0.0004155215,0.1270608],"genre_scores_gemma":[0.7260959,0.00166692,0.255861,0.002091959,0.0002619608,0.0103169,0.0001353078,0.00008694157,0.003483124],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1246737,"threshold_uncertainty_score":0.6593454,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2971517022","doi":"10.1177/1098214019866260","title":"Evaluations in the English-Speaking Commonwealth Caribbean Region: Lessons From the Field","year":2019,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Commonwealth; Pride; Nexus (standard); Face (sociological concept); Public relations; Political science; Sociology; Psychology; Social psychology; Social science; Law","authors":[{"name":"Nadini Persaud","is_ca":false},{"name":"Ruby Dagher","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2049595892506882,"gpt":0.5292134531119421,"spread":0.3242538638612539,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06784406,0.0004699727,0.0009769626,0.002631188,0.01250366,0.01722516,0.002446722,0.003635935,0.004706574],"category_scores_gemma":[0.09283815,0.0002976203,0.0005164228,0.003445483,0.01193645,0.006967639,0.006848414,0.005380101,0.0003734891],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02753032,"about_ca_system_score_gemma":0.08780412,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1407157,"about_ca_topic_score_gemma":0.2068408,"domain_scores_codex":[0.9343705,0.0534496,0.001425875,0.001064487,0.004086589,0.005602916],"domain_scores_gemma":[0.8056085,0.1313509,0.005061563,0.004038375,0.03522698,0.01871382],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005963822,0.002205679,0.02472986,0.005378432,0.0001603622,0.003991125,0.2146297,0.002206209,0.0008000488,0.116395,0.07128651,0.5576207],"study_design_scores_gemma":[0.000189113,0.0007021939,0.03285742,0.01863542,0.00008382687,0.0006789462,0.5820024,0.0009592334,0.0009963022,0.04482389,0.3178438,0.0002273853],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.2182844,0.1004576,0.007946042,0.4353076,0.002155151,0.001343124,0.0001739578,0.0001048574,0.2342272],"genre_scores_gemma":[0.935555,0.02848557,0.005809742,0.02115004,0.0005301274,0.0005125653,0.00005558377,0.00006090639,0.00784058],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1407157,"threshold_uncertainty_score":0.358798,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2884511286","doi":"10.1177/1098214018781506","title":"Making Space for Adaptive Learning","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"University of Ottawa","keywords":"Interrogation; Space (punctuation); Mediation; Psychology; Social learning; Process (computing); Social psychology; Computer science; Sociology; Pedagogy; Political science; Social science","authors":[{"name":"Barbara Szijarto","is_ca":true},{"name":"J. Bradley Cousins","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3892532860461554,"gpt":0.5842406175819069,"spread":0.1949873315357515,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0186091,0.0006270022,0.0004139149,0.001349034,0.00535308,0.01047041,0.002437486,0.002712247,0.00917422],"category_scores_gemma":[0.05144765,0.0004481327,0.0007424307,0.0005927907,0.02229178,0.01966351,0.01844763,0.00355641,0.001025943],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002401543,"about_ca_system_score_gemma":0.007052934,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001064907,"about_ca_topic_score_gemma":0.001601221,"domain_scores_codex":[0.9786142,0.01535486,0.0006966915,0.00152188,0.002547587,0.001264846],"domain_scores_gemma":[0.9583796,0.02489447,0.003423658,0.006324415,0.003670824,0.003306973],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001504053,0.0003393677,0.01052678,0.0005899107,0.00005522512,0.0008411006,0.1239853,0.003356074,0.004367659,0.6451216,0.005387417,0.2052791],"study_design_scores_gemma":[0.00008943997,0.0003844455,0.005114081,0.000821802,0.00005288626,0.0006603855,0.1048792,0.004386424,0.006475233,0.6269127,0.250087,0.0001363877],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3378102,0.00271084,0.293062,0.05286011,0.0006680273,0.0009353665,0.0001071997,0.0009461883,0.3109002],"genre_scores_gemma":[0.9586256,0.0004049216,0.03619913,0.0003953966,0.00005937896,0.0002248761,0.00002753416,0.0000692914,0.00399389],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0186091,"threshold_uncertainty_score":0.09841555,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2034211783","doi":"10.1177/1098214015578731","title":"Merging Developmental and Feminist Evaluation to Monitor and Evaluate Transformative Social Change","year":2015,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Transformative learning; Theory of change; Sociology; Social transformation; Social change; Monitoring and evaluation; Participatory evaluation; Program evaluation; Political science; Pedagogy; Social science; Public administration; Law","authors":[{"name":"Laura Haylock","is_ca":false},{"name":"Carol Miller","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3677671345545405,"gpt":0.5374349974415299,"spread":0.1696678628869894,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1393768,0.0008133007,0.0008136886,0.007912003,0.005194254,0.01081604,0.002562611,0.001273982,0.003613068],"category_scores_gemma":[0.09939618,0.0004241999,0.0004745936,0.004172122,0.01484973,0.006252483,0.01101344,0.002871532,0.0003031445],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.07248326,"about_ca_system_score_gemma":0.07787453,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1451645,"about_ca_topic_score_gemma":0.2217649,"domain_scores_codex":[0.840982,0.121491,0.003475104,0.00372838,0.02614167,0.004181928],"domain_scores_gemma":[0.8811569,0.05683735,0.005869173,0.006878764,0.04478983,0.004467917],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001860693,0.0003430152,0.02238365,0.001233258,0.000110012,0.0001381138,0.03057943,0.004928321,0.001191442,0.3704319,0.01312294,0.555352],"study_design_scores_gemma":[0.0003190648,0.001140426,0.05593437,0.005716819,0.0002716323,0.0003192745,0.0806702,0.03250471,0.01499909,0.3051062,0.5026208,0.0003974304],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07481776,0.00738579,0.4392272,0.08041549,0.0006592595,0.004797498,0.0004693532,0.000930396,0.3912974],"genre_scores_gemma":[0.7641295,0.001922741,0.2170396,0.00427318,0.0001257388,0.002068864,0.0001413751,0.0001871734,0.01011191],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8606232,"threshold_uncertainty_score":0.7371037,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2097114336","doi":"10.1177/1098214010378355","title":"Evaluating Capacity Building for Policy Research Organizations","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"International Development and Aid","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"International Development Research Centre","keywords":"Capacity building; Legislation; Business; Key (lock); Corporate governance; Process management; Program evaluation; Public relations; Economics; Political science; Computer science; Economic growth; Public administration; Computer security; Finance","authors":[{"name":"Raymond J. Struyk","is_ca":false},{"name":"Mawadda Damon","is_ca":false},{"name":"Samuel R. Haddaway","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1885404601512432,"gpt":0.5539368367300723,"spread":0.3653963765788291,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3844558,0.001681193,0.001531401,0.008924828,0.00379371,0.008867854,0.003583832,0.003543557,0.007501693],"category_scores_gemma":[0.4514221,0.0008325402,0.001859202,0.005537134,0.003993255,0.009289513,0.01350367,0.002997438,0.0008789256],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0204154,"about_ca_system_score_gemma":0.03125836,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002288143,"about_ca_topic_score_gemma":0.003007676,"domain_scores_codex":[0.5385078,0.4070103,0.01324778,0.005185752,0.02511694,0.01093144],"domain_scores_gemma":[0.3303434,0.501613,0.04443618,0.02883071,0.07273974,0.02203702],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.006413781,0.009451305,0.08418372,0.007866802,0.002243644,0.0003396001,0.01845968,0.05208183,0.002472569,0.1283617,0.0129656,0.6751598],"study_design_scores_gemma":[0.0118624,0.07764861,0.1937912,0.02270296,0.006904203,0.000510018,0.0933542,0.1299383,0.04000301,0.2559446,0.1661649,0.001175602],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6341571,0.004565605,0.112258,0.01869892,0.001085791,0.0582753,0.001486848,0.0007727506,0.1686997],"genre_scores_gemma":[0.864397,0.001258882,0.09908035,0.0009467363,0.0001851769,0.03066466,0.0005509839,0.0000532349,0.002862934],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6155442,"threshold_uncertainty_score":0.7590756,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2033993956","doi":"10.1177/1098214013478146","title":"The Practice of Evaluation in Public Sector Contexts","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Accountability; Public sector; Citizen journalism; Perception; Diversity (politics); Reflection (computer programming); Sociology; Evaluation methods; Public relations; Public administration; Political science; Psychology; Law; Computer science; Engineering","authors":[{"name":"Jill Anne Chouinard","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1817166479435985,"gpt":0.51942008015783,"spread":0.3377034322142315,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3425305,0.001525801,0.003184569,0.01070911,0.01150816,0.03474826,0.006107853,0.01293127,0.003956072],"category_scores_gemma":[0.2523872,0.001293159,0.00146064,0.01041536,0.1520651,0.03228252,0.01689216,0.01562345,0.001062139],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02182857,"about_ca_system_score_gemma":0.02793731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004794573,"about_ca_topic_score_gemma":0.00367116,"domain_scores_codex":[0.4587395,0.4865401,0.01482103,0.01322739,0.02283511,0.003836682],"domain_scores_gemma":[0.534387,0.3945962,0.01127451,0.03503078,0.02136959,0.003341969],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.00002822009,0.00003634308,0.0005503435,0.000721532,0.00003975028,0.0001228394,0.02083293,0.0006421516,0.0001228589,0.9327463,0.004572503,0.0395842],"study_design_scores_gemma":[0.00003760223,0.00007360461,0.000428561,0.003290885,0.00002373063,0.0001907466,0.01468482,0.001460876,0.0005287577,0.863371,0.1158419,0.00006755743],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01139737,0.04191578,0.4885384,0.2120837,0.004523337,0.001351234,0.0001140648,0.0005525623,0.2395235],"genre_scores_gemma":[0.6548187,0.01573702,0.2919558,0.01951932,0.002384825,0.004506476,0.00006840732,0.0003642057,0.01064527],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3425305,"threshold_uncertainty_score":0.8107769,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2046660743","doi":"10.1177/1098214008327931","title":"Do Self-Assessments Work to Detect Workshop Success?","year":2009,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Human Resource Development and Performance Evaluation","field":"Psychology","cited_by":10,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Self-assessment; Applied psychology; Psychology; Work (physics); Multilevel model; Self-report study; Computer science; Social psychology; Machine learning; Engineering","authors":[{"name":"Tony C. M. Lam","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04228701723909126,"gpt":0.418141853921784,"spread":0.3758548366826928,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07119654,0.0006593264,0.0009960586,0.003704401,0.0006668904,0.003120743,0.002024635,0.002151313,0.001194813],"category_scores_gemma":[0.2486245,0.0004170288,0.0009648716,0.002080794,0.002349217,0.004591782,0.001790075,0.00212793,0.001017013],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009889831,"about_ca_system_score_gemma":0.001189361,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002584303,"about_ca_topic_score_gemma":0.005164067,"domain_scores_codex":[0.9335927,0.04304506,0.006245525,0.003460709,0.01254121,0.001114803],"domain_scores_gemma":[0.7249181,0.1925206,0.03685825,0.01136141,0.03192366,0.002418126],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0005221828,0.000320446,0.6823725,0.001324259,0.0006326722,0.0000935049,0.006867929,0.0006157076,0.0006349655,0.00454675,0.01392301,0.2881461],"study_design_scores_gemma":[0.0001889964,0.002021558,0.8931494,0.004440474,0.0006509661,0.0008520141,0.01726391,0.01185512,0.007469099,0.02515802,0.03655847,0.0003918443],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7259948,0.01721677,0.147465,0.05053078,0.005926317,0.001091109,0.002127107,0.001160264,0.04848788],"genre_scores_gemma":[0.9659453,0.001624777,0.0254622,0.003654249,0.0003399224,0.0006784716,0.0002824784,0.00005889834,0.001953763],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07119654,"threshold_uncertainty_score":0.3765278,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2063748659","doi":"10.1016/s1098-2140(00)00090-4","title":"Planning for community-based evaluation","year":2000,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":9,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; Centre for Global Health Research","funders":"","keywords":"Management science; Program evaluation; Set (abstract data type); Unit (ring theory); Process (computing); Conflict resolution; Evaluation methods; Process management; Computer science; Psychology; Sociology; Political science; Business; Engineering; Mathematics education","authors":[{"name":"Rhonda Cockerill","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.355370496696108,"gpt":0.5926022913441982,"spread":0.2372317946480902,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03541558,0.001059823,0.000763468,0.005688754,0.01416916,0.01451943,0.00456388,0.007641082,0.03283267],"category_scores_gemma":[0.1006708,0.001013874,0.00117812,0.004751764,0.003748498,0.01008135,0.008961415,0.006964405,0.003007591],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02308473,"about_ca_system_score_gemma":0.1173212,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.05777533,"about_ca_topic_score_gemma":0.1578162,"domain_scores_codex":[0.9593432,0.02699618,0.001229978,0.001494077,0.005663515,0.005273074],"domain_scores_gemma":[0.9273653,0.0208072,0.003749718,0.00246478,0.02383917,0.02177386],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003556554,0.003503864,0.02515558,0.001081554,0.0001440962,0.001704336,0.006809134,0.02712192,0.001016256,0.3445016,0.1770607,0.4115453],"study_design_scores_gemma":[0.0004240492,0.001133235,0.02291688,0.002583122,0.0001161933,0.0007449423,0.04215396,0.0478591,0.001814857,0.5959497,0.2839994,0.0003045997],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.09561433,0.004650143,0.2853369,0.1634026,0.001246452,0.01977744,0.000858921,0.001439153,0.4276741],"genre_scores_gemma":[0.7000952,0.001576253,0.2598679,0.004567833,0.0003073787,0.00725451,0.0008740968,0.0002207407,0.02523619],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.05777533,"threshold_uncertainty_score":0.1872977,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4235318396","doi":"10.1177/109821400002100306","title":"Planning for Community-based Evaluation","year":2000,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Program evaluation; Unit (ring theory); Management science; Set (abstract data type); Process (computing); Conflict resolution; Process management; Psychology; Computer science; Sociology; Political science; Business; Engineering; Mathematics education; Social science","authors":[{"name":"Rhonda Cockerill","is_ca":true},{"name":"Ted Myers","is_ca":false},{"name":"Dan Allman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.355370496696108,"gpt":0.5926022913441982,"spread":0.2372317946480902,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1871466,0.002344339,0.001580929,0.007611756,0.009463544,0.01313531,0.006426939,0.006671787,0.03626387],"category_scores_gemma":[0.237008,0.001647365,0.00208592,0.006857105,0.005060877,0.01194636,0.01481273,0.01021183,0.01191213],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01655494,"about_ca_system_score_gemma":0.1151069,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01629358,"about_ca_topic_score_gemma":0.03197418,"domain_scores_codex":[0.7813758,0.1759454,0.0101663,0.003452887,0.02401144,0.005048173],"domain_scores_gemma":[0.7509513,0.1094206,0.009600819,0.01759967,0.0913053,0.02112232],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001667328,0.000945856,0.001251102,0.002960956,0.00007784449,0.0007993412,0.008530861,0.01067541,0.0006389047,0.1386033,0.2787861,0.5565636],"study_design_scores_gemma":[0.0003386522,0.0005209848,0.001663537,0.007570773,0.00006353058,0.0004194226,0.01253003,0.00702796,0.001026226,0.1750551,0.7935275,0.000256186],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005045379,0.004836369,0.6097273,0.07798723,0.002684036,0.09517651,0.001086757,0.002389382,0.201067],"genre_scores_gemma":[0.02839404,0.002880281,0.8865543,0.004423356,0.0002831589,0.05257732,0.0009998729,0.0004765505,0.02341109],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1871466,"threshold_uncertainty_score":0.9897375,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4283033908","doi":"10.1177/10982140211056913","title":"Developing Evaluation Approaches for an Anti-Human Trafficking Housing Program","year":2022,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Migration, Health and Trauma","field":"Psychology","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Formative assessment; Protocol (science); Program evaluation; Evaluation methods; Human trafficking; Public relations; Best practice; Psychology; Process management; Political science; Engineering ethics; Business; Medicine; Engineering; Public administration; Pedagogy; Alternative medicine","authors":[{"name":"Rebecca J. Macy","is_ca":false},{"name":"Amanda Eckhardt","is_ca":false},{"name":"Christopher J. Wretman","is_ca":false},{"name":"Ran Hu","is_ca":true},{"name":"Jeong-Suk Kim","is_ca":false},{"name":"Xinyi Wang","is_ca":false},{"name":"Cindy Bombeeck","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2469876266596626,"gpt":0.4708132701818361,"spread":0.2238256435221735,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.4301727,0.001479034,0.001401351,0.006568038,0.006747271,0.008569424,0.004392555,0.003011563,0.007025877],"category_scores_gemma":[0.3276603,0.001376304,0.002146457,0.003138803,0.007648983,0.01071438,0.008928987,0.004893059,0.0007897459],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02222527,"about_ca_system_score_gemma":0.05726645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003124406,"about_ca_topic_score_gemma":0.006589245,"domain_scores_codex":[0.5100284,0.4487361,0.01793897,0.003455813,0.01671206,0.003128754],"domain_scores_gemma":[0.6324105,0.2907858,0.0149688,0.01514608,0.04242589,0.004262873],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.001708419,0.01060086,0.008788983,0.01614957,0.0003677924,0.0005086471,0.1191531,0.009049056,0.003438036,0.1258158,0.008132579,0.6962871],"study_design_scores_gemma":[0.01393693,0.04898214,0.03336895,0.05879096,0.001869952,0.001018008,0.3874185,0.03728649,0.03372506,0.221733,0.1607341,0.00113606],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1254207,0.001948598,0.4189019,0.01326807,0.0006817193,0.3958716,0.0004726797,0.0004665399,0.04296818],"genre_scores_gemma":[0.1417805,0.0007974623,0.6362047,0.001137748,0.00004257878,0.218584,0.00009266843,0.000040691,0.001319682],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4301727,"threshold_uncertainty_score":0.7026985,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2063290014","doi":"10.1177/1098214012464426","title":"Improving Program Results Through the Use of Predictive Operational Performance Indicators","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Employment and Social Development Canada; Carleton University","funders":"","keywords":"Accountability; Context (archaeology); Performance indicator; Program evaluation; Process management; Quality (philosophy); Computer science; Risk analysis (engineering); Term (time); Operations management; Environmental economics; Business; Engineering; Marketing; Economics; Political science; Public administration","authors":[{"name":"Maria Barrados","is_ca":true},{"name":"Julie Blain","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1859485882753664,"gpt":0.471487866355326,"spread":0.2855392780799596,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05813128,0.001552692,0.0007739289,0.009569635,0.001261711,0.007123193,0.001489517,0.0004943052,0.001675977],"category_scores_gemma":[0.1183315,0.0004009379,0.0005368164,0.008525653,0.0009858073,0.004599698,0.003065695,0.001584441,0.0004939375],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00949736,"about_ca_system_score_gemma":0.0187986,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.05525202,"about_ca_topic_score_gemma":0.07204926,"domain_scores_codex":[0.963301,0.01957917,0.002756786,0.001570284,0.01153285,0.001259909],"domain_scores_gemma":[0.8646865,0.05630993,0.03048848,0.007571346,0.03844741,0.002496236],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000571951,0.001399439,0.1998896,0.001294133,0.0002438279,0.00008665925,0.005307578,0.02166127,0.002688398,0.01706812,0.01268959,0.7370995],"study_design_scores_gemma":[0.0002296008,0.005013559,0.6377665,0.004810644,0.0007745132,0.0002304555,0.01484021,0.1830304,0.03400754,0.0338287,0.08462436,0.0008434514],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.4325708,0.002647051,0.3632582,0.006672709,0.000358722,0.007075882,0.006767347,0.006198488,0.1744509],"genre_scores_gemma":[0.8206455,0.001085532,0.1727144,0.0003065354,0.0000459121,0.001261934,0.00146061,0.0001514415,0.002328147],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.05813128,"threshold_uncertainty_score":0.3074313,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4392775921","doi":"10.1177/10982140241234841","title":"Mapping Evaluation Use: A Scoping Review of Extant Literature (2005–2022)","year":2024,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta; Queen's University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Extant taxon; Program evaluation; Management science; Evaluation methods; Psychology; Sociology; Political science; Engineering; Public administration","authors":[{"name":"Michelle Searle","is_ca":true},{"name":"Amanda Cooper","is_ca":true},{"name":"Paisley Worthington","is_ca":true},{"name":"Jennifer Hughes","is_ca":true},{"name":"Rebecca Gokiert","is_ca":true},{"name":"Cheryl Poth","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3963496705860396,"gpt":0.6096230452322937,"spread":0.2132733746462541,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07300958,0.00161291,0.003423298,0.05725409,0.002237503,0.005595818,0.002087658,0.002427098,0.002994407],"category_scores_gemma":[0.1932113,0.001888225,0.003797541,0.05570759,0.001973459,0.007174575,0.00520608,0.001812442,0.0007241163],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008754549,"about_ca_system_score_gemma":0.04775911,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01589311,"about_ca_topic_score_gemma":0.04680869,"domain_scores_codex":[0.9604685,0.01417315,0.01394209,0.001788268,0.008797949,0.0008301027],"domain_scores_gemma":[0.8542209,0.09033804,0.01778328,0.003875351,0.03264312,0.001139323],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0001244738,0.0000486387,0.002948867,0.4582332,0.001288698,0.0003068826,0.004556241,0.000406925,0.0006188703,0.002326848,0.01247199,0.5166683],"study_design_scores_gemma":[0.00002662452,0.00009766914,0.005985023,0.9151731,0.002793818,0.0002630125,0.002933895,0.0001555871,0.0004249375,0.0008651129,0.07123532,0.00004597777],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.004342615,0.9843659,0.002433128,0.00244256,0.0004922927,0.001889023,0.0009830904,0.0000419404,0.003009347],"genre_scores_gemma":[0.01913995,0.9691161,0.006462206,0.000873099,0.0001177709,0.002879239,0.000890885,0.00003078844,0.000489886],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9269904,"threshold_uncertainty_score":0.3861162,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2140628270","doi":"10.1177/1098214014532166","title":"How Analogue Research Can Advance Descriptive Evaluation Theory","year":2014,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Interpersonal communication; Frame (networking); Computer science; Management science; Accountability; Evaluation methods; Epistemology; Psychology; Social psychology; Political science","authors":[{"name":"Bernadette Campbell","is_ca":true},{"name":"Melvin M. Mark","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3507628690464508,"gpt":0.5660418537784816,"spread":0.2152789847320308,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08524376,0.001336414,0.001762064,0.006988388,0.002386783,0.01165511,0.003594596,0.003330301,0.01533156],"category_scores_gemma":[0.2189586,0.0009619939,0.001317375,0.004853977,0.02848795,0.02693962,0.00790206,0.0075793,0.001554388],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005549106,"about_ca_system_score_gemma":0.004701214,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002239178,"about_ca_topic_score_gemma":0.00177456,"domain_scores_codex":[0.9073715,0.07821893,0.002632111,0.003418546,0.007393396,0.0009653968],"domain_scores_gemma":[0.6768851,0.2789116,0.00533118,0.02563858,0.01159243,0.001641084],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00005625774,0.00007432199,0.0004255092,0.0002968489,0.00002143644,0.00006309299,0.002335301,0.0009471654,0.00008638779,0.9712147,0.001069983,0.02340902],"study_design_scores_gemma":[0.00004492092,0.00006171782,0.0002187652,0.000340607,0.00001065482,0.00005518501,0.0009373698,0.001876491,0.0001784097,0.9781317,0.01811679,0.00002750367],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02079582,0.01139256,0.6458919,0.03054238,0.002611715,0.001443965,0.0002829244,0.0003653469,0.2866733],"genre_scores_gemma":[0.6218798,0.007308126,0.3520033,0.005404422,0.001807848,0.002929332,0.0001848426,0.0002189121,0.008263374],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.08524376,"threshold_uncertainty_score":0.4508175,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2901688885","doi":"10.1177/1098214018796319","title":"Honoring Lived Experience: Life Histories as a Realist Evaluation Method","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University; University of British Columbia; Impact; St. Michael's Hospital","funders":"Tehran University of Medical Sciences and Health Services","keywords":"Lived experience; Context (archaeology); Set (abstract data type); Psychology; Indigenous; Sociology; Epistemology; Computer science; History; Psychotherapist","authors":[{"name":"Emma Richardson","is_ca":true},{"name":"Mary Phillips","is_ca":false},{"name":"Alejandra Colom","is_ca":false},{"name":"Jennica Nichols","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.365242125327843,"gpt":0.5972764607902662,"spread":0.2320343354624232,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0729803,0.0009289782,0.0007014651,0.005696209,0.004515259,0.005820823,0.002716025,0.001308468,0.009001489],"category_scores_gemma":[0.095426,0.0008088898,0.0006129029,0.003820658,0.008705724,0.007340001,0.01035491,0.002432824,0.0008987842],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0056477,"about_ca_system_score_gemma":0.005167214,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001636173,"about_ca_topic_score_gemma":0.004881443,"domain_scores_codex":[0.8772202,0.1124642,0.002744113,0.002658451,0.003816414,0.001096617],"domain_scores_gemma":[0.9042284,0.0602576,0.008371769,0.01200174,0.0116416,0.003498934],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001369348,0.001826032,0.0451367,0.002945295,0.000234078,0.000937641,0.4384934,0.001420649,0.003067062,0.08399209,0.009277587,0.4113002],"study_design_scores_gemma":[0.0008560201,0.003697504,0.044974,0.005340539,0.0003770512,0.0009408385,0.5641378,0.00567217,0.009462442,0.1255277,0.2386617,0.0003523042],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5440579,0.003441738,0.2770446,0.009697029,0.0006826438,0.03153413,0.002435398,0.0002530837,0.1308535],"genre_scores_gemma":[0.7819418,0.0009930116,0.1483752,0.001317767,0.0001181824,0.05970327,0.0006954882,0.0001269977,0.006728238],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9270197,"threshold_uncertainty_score":0.3859614,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}