{"meta":{"query_hash":"24b6424d1b26","filters":{"venue":"American Journal of Evaluation"},"cohort_total":60,"direct_labels_cover":1,"predictions_cover":60,"exported":60,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/24b6424d1b26","api":"https://metacan.xera.ac/api/v1/cohort?venue=American+Journal+of+Evaluation"},"results":[{"id":"W1965824013","doi":"10.1177/1098214013478142","title":"The Case for Participatory Evaluation in an Era of Accountability","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":165,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Accountability; Technocracy; Citizen journalism; Participatory evaluation; Context (archaeology); Participatory GIS; Public sector; Sociology; Politics; Public administration; Public relations; Government (linguistics); Social accounting; Political science; Business; Accounting; Law","score_opus":0.3658990701241563,"score_gpt":0.5819220676204166,"score_spread":0.21602299749626036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1965824013","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":"methods","domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006311462,0.00987296,0.14392485,0.63037205,0.0034831143,0.00074402615,0.00006235111,0.00018744558,0.20504183],"genre_scores_gemma":[0.7358721,0.006076299,0.14036028,0.08818028,0.0038838943,0.0041635013,0.000060618866,0.00040705773,0.020996025],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.43638754,0.48815,0.0075266217,0.01924803,0.03947902,0.009208781],"domain_scores_gemma":[0.5521003,0.3530542,0.01426022,0.04126886,0.028409185,0.010907163],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.38410926,0.0015802792,0.002539853,0.0054246793,0.020372022,0.037460733,0.005565479,0.020682137,0.00607849],"category_scores_gemma":[0.28712684,0.0014739732,0.002193286,0.004498021,0.14775026,0.05836201,0.02999454,0.03143195,0.0011317369],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002612646,0.000026728818,0.0003052581,0.00018585566,0.000025059571,0.000107909415,0.011262294,0.00031122295,0.000055486067,0.97175795,0.0055352114,0.010400836],"study_design_scores_gemma":[0.000039570124,0.00004198078,0.00022755687,0.0009409554,0.000012860869,0.00012095843,0.0059619765,0.0006141496,0.00013311546,0.8933764,0.098486125,0.000044307144],"about_ca_topic_score_codex":0.007304897,"about_ca_topic_score_gemma":0.005667371,"teacher_disagreement_score":0.38410926,"about_ca_system_score_codex":0.024871014,"about_ca_system_score_gemma":0.053899303,"threshold_uncertainty_score":0.75950295},"labels":[],"label_agreement":null},{"id":"W1975564755","doi":"10.1177/1098214013477235","title":"Understanding Dimensions of Organizational Evaluation Capacity","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":118,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa; École Nationale d'Administration Publique","funders":"Australian Government","keywords":"Dimension (graph theory); Capacity building; Government (linguistics); Knowledge management; Organization development; Business; Organizational learning; Process management; Computer science; Economics; Economic growth","score_opus":0.5129125946126178,"score_gpt":0.4844682783398096,"score_spread":0.028444316272808245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975564755","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.83187836,0.0013039424,0.018886829,0.0074680145,0.000027714532,0.00015211888,0.0001293365,0.000053938566,0.1400998],"genre_scores_gemma":[0.9981254,0.00012653985,0.0013403636,0.000051440165,0.0000032508858,0.000025150925,0.000025004132,0.0000028249628,0.00030013613],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9903617,0.004234136,0.0005569273,0.00051606324,0.0022202667,0.002111042],"domain_scores_gemma":[0.9601122,0.022811536,0.0038582396,0.0021953867,0.007701904,0.0033207794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011760495,0.00033100837,0.000251636,0.0054334537,0.0031474584,0.0076675583,0.0010519155,0.0010782866,0.001955654],"category_scores_gemma":[0.032389373,0.00026762643,0.0003565613,0.0028981185,0.013836627,0.008487763,0.0061709597,0.0013783795,0.00010399176],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000085377585,0.00016612693,0.16933325,0.00042282892,0.000068194546,0.00028454151,0.15818077,0.0053022667,0.0015419619,0.5604243,0.0025243019,0.10166609],"study_design_scores_gemma":[0.000028262468,0.00012419184,0.28397694,0.0011232314,0.000055996043,0.00032991258,0.3372654,0.0130932415,0.0016575905,0.29944205,0.06274616,0.00015701498],"about_ca_topic_score_codex":0.076252244,"about_ca_topic_score_gemma":0.05396079,"teacher_disagreement_score":0.076252244,"about_ca_system_score_codex":0.01782574,"about_ca_system_score_gemma":0.01767699,"threshold_uncertainty_score":0.15161681},"labels":[],"label_agreement":null},{"id":"W1989835141","doi":"10.1177/1098214009349865","title":"A Review and Synthesis of Current Research on Cross-Cultural Evaluation","year":2009,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":91,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Construct (python library); Context (archaeology); Indigenous; Sociology; Empirical research; Management science; Knowledge management; Engineering ethics; Epistemology; Computer science; Ecology","score_opus":0.7361957222054291,"score_gpt":0.7346471200265722,"score_spread":0.0015486021788568838,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1989835141","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00095623214,0.98962116,0.002019227,0.001966982,0.00039835807,0.00006842449,0.000055131783,0.000019862306,0.00489454],"genre_scores_gemma":[0.008049755,0.9878169,0.002691254,0.00055314874,0.00020570283,0.00012167224,0.00007281925,0.0000118760845,0.00047693407],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.98687965,0.006160551,0.002669234,0.0007319265,0.0032681176,0.00029059852],"domain_scores_gemma":[0.9126095,0.07110912,0.0038332676,0.001738885,0.01003186,0.00067736336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018790733,0.00096377137,0.0032914632,0.0133009525,0.0011178997,0.005517026,0.0015600601,0.002156009,0.0068840417],"category_scores_gemma":[0.05930339,0.00059390854,0.0011931026,0.02032941,0.0017743543,0.0061834157,0.0019225843,0.0014434913,0.0015423761],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000071924565,0.00009402713,0.00070553,0.06813958,0.00019573886,0.00015322951,0.001493093,0.00038729602,0.00037091502,0.005670491,0.012805255,0.9099128],"study_design_scores_gemma":[0.000040429924,0.00036525904,0.009699894,0.30989188,0.0010481972,0.0012949496,0.0065193097,0.00051750254,0.0010586547,0.013218383,0.65624017,0.00010531727],"about_ca_topic_score_codex":0.0042713624,"about_ca_topic_score_gemma":0.009102548,"teacher_disagreement_score":0.018790733,"about_ca_system_score_codex":0.004478382,"about_ca_system_score_gemma":0.011794443,"threshold_uncertainty_score":0.09937608},"labels":[],"label_agreement":null},{"id":"W2000821194","doi":"10.1177/1098214009340580","title":"Toward Accurate Measurement of Participation","year":2009,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":86,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Conceptualization; Operationalization; Citizen journalism; Stakeholder; Field (mathematics); Participatory evaluation; Psychology; Sociology; Management science; Computer science; Epistemology; Political science; Public relations; Social science; Artificial intelligence; Engineering; Mathematics","score_opus":0.4707113104953488,"score_gpt":0.5611236037354239,"score_spread":0.0904122932400751,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2000821194","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.064128764,0.0029902128,0.8594982,0.01758174,0.00077174813,0.0014899602,0.0005701922,0.00050561945,0.052463572],"genre_scores_gemma":[0.4866986,0.00227063,0.49996126,0.0027403398,0.00027463742,0.004670847,0.0005224606,0.000117695374,0.0027434567],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.79922456,0.13950288,0.01081099,0.011798757,0.035936866,0.0027259856],"domain_scores_gemma":[0.7364173,0.13909146,0.024665508,0.032026492,0.06439362,0.0034056455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.12904495,0.0010501967,0.0013805538,0.005572424,0.0024632437,0.007968769,0.0022867443,0.0037425251,0.0019502039],"category_scores_gemma":[0.25538138,0.0006870954,0.0006551579,0.005905854,0.006546078,0.016917355,0.013247183,0.005529729,0.0008727475],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016640409,0.00042706507,0.05150686,0.0017323539,0.00012743808,0.00008471746,0.028026538,0.0032806024,0.004387979,0.4711637,0.012317078,0.4267793],"study_design_scores_gemma":[0.00008981324,0.00088985526,0.05800097,0.0046851,0.00013651814,0.00033837513,0.027526962,0.018868204,0.010591007,0.7301467,0.14844984,0.00027667053],"about_ca_topic_score_codex":0.002586099,"about_ca_topic_score_gemma":0.0020700553,"teacher_disagreement_score":0.12904495,"about_ca_system_score_codex":0.0041024745,"about_ca_system_score_gemma":0.007649344,"threshold_uncertainty_score":0.68246305},"labels":[],"label_agreement":null},{"id":"W2014357589","doi":"10.1177/1098214011405311","title":"Legislator Uses of Public Performance Reports","year":2011,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Public Policy and Administration Research","field":"Social Sciences","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Victoria","funders":"","keywords":"Accountability; Legislator; Government (linguistics); Performance measurement; Dual (grammatical number); Key (lock); Public relations; Business; Public administration; Accounting; Political science; Computer science; Computer security; Marketing; Law; Legislation","score_opus":0.2353065547670905,"score_gpt":0.4390290283376902,"score_spread":0.2037224735705997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2014357589","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93072486,0.00079733215,0.0062051388,0.005783147,0.00010549952,0.00015556382,0.0017345652,0.00046085773,0.054033082],"genre_scores_gemma":[0.996416,0.0002382437,0.0009987619,0.00021938872,0.000033805518,0.00006093483,0.00036694302,0.00002533612,0.0016405098],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9213498,0.044879336,0.005270212,0.003529714,0.022701804,0.0022691141],"domain_scores_gemma":[0.71742195,0.13199867,0.070200615,0.028199222,0.04871816,0.0034614392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.039516293,0.00021187708,0.0003623622,0.003598166,0.0013854476,0.0043992274,0.0011569473,0.00058603234,0.0018728818],"category_scores_gemma":[0.17623831,0.00034003737,0.00035196773,0.004147693,0.0015276704,0.0017966682,0.0016522391,0.0011940821,0.00075200654],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027555338,0.00024139101,0.6699903,0.00044048997,0.0001942282,0.0002537285,0.052669708,0.0015256862,0.002409222,0.012832542,0.014388602,0.2447785],"study_design_scores_gemma":[0.00002756161,0.0005873028,0.8613518,0.0005436758,0.000115217845,0.00034213436,0.027996529,0.0036621653,0.006310034,0.0026326699,0.0962138,0.0002171824],"about_ca_topic_score_codex":0.017049719,"about_ca_topic_score_gemma":0.01546361,"teacher_disagreement_score":0.039516293,"about_ca_system_score_codex":0.0030767517,"about_ca_system_score_gemma":0.004575052,"threshold_uncertainty_score":0.20898461},"labels":[],"label_agreement":null},{"id":"W2032438539","doi":"10.1177/1098214009349792","title":"Exploring the Intervention— Context Interface","year":2009,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre Hospitalier de l’Université de Montréal; Université de Montréal","funders":"","keywords":"Sociotechnical system; Context (archaeology); Adaptation (eye); Workaround; Process (computing); Knowledge management; Computer science; Social network analysis; Psychology; Social media; World Wide Web","score_opus":0.8024925403175549,"score_gpt":0.7139293653214512,"score_spread":0.08856317499610367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032438539","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5277456,0.0022369286,0.1766034,0.02083708,0.000343953,0.004159653,0.0003942384,0.0004954281,0.26718375],"genre_scores_gemma":[0.9540952,0.00038761576,0.04019307,0.00091796735,0.000015858874,0.002079258,0.00005240169,0.00003519754,0.0022234896],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9768754,0.020229531,0.00029855318,0.0009760637,0.00081268465,0.0008077003],"domain_scores_gemma":[0.98033476,0.017631514,0.0005010969,0.00040235763,0.0005010281,0.0006292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014959271,0.00050442875,0.0005610331,0.0011794093,0.0031757953,0.005801539,0.0014716934,0.0022171743,0.012470864],"category_scores_gemma":[0.022352226,0.00041897138,0.0005658684,0.0010198215,0.0052907434,0.006044458,0.007243447,0.0018547093,0.00048540698],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009916791,0.0023442172,0.02479144,0.0038629177,0.00018247946,0.0021304977,0.30813584,0.0037149936,0.009706903,0.393081,0.0034647188,0.24759331],"study_design_scores_gemma":[0.0013027007,0.0040548523,0.03910056,0.0069161807,0.001007996,0.0016037344,0.37841246,0.020429758,0.01398432,0.26268896,0.27029702,0.0002014441],"about_ca_topic_score_codex":0.00313137,"about_ca_topic_score_gemma":0.0034018655,"teacher_disagreement_score":0.014959271,"about_ca_system_score_codex":0.004509752,"about_ca_system_score_gemma":0.0056584324,"threshold_uncertainty_score":0.079113185},"labels":[],"label_agreement":null},{"id":"W2033993956","doi":"10.1177/1098214013478146","title":"The Practice of Evaluation in Public Sector Contexts","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Accountability; Public sector; Citizen journalism; Perception; Diversity (politics); Reflection (computer programming); Sociology; Evaluation methods; Public relations; Public administration; Political science; Psychology; Law; Computer science; Engineering","score_opus":0.1817166479435985,"score_gpt":0.51942008015783,"score_spread":0.3377034322142315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2033993956","genre_codex":"methods","genre_gemma":"empirical","domain_codex":"methods","domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011397371,0.04191578,0.48853835,0.2120837,0.004523337,0.0013512343,0.000114064766,0.0005525623,0.23952349],"genre_scores_gemma":[0.65481865,0.015737018,0.29195583,0.019519318,0.0023848251,0.004506476,0.000068407324,0.00036420565,0.010645274],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.45873955,0.48654014,0.014821034,0.01322739,0.022835111,0.003836682],"domain_scores_gemma":[0.534387,0.3945962,0.01127451,0.03503078,0.021369586,0.0033419686],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.34253052,0.0015258008,0.0031845686,0.010709114,0.011508159,0.034748256,0.006107853,0.012931271,0.003956072],"category_scores_gemma":[0.25238717,0.0012931586,0.0014606398,0.010415365,0.15206508,0.03228252,0.016892163,0.015623447,0.0010621389],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002822009,0.000036343085,0.00055034354,0.000721532,0.000039750284,0.00012283937,0.020832933,0.0006421516,0.00012285894,0.9327463,0.0045725033,0.039584205],"study_design_scores_gemma":[0.00003760223,0.000073604606,0.00042856098,0.0032908851,0.000023730632,0.00019074656,0.014684823,0.0014608759,0.0005287577,0.863371,0.11584186,0.00006755743],"about_ca_topic_score_codex":0.0047945734,"about_ca_topic_score_gemma":0.0036711604,"teacher_disagreement_score":0.34253052,"about_ca_system_score_codex":0.021828573,"about_ca_system_score_gemma":0.027937314,"threshold_uncertainty_score":0.81077695},"labels":[],"label_agreement":null},{"id":"W2034211783","doi":"10.1177/1098214015578731","title":"Merging Developmental and Feminist Evaluation to Monitor and Evaluate Transformative Social Change","year":2015,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Transformative learning; Theory of change; Sociology; Social transformation; Social change; Monitoring and evaluation; Participatory evaluation; Program evaluation; Political science; Pedagogy; Social science; Public administration; Law","score_opus":0.3677671345545405,"score_gpt":0.5374349974415299,"score_spread":0.16966786288698937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034211783","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07481776,0.0073857903,0.43922716,0.08041549,0.0006592595,0.004797498,0.0004693532,0.00093039597,0.3912974],"genre_scores_gemma":[0.76412946,0.0019227411,0.21703957,0.00427318,0.00012573876,0.002068864,0.00014137512,0.0001871734,0.010111906],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.840982,0.121491,0.0034751035,0.0037283804,0.026141666,0.0041819285],"domain_scores_gemma":[0.8811569,0.05683735,0.0058691734,0.0068787644,0.044789832,0.004467917],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.13937679,0.0008133007,0.0008136886,0.0079120025,0.005194254,0.010816036,0.002562611,0.0012739822,0.0036130683],"category_scores_gemma":[0.09939618,0.00042419985,0.0004745936,0.004172122,0.014849733,0.0062524835,0.011013442,0.0028715325,0.00030314445],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018606934,0.00034301516,0.022383653,0.0012332582,0.000110012006,0.00013811383,0.030579425,0.004928321,0.0011914418,0.37043187,0.013122942,0.555352],"study_design_scores_gemma":[0.00031906477,0.0011404257,0.055934373,0.005716819,0.00027163228,0.00031927455,0.0806702,0.032504708,0.0149990935,0.3051062,0.50262076,0.00039743035],"about_ca_topic_score_codex":0.14516449,"about_ca_topic_score_gemma":0.22176486,"teacher_disagreement_score":0.86062324,"about_ca_system_score_codex":0.07248326,"about_ca_system_score_gemma":0.07787453,"threshold_uncertainty_score":0.7371037},"labels":[],"label_agreement":null},{"id":"W2034409422","doi":"10.1177/1098214008316655","title":"Cross-Disciplinarization: A New Talisman for Evaluation?","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Cross disciplinary; Discipline; Field (mathematics); Engineering ethics; Face (sociological concept); Management science; Sociology; Interdisciplinarity; Computer science; Data science; Social science; Engineering","score_opus":0.2784832306773482,"score_gpt":0.5746443419319662,"score_spread":0.296161111254618,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034409422","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004114204,0.055987626,0.14384504,0.7184181,0.0059564766,0.0005496954,0.00003103897,0.00023189976,0.07086591],"genre_scores_gemma":[0.58970565,0.041915353,0.19414859,0.14696112,0.0089409035,0.0044060973,0.00007467626,0.0006638358,0.013183791],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.5601331,0.38849798,0.010774117,0.008191075,0.027894458,0.0045093084],"domain_scores_gemma":[0.5856812,0.33552265,0.0089923,0.03403752,0.027710719,0.008055558],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3271042,0.0017551951,0.004173931,0.009414955,0.014062099,0.047096685,0.006636934,0.0152758695,0.0046007982],"category_scores_gemma":[0.28594816,0.0010711249,0.0021776606,0.009106935,0.1333352,0.07546074,0.032631565,0.030008594,0.0009263376],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058936894,0.00006824936,0.0005406922,0.000556172,0.00006417742,0.00008577467,0.018762466,0.0004504121,0.000059196183,0.90221417,0.013344125,0.0637955],"study_design_scores_gemma":[0.00004456188,0.00007636595,0.000268363,0.0026752953,0.000033696437,0.00015748074,0.015856408,0.001122381,0.00020703868,0.8963106,0.083192244,0.000055462595],"about_ca_topic_score_codex":0.005559109,"about_ca_topic_score_gemma":0.004887207,"teacher_disagreement_score":0.6728958,"about_ca_system_score_codex":0.029862046,"about_ca_system_score_gemma":0.03743245,"threshold_uncertainty_score":0.8298003},"labels":[],"label_agreement":null},{"id":"W2041989608","doi":"10.1177/1098214013487426","title":"The Complexity of Institutionalizing Evaluation as a Best Practice in North American Quitlines","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Centers for Disease Control and Prevention; National Institutes of Health","keywords":"Quitline; Thematic analysis; Institutionalisation; Qualitative research; Process (computing); Best practice; Process management; Program evaluation; Management science; Computer science; Psychology; Medicine; Sociology; Business; Intervention (counseling); Political science; Nursing; Engineering","score_opus":0.6067356275386427,"score_gpt":0.6851844173352069,"score_spread":0.07844878979656411,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2041989608","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.874904,0.0021333597,0.026834426,0.06891046,0.00019713347,0.0012363875,0.00003959146,0.00016291384,0.025581755],"genre_scores_gemma":[0.98187673,0.00050763454,0.014488523,0.0016112598,0.000027367221,0.0004585239,0.000016692942,0.000037288093,0.000975974],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.7494166,0.21661267,0.0065980926,0.0055328747,0.014984209,0.0068555865],"domain_scores_gemma":[0.80588514,0.13375567,0.01582186,0.01244069,0.020060686,0.0120359985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.14919092,0.00038659023,0.0006543626,0.0018251784,0.017171703,0.012185296,0.0033301439,0.0026386417,0.0017265527],"category_scores_gemma":[0.13381967,0.0009806359,0.00056484615,0.002043452,0.017888656,0.008623518,0.0152863525,0.0039760126,0.00014977327],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015166102,0.0009897291,0.056829616,0.0011446178,0.000089035915,0.0009861753,0.7195004,0.0019334124,0.0020516026,0.034069177,0.005226096,0.17702843],"study_design_scores_gemma":[0.00010699105,0.00083296315,0.051407356,0.002015153,0.000059674392,0.0003937113,0.8713008,0.003862091,0.0019304785,0.015803013,0.05209273,0.00019496071],"about_ca_topic_score_codex":0.025659028,"about_ca_topic_score_gemma":0.058884066,"teacher_disagreement_score":0.14919092,"about_ca_system_score_codex":0.04212878,"about_ca_system_score_gemma":0.0750586,"threshold_uncertainty_score":0.7890064},"labels":[],"label_agreement":null},{"id":"W2043887666","doi":"10.1177/1098214005278752","title":"Is Sustainability Possible? A Review and Commentary on Empirical Studies of Program Sustainability","year":2005,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":715,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Sustainability; Champion; Sustainability organizations; Empirical research; Program evaluation; Sustainability science; Social sustainability; Business; Environmental resource management; Political science; Economics; Public administration","score_opus":0.712236943529094,"score_gpt":0.7821963698676211,"score_spread":0.06995942633852714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2043887666","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000135166,0.94108564,0.00015858759,0.053254902,0.004318345,0.000017905551,0.000055179287,0.000006901454,0.00096727104],"genre_scores_gemma":[0.0058851326,0.95149875,0.0004953026,0.037122007,0.0043378426,0.00014306956,0.00006464641,0.00001824022,0.0004350924],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9571732,0.022161355,0.006279396,0.0025450464,0.0109714065,0.00086957903],"domain_scores_gemma":[0.6365977,0.31680343,0.012474363,0.0030820465,0.0297127,0.0013297766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.047664847,0.001465676,0.005013492,0.009062669,0.00166231,0.004907587,0.005507357,0.00648641,0.0041802465],"category_scores_gemma":[0.20532985,0.00084131013,0.002834513,0.01742792,0.0070571965,0.0070988005,0.0024228226,0.0065295855,0.0011240832],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002663283,0.00004963698,0.0006842408,0.21645097,0.0008997972,0.0004773735,0.0029616922,0.00042370462,0.00019484671,0.02724699,0.4412705,0.30907387],"study_design_scores_gemma":[0.000084557425,0.00008503868,0.0016980264,0.34365338,0.0009639572,0.00040453856,0.0032907845,0.00013699733,0.00021266007,0.008895287,0.64050967,0.000065019914],"about_ca_topic_score_codex":0.027676128,"about_ca_topic_score_gemma":0.04067196,"teacher_disagreement_score":0.047664847,"about_ca_system_score_codex":0.012584332,"about_ca_system_score_gemma":0.03787336,"threshold_uncertainty_score":0.25207883},"labels":[],"label_agreement":null},{"id":"W2046660743","doi":"10.1177/1098214008327931","title":"Do Self-Assessments Work to Detect Workshop Success?","year":2009,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Human Resource Development and Performance Evaluation","field":"Psychology","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Self-assessment; Applied psychology; Psychology; Work (physics); Multilevel model; Self-report study; Computer science; Social psychology; Machine learning; Engineering","score_opus":0.042287017239091264,"score_gpt":0.41814185392178405,"score_spread":0.37585483668269276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046660743","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7259948,0.017216768,0.14746504,0.050530776,0.005926317,0.0010911087,0.0021271072,0.001160264,0.048487876],"genre_scores_gemma":[0.9659453,0.0016247773,0.025462205,0.0036542485,0.00033992238,0.00067847164,0.00028247837,0.000058898342,0.0019537627],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.93359274,0.04304506,0.006245525,0.0034607095,0.012541207,0.0011148026],"domain_scores_gemma":[0.72491807,0.19252059,0.036858246,0.011361409,0.031923655,0.0024181257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07119654,0.00065932644,0.0009960586,0.0037044007,0.0006668904,0.0031207425,0.0020246347,0.0021513125,0.001194813],"category_scores_gemma":[0.24862455,0.00041702882,0.0009648716,0.0020807944,0.0023492174,0.0045917816,0.0017900749,0.0021279298,0.0010170131],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052218285,0.00032044598,0.68237245,0.0013242592,0.0006326722,0.000093504896,0.006867929,0.0006157076,0.0006349655,0.0045467503,0.0139230145,0.28814608],"study_design_scores_gemma":[0.00018899639,0.0020215583,0.89314944,0.004440474,0.00065096613,0.0008520141,0.01726391,0.011855122,0.0074690995,0.025158018,0.03655847,0.00039184425],"about_ca_topic_score_codex":0.002584303,"about_ca_topic_score_gemma":0.0051640673,"teacher_disagreement_score":0.07119654,"about_ca_system_score_codex":0.0009889831,"about_ca_system_score_gemma":0.001189361,"threshold_uncertainty_score":0.3765278},"labels":[],"label_agreement":null},{"id":"W2053636313","doi":"10.1177/1098214008319012","title":"The Case of Top Beginnings and the Missing Child Outcomes","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Education and Early Childhood Development","funders":"","keywords":"Formative assessment; Work (physics); Set (abstract data type); Foundation (evidence); Program evaluation; Political science; Public relations; Psychology; Management science; Engineering ethics; Pedagogy; Public administration; Engineering; Computer science","score_opus":0.023099755106382413,"score_gpt":0.35318993430323464,"score_spread":0.33009017919685224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053636313","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37501332,0.002087054,0.013638664,0.3675863,0.000817497,0.0005213179,0.00057266135,0.000075607946,0.2396876],"genre_scores_gemma":[0.9618058,0.00096172,0.0072511365,0.0137191275,0.00014084845,0.00044932455,0.00007732291,0.000032128733,0.015562652],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9734735,0.016530402,0.00052281807,0.0011664669,0.0030185024,0.0052883034],"domain_scores_gemma":[0.97686994,0.012812533,0.0022930054,0.0012289232,0.0021055029,0.0046900506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020036375,0.0005283572,0.0010361095,0.0018098562,0.02224583,0.0062515037,0.001977219,0.007826986,0.006260536],"category_scores_gemma":[0.028562075,0.00062731595,0.0013284759,0.002188547,0.012711171,0.0067116315,0.00884789,0.0112326965,0.00048362082],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033371587,0.00056103274,0.036249425,0.0003676587,0.000077631776,0.06024033,0.20778218,0.001235971,0.000644872,0.5998783,0.05089326,0.041735657],"study_design_scores_gemma":[0.00022075631,0.0005964159,0.02962264,0.0019653805,0.00020837021,0.028940855,0.4994011,0.0026303986,0.0020275733,0.14296691,0.2911792,0.0002404387],"about_ca_topic_score_codex":0.07082334,"about_ca_topic_score_gemma":0.11209501,"teacher_disagreement_score":0.07082334,"about_ca_system_score_codex":0.010919155,"about_ca_system_score_gemma":0.01485117,"threshold_uncertainty_score":0.14082223},"labels":[],"label_agreement":null},{"id":"W2060998986","doi":"10.1177/1098214006287990","title":"Developing a Stakeholder-Driven Anticipated Timeline of Impact for Evaluation of Social Programs","year":2006,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Timeline; Stakeholder; Stakeholder engagement; Stakeholder analysis; Program evaluation; Process (computing); Process management; Computer science; Management science; Public relations; Business; Political science; Engineering; Geography","score_opus":0.47865420615752025,"score_gpt":0.5785712379293496,"score_spread":0.09991703177182937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2060998986","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018989965,0.0001570164,0.9613044,0.00068745686,0.00008330184,0.0019157925,0.0005085217,0.0008184149,0.015534986],"genre_scores_gemma":[0.09838486,0.00010752091,0.89745474,0.000084190586,0.000011379232,0.002569891,0.00031154277,0.00012219438,0.00095373776],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9725398,0.018683542,0.0015980289,0.0010221782,0.005680006,0.00047632866],"domain_scores_gemma":[0.9228941,0.040437464,0.007051401,0.0043652086,0.023994377,0.0012574648],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.034804657,0.0014239325,0.00067624426,0.005121,0.0014334584,0.0036173228,0.0015596293,0.001140809,0.0050509297],"category_scores_gemma":[0.073782176,0.0007407424,0.0008319034,0.0032453602,0.001048645,0.004985571,0.0022283928,0.002219961,0.000969056],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065211894,0.0005331055,0.014181756,0.0019966555,0.00016117834,0.0003135066,0.011554842,0.056096744,0.013482883,0.14537644,0.012793837,0.74285686],"study_design_scores_gemma":[0.0007920281,0.004814674,0.034229487,0.0033920463,0.00031539603,0.00077759346,0.019357536,0.41464618,0.0698073,0.23119988,0.21956101,0.0011069241],"about_ca_topic_score_codex":0.004698495,"about_ca_topic_score_gemma":0.008434633,"teacher_disagreement_score":0.96519536,"about_ca_system_score_codex":0.0049057454,"about_ca_system_score_gemma":0.0069001154,"threshold_uncertainty_score":0.18406683},"labels":[],"label_agreement":null},{"id":"W2061653962","doi":"10.1177/1098214007307942","title":"Evaluations That Consider the Cost of Educational Programs","year":2007,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"School Choice and Performance","field":"Social Sciences","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Program evaluation; Cost effectiveness; Set (abstract data type); Cost–benefit analysis; Computer science; Cost estimate; Macro; Program Design Language; Actuarial science; Cost contingency; Management science; Relevant cost; Risk analysis (engineering); Economics; Business","score_opus":0.10349334241685199,"score_gpt":0.4664295884873363,"score_spread":0.3629362460704843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061653962","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.189585,0.28183424,0.11092575,0.030648127,0.009780872,0.027660497,0.021830602,0.00076463615,0.32697022],"genre_scores_gemma":[0.8864071,0.04200588,0.04991212,0.004180188,0.0012448156,0.008365535,0.0024568946,0.00019235794,0.005235195],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.8281466,0.124128416,0.012429426,0.001994535,0.031757545,0.001543563],"domain_scores_gemma":[0.5234043,0.4207718,0.022873363,0.006427344,0.0243072,0.002216029],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07197724,0.001717571,0.0031515183,0.0116034085,0.0007077509,0.006687449,0.0016228418,0.0027912068,0.009782809],"category_scores_gemma":[0.39140338,0.00065057434,0.0048770765,0.011174616,0.0016649892,0.007130791,0.0022067535,0.0031360176,0.00064821943],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.012236559,0.0018994017,0.03439082,0.05075409,0.02755006,0.00035210414,0.00097610004,0.041164573,0.0007659221,0.08799132,0.031887185,0.71003187],"study_design_scores_gemma":[0.0112749105,0.02204353,0.12480087,0.10312262,0.10180347,0.0017999453,0.0044021895,0.055846464,0.011084125,0.21416637,0.34829637,0.0013591722],"about_ca_topic_score_codex":0.0035993082,"about_ca_topic_score_gemma":0.0045965374,"teacher_disagreement_score":0.07197724,"about_ca_system_score_codex":0.00687873,"about_ca_system_score_gemma":0.0053858375,"threshold_uncertainty_score":0.3806566},"labels":[],"label_agreement":null},{"id":"W2063290014","doi":"10.1177/1098214012464426","title":"Improving Program Results Through the Use of Predictive Operational Performance Indicators","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Employment and Social Development Canada; Carleton University","funders":"","keywords":"Accountability; Context (archaeology); Performance indicator; Program evaluation; Process management; Quality (philosophy); Computer science; Risk analysis (engineering); Term (time); Operations management; Environmental economics; Business; Engineering; Marketing; Economics; Political science; Public administration","score_opus":0.1859485882753664,"score_gpt":0.471487866355326,"score_spread":0.2855392780799596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063290014","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4325708,0.0026470511,0.3632582,0.006672709,0.00035872197,0.007075882,0.006767347,0.006198488,0.17445086],"genre_scores_gemma":[0.8206455,0.0010855318,0.17271441,0.0003065354,0.0000459121,0.001261934,0.0014606102,0.00015144155,0.0023281472],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.963301,0.019579174,0.0027567858,0.001570284,0.011532848,0.0012599092],"domain_scores_gemma":[0.86468655,0.05630993,0.030488476,0.0075713457,0.038447406,0.0024962365],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05813128,0.0015526925,0.00077392894,0.009569635,0.0012617115,0.007123193,0.0014895174,0.00049430516,0.0016759768],"category_scores_gemma":[0.11833154,0.00040093795,0.0005368164,0.008525653,0.0009858073,0.004599698,0.0030656953,0.0015844406,0.0004939375],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057195104,0.0013994392,0.19988959,0.0012941328,0.00024382795,0.000086659245,0.0053075785,0.021661269,0.0026883984,0.017068116,0.012689592,0.73709947],"study_design_scores_gemma":[0.00022960085,0.005013559,0.63776654,0.0048106443,0.0007745132,0.0002304555,0.0148402145,0.18303044,0.034007538,0.033828698,0.08462436,0.00084345136],"about_ca_topic_score_codex":0.05525202,"about_ca_topic_score_gemma":0.07204926,"teacher_disagreement_score":0.05813128,"about_ca_system_score_codex":0.00949736,"about_ca_system_score_gemma":0.018798599,"threshold_uncertainty_score":0.30743128},"labels":[],"label_agreement":null},{"id":"W2063748659","doi":"10.1016/s1098-2140(00)00090-4","title":"Planning for community-based evaluation","year":2000,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Centre for Global Health Research","funders":"","keywords":"Management science; Program evaluation; Set (abstract data type); Unit (ring theory); Process (computing); Conflict resolution; Evaluation methods; Process management; Computer science; Psychology; Sociology; Political science; Business; Engineering; Mathematics education","score_opus":0.355370496696108,"score_gpt":0.5926022913441982,"score_spread":0.2372317946480902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063748659","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09561433,0.004650143,0.28533685,0.16340256,0.0012464517,0.019777441,0.00085892103,0.0014391534,0.4276741],"genre_scores_gemma":[0.70009524,0.0015762532,0.25986788,0.0045678327,0.00030737865,0.00725451,0.00087409676,0.00022074065,0.025236195],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9593432,0.026996182,0.0012299784,0.0014940768,0.005663515,0.0052730744],"domain_scores_gemma":[0.9273653,0.020807197,0.0037497182,0.00246478,0.023839168,0.021773856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.035415582,0.0010598227,0.000763468,0.0056887544,0.014169156,0.014519434,0.0045638797,0.0076410817,0.03283267],"category_scores_gemma":[0.100670785,0.0010138742,0.0011781197,0.004751764,0.0037484984,0.0100813545,0.008961415,0.006964405,0.0030075912],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035565544,0.003503864,0.025155583,0.0010815543,0.00014409616,0.001704336,0.006809134,0.027121916,0.0010162559,0.34450155,0.17706072,0.4115453],"study_design_scores_gemma":[0.00042404915,0.0011332348,0.022916881,0.002583122,0.00011619333,0.00074494234,0.042153962,0.0478591,0.0018148571,0.5959497,0.28399944,0.00030459967],"about_ca_topic_score_codex":0.057775334,"about_ca_topic_score_gemma":0.15781619,"teacher_disagreement_score":0.057775334,"about_ca_system_score_codex":0.023084732,"about_ca_system_score_gemma":0.117321245,"threshold_uncertainty_score":0.1872977},"labels":[],"label_agreement":null},{"id":"W2063922189","doi":"10.1177/1098214007304536","title":"Analysis of Thin Online Interview Data","year":2007,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Focus Groups and Qualitative Methods","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Meaning (existential); Computer science; Context (archaeology); Qualitative property; Perception; Data collection; Semantics (computer science); Data science; Qualitative analysis; Qualitative research; Psychology; Sociology; Machine learning","score_opus":0.34668716451748016,"score_gpt":0.5830954045773991,"score_spread":0.2364082400599189,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063922189","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4821796,0.0008030158,0.3841838,0.0035135187,0.00057841657,0.040375248,0.030376881,0.0012956039,0.05669388],"genre_scores_gemma":[0.50665534,0.0008213947,0.37467813,0.0016951107,0.00016065671,0.08352979,0.015958978,0.0008277086,0.015672907],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.94344413,0.035499245,0.006111395,0.0031965538,0.010337129,0.0014115623],"domain_scores_gemma":[0.7354707,0.16419882,0.017293362,0.024998879,0.05621768,0.0018205493],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04955923,0.0008848597,0.0010414147,0.006723556,0.0031451883,0.0024813404,0.0019061799,0.00090340077,0.012136119],"category_scores_gemma":[0.16789438,0.0007579456,0.00046858506,0.0070974636,0.0024539405,0.0027945084,0.0049520843,0.0018120575,0.0027261802],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012240498,0.0007026765,0.035118043,0.007702193,0.00014079553,0.002539048,0.3773907,0.0022647032,0.02557844,0.04228316,0.032464568,0.4725916],"study_design_scores_gemma":[0.00022902452,0.0010997177,0.065472096,0.0076311626,0.00013318968,0.0017593075,0.41277647,0.015509209,0.028645348,0.069920674,0.39653727,0.00028651473],"about_ca_topic_score_codex":0.002268156,"about_ca_topic_score_gemma":0.0036238492,"teacher_disagreement_score":0.95044076,"about_ca_system_score_codex":0.0031785031,"about_ca_system_score_gemma":0.005330491,"threshold_uncertainty_score":0.26209742},"labels":[],"label_agreement":null},{"id":"W2070663696","doi":"10.1177/1098214010371817","title":"A Realist Evaluation Approach to Unpacking the Impacts of the Sentencing Guidelines","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Criminal Justice and Corrections Analysis","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; St. Michael's Hospital","funders":"","keywords":"Unpacking; Prison; Sentencing guidelines; Work (physics); Psychological intervention; Resource (disambiguation); Impact evaluation; Key (lock); Control (management); Linkage (software); Public economics; Sociology; Management science; Criminology; Political science; Psychology; Economics; Computer science; Computer security","score_opus":0.08324908097625391,"score_gpt":0.42450876934372395,"score_spread":0.34125968836747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070663696","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07558848,0.004276105,0.75145257,0.029236484,0.001053528,0.010082519,0.0008339921,0.00041552153,0.12706077],"genre_scores_gemma":[0.72609586,0.0016669205,0.25586104,0.0020919594,0.0002619608,0.010316905,0.00013530783,0.00008694157,0.0034831243],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.7963806,0.182933,0.0036569778,0.0038911696,0.011345575,0.0017927218],"domain_scores_gemma":[0.7888367,0.18399839,0.009783783,0.008729085,0.007505709,0.0011463502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1246737,0.00189923,0.0024523851,0.0059081,0.002642897,0.009041074,0.002972827,0.0027980905,0.008703293],"category_scores_gemma":[0.18415324,0.0010143849,0.0018338033,0.0029463952,0.012773887,0.0113704745,0.0053289086,0.00477653,0.00032370363],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005399717,0.0007182433,0.005145205,0.0030776795,0.00094106916,0.00021644593,0.0044399663,0.029946717,0.0006411594,0.8609806,0.0028060086,0.090546936],"study_design_scores_gemma":[0.00074119715,0.0029345811,0.0076490175,0.0022533862,0.0008524348,0.00014159971,0.0057444773,0.07195447,0.0021842849,0.87117684,0.03414981,0.00021789857],"about_ca_topic_score_codex":0.005303892,"about_ca_topic_score_gemma":0.007551192,"teacher_disagreement_score":0.1246737,"about_ca_system_score_codex":0.013628431,"about_ca_system_score_gemma":0.016846746,"threshold_uncertainty_score":0.65934545},"labels":[],"label_agreement":null},{"id":"W2073141549","doi":"10.1177/1098214007312630","title":"Using Self-Assessments to Detect Workshop Success","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Communication in Education and Healthcare","field":"Psychology","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Alberta Health Services; University of British Columbia; University of Calgary; University of Saskatchewan","funders":"","keywords":"Psychology; Reliability (semiconductor); Applied psychology; Program evaluation; Scale (ratio); Gold standard (test); Medical education; Medicine; Statistics","score_opus":0.2077316628935172,"score_gpt":0.5490251839928058,"score_spread":0.34129352109928857,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073141549","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96569324,0.00036325504,0.021618301,0.00015618787,0.00009625824,0.0013191173,0.0010507045,0.00039811197,0.009304901],"genre_scores_gemma":[0.96751326,0.00033153748,0.025111811,0.000107984226,0.00007575933,0.0019273572,0.0015303325,0.000057119705,0.003344715],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98915124,0.004466258,0.0016480051,0.0008204754,0.003627996,0.00028592514],"domain_scores_gemma":[0.9420525,0.023605084,0.0130197145,0.0038943356,0.015711693,0.0017167913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012436915,0.0006488789,0.0005104873,0.002369971,0.00037097055,0.0009779796,0.0006853293,0.00049064594,0.0016948404],"category_scores_gemma":[0.039331242,0.0002533481,0.0005235733,0.0008335364,0.0003641522,0.0010067228,0.00096894527,0.00072184653,0.0012114224],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090816955,0.0013856953,0.8139325,0.0004976636,0.00027713802,0.00014486503,0.0036409693,0.00085370947,0.00731396,0.00026351344,0.0028121695,0.16796966],"study_design_scores_gemma":[0.000092164715,0.0043093953,0.97112435,0.00017772806,0.000101210004,0.00036600488,0.002599628,0.005837168,0.010202877,0.000503251,0.0045892894,0.00009699782],"about_ca_topic_score_codex":0.0005822537,"about_ca_topic_score_gemma":0.0010571203,"teacher_disagreement_score":0.012436915,"about_ca_system_score_codex":0.00024308423,"about_ca_system_score_gemma":0.0004481417,"threshold_uncertainty_score":0.06577349},"labels":[],"label_agreement":null},{"id":"W2093411079","doi":"10.1177/1098214010379038","title":"Evaluating the Science of Discovery in Complex Health Systems","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Interdisciplinary Research and Collaboration","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Nutrasource; University of Toronto","funders":"","keywords":"Process (computing); Discipline; Plan (archaeology); Data science; Work (physics); Engineering ethics; Computer science; Scientific discovery; Management science; Logic model; Translational science; Health science; Sociology; Psychology; Engineering; Medicine; Social science; Medical education","score_opus":0.2891501171514995,"score_gpt":0.5838757997116935,"score_spread":0.29472568256019394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2093411079","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6285566,0.014463099,0.24778342,0.024313714,0.000682074,0.0071897376,0.0009884919,0.0002538754,0.07576893],"genre_scores_gemma":[0.90829045,0.0022277182,0.08634306,0.0005487455,0.00011392577,0.0017509607,0.00015253125,0.000021461434,0.0005511542],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.65728897,0.31189826,0.006916304,0.003412491,0.018467916,0.0020159923],"domain_scores_gemma":[0.37114874,0.5841909,0.017356886,0.009757331,0.014112458,0.003433753],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.17494297,0.0012115897,0.0023186256,0.0065006316,0.0032606155,0.011385397,0.0017679781,0.0023780207,0.0029027478],"category_scores_gemma":[0.3850008,0.0005507854,0.0015600118,0.006591642,0.009825239,0.009418487,0.0075366013,0.0026186025,0.00018859397],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024567067,0.0022264326,0.08196265,0.0058311983,0.0028920698,0.00038435953,0.0060332217,0.27802134,0.0010543878,0.35666558,0.0037360087,0.25873604],"study_design_scores_gemma":[0.0006542319,0.008235364,0.023106763,0.0020910865,0.0011257767,0.00023902513,0.010887673,0.36922985,0.0033233983,0.5690762,0.011760826,0.00026988043],"about_ca_topic_score_codex":0.005463939,"about_ca_topic_score_gemma":0.004859953,"teacher_disagreement_score":0.825057,"about_ca_system_score_codex":0.0142338015,"about_ca_system_score_gemma":0.015280352,"threshold_uncertainty_score":0.9251979},"labels":[],"label_agreement":null},{"id":"W2097114336","doi":"10.1177/1098214010378355","title":"Evaluating Capacity Building for Policy Research Organizations","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"International Development and Aid","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"International Development Research Centre","keywords":"Capacity building; Legislation; Business; Key (lock); Corporate governance; Process management; Program evaluation; Public relations; Economics; Political science; Computer science; Economic growth; Public administration; Computer security; Finance","score_opus":0.1885404601512432,"score_gpt":0.5539368367300723,"score_spread":0.3653963765788291,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097114336","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63415706,0.0045656054,0.11225803,0.018698923,0.0010857908,0.058275297,0.0014868476,0.00077275065,0.16869973],"genre_scores_gemma":[0.864397,0.0012588821,0.09908035,0.00094673625,0.00018517688,0.030664658,0.0005509839,0.0000532349,0.0028629343],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.5385078,0.4070103,0.013247776,0.0051857517,0.02511694,0.010931443],"domain_scores_gemma":[0.3303434,0.50161296,0.044436183,0.028830709,0.072739735,0.022037022],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.38445577,0.0016811931,0.0015314012,0.008924828,0.0037937097,0.008867854,0.0035838322,0.0035435571,0.0075016934],"category_scores_gemma":[0.4514221,0.0008325402,0.0018592019,0.0055371337,0.0039932546,0.009289513,0.013503668,0.0029974382,0.00087892555],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006413781,0.0094513055,0.084183715,0.007866802,0.0022436436,0.00033960008,0.01845968,0.052081835,0.002472569,0.12836175,0.012965601,0.6751598],"study_design_scores_gemma":[0.011862399,0.07764861,0.19379115,0.022702957,0.0069042025,0.00051001797,0.0933542,0.12993832,0.04000301,0.2559446,0.16616489,0.0011756021],"about_ca_topic_score_codex":0.002288143,"about_ca_topic_score_gemma":0.0030076755,"teacher_disagreement_score":0.6155442,"about_ca_system_score_codex":0.0204154,"about_ca_system_score_gemma":0.031258363,"threshold_uncertainty_score":0.75907564},"labels":[],"label_agreement":null},{"id":"W2102960304","doi":"10.1177/1098214012464037","title":"Arguments for a Common Set of Principles for Collaborative Inquiry in Evaluation","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":113,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Carleton University; University of Ottawa","funders":"","keywords":"Logic model; Set (abstract data type); Context (archaeology); Field (mathematics); Stakeholder; Program evaluation; Management science; Engineering ethics; Sociology; Computer science; Epistemology; Knowledge management; Public relations; Political science; Social science; Public administration; Engineering","score_opus":0.39253977170837995,"score_gpt":0.5662179449091042,"score_spread":0.1736781732007242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102960304","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029092615,0.0026396455,0.78865635,0.11855148,0.0009607204,0.0012405713,0.000096071984,0.00034706647,0.08459885],"genre_scores_gemma":[0.2712866,0.0021305142,0.6895554,0.021316571,0.0010140915,0.008652979,0.00015951655,0.00039176774,0.005492499],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.653598,0.23668715,0.023369972,0.01930781,0.061435148,0.0056018946],"domain_scores_gemma":[0.6968179,0.21434493,0.010473255,0.04115148,0.031270288,0.0059421356],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3024054,0.002648088,0.005341647,0.009153823,0.01660413,0.03869847,0.013435098,0.029608075,0.007384952],"category_scores_gemma":[0.23804352,0.0026459417,0.006060009,0.007560718,0.14169532,0.054811884,0.028114997,0.035474144,0.0031853241],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009428871,0.000021035194,0.00005940397,0.000103843326,0.000012538738,0.000019961071,0.002033072,0.00023596987,0.00002057937,0.99409544,0.00091309566,0.0024756247],"study_design_scores_gemma":[0.000050206123,0.00002049205,0.000053584536,0.00029001915,0.00000860828,0.00004477476,0.0007774134,0.000804695,0.00008450032,0.98587996,0.011966064,0.00001973255],"about_ca_topic_score_codex":0.0044825273,"about_ca_topic_score_gemma":0.0028709907,"teacher_disagreement_score":0.3024054,"about_ca_system_score_codex":0.019950347,"about_ca_system_score_gemma":0.027325708,"threshold_uncertainty_score":0.86025834},"labels":[],"label_agreement":null},{"id":"W2120292430","doi":"10.1177/1098214012440030","title":"A New Realistic Evaluation Analysis Method","year":2012,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":144,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Public Health Ontario; University of Toronto","funders":"Health Research Board","keywords":"Coding (social sciences); Computer science; Qualitative research; Qualitative analysis; Narrative; Management science; Context (archaeology); Evaluation methods; Data science; Psychology; Sociology; Social science; Engineering","score_opus":0.29742318158161135,"score_gpt":0.6135208376358036,"score_spread":0.3160976560541922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120292430","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00711156,0.00020321546,0.9453391,0.000963445,0.00031175173,0.0059829475,0.00063282053,0.0005342365,0.038920928],"genre_scores_gemma":[0.061658025,0.00017076069,0.91283685,0.0002512664,0.000063313615,0.016809078,0.00044335081,0.00032251983,0.007444848],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.8707089,0.09604144,0.008519234,0.009061929,0.014349238,0.0013192858],"domain_scores_gemma":[0.8829531,0.06641935,0.004977448,0.01471382,0.0296535,0.0012827228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.069573626,0.0014923605,0.0012226688,0.008029932,0.0039070887,0.010311433,0.002786169,0.0017814755,0.023193695],"category_scores_gemma":[0.15073581,0.0010192241,0.0015083776,0.0068026553,0.0041712266,0.0080764,0.0072990702,0.0034116697,0.003351172],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038477714,0.00025149735,0.0029141265,0.0016551182,0.00014594175,0.00021036727,0.0267929,0.004322817,0.0033873164,0.43147826,0.015792899,0.5126641],"study_design_scores_gemma":[0.0004987163,0.0008429159,0.0054448624,0.0024429413,0.00021989952,0.00082820404,0.024840578,0.07299381,0.009699829,0.3805006,0.50134367,0.0003438703],"about_ca_topic_score_codex":0.0027998271,"about_ca_topic_score_gemma":0.0033972354,"teacher_disagreement_score":0.069573626,"about_ca_system_score_codex":0.008372096,"about_ca_system_score_gemma":0.010421578,"threshold_uncertainty_score":0.3679449},"labels":[],"label_agreement":null},{"id":"W2140628270","doi":"10.1177/1098214014532166","title":"How Analogue Research Can Advance Descriptive Evaluation Theory","year":2014,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Interpersonal communication; Frame (networking); Computer science; Management science; Accountability; Evaluation methods; Epistemology; Psychology; Social psychology; Political science","score_opus":0.3507628690464508,"score_gpt":0.5660418537784816,"score_spread":0.21527898473203078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140628270","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020795817,0.01139256,0.6458919,0.030542381,0.0026117153,0.0014439652,0.0002829244,0.00036534687,0.2866733],"genre_scores_gemma":[0.6218798,0.007308126,0.35200328,0.005404422,0.0018078481,0.0029293324,0.00018484259,0.00021891209,0.008263374],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9073715,0.07821893,0.0026321106,0.0034185462,0.0073933965,0.00096539676],"domain_scores_gemma":[0.67688507,0.27891162,0.00533118,0.025638578,0.011592434,0.0016410842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08524376,0.0013364136,0.0017620643,0.006988388,0.0023867835,0.011655115,0.0035945964,0.0033303006,0.0153315645],"category_scores_gemma":[0.21895865,0.00096199394,0.001317375,0.0048539774,0.028487947,0.026939623,0.00790206,0.0075792996,0.0015543876],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056257737,0.00007432199,0.00042550924,0.00029684888,0.000021436439,0.000063092986,0.0023353014,0.0009471654,0.00008638779,0.97121465,0.0010699832,0.02340902],"study_design_scores_gemma":[0.000044920922,0.000061717816,0.00021876524,0.00034060696,0.000010654819,0.00005518501,0.00093736977,0.0018764908,0.00017840973,0.9781317,0.01811679,0.000027503665],"about_ca_topic_score_codex":0.0022391777,"about_ca_topic_score_gemma":0.0017745602,"teacher_disagreement_score":0.08524376,"about_ca_system_score_codex":0.0055491063,"about_ca_system_score_gemma":0.004701214,"threshold_uncertainty_score":0.45081747},"labels":[],"label_agreement":null},{"id":"W2147607381","doi":"10.1177/1098214007309280","title":"The Evaluation of Large Research Initiatives","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":144,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Program evaluation; Agency (philosophy); Government (linguistics); Work (physics); Political science; Management science; Public relations; Public administration; Sociology; Engineering; Social science","score_opus":0.8816684527440397,"score_gpt":0.727468180018336,"score_spread":0.1542002727257037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2147607381","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7063445,0.014783351,0.05792497,0.019530876,0.0013841082,0.0283512,0.002609863,0.0017820799,0.16728903],"genre_scores_gemma":[0.90006304,0.002928243,0.078293025,0.0014007032,0.0005232908,0.011669555,0.0012996595,0.00021097639,0.0036114464],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.4545095,0.40344876,0.02540201,0.012557779,0.098178424,0.005903495],"domain_scores_gemma":[0.25431406,0.51560307,0.04259395,0.045316815,0.1319751,0.010197001],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.36134726,0.0017855462,0.0024292078,0.02055239,0.005477903,0.016714787,0.0036273545,0.0022487328,0.0055206297],"category_scores_gemma":[0.5048438,0.0007501174,0.0011117503,0.022514656,0.005659251,0.012080372,0.0131008485,0.0020912874,0.0010087212],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021461872,0.0026326424,0.06772961,0.00753504,0.00152378,0.00040250568,0.027680939,0.0048174956,0.003534869,0.053502772,0.016172161,0.8123221],"study_design_scores_gemma":[0.0040829703,0.023065165,0.33048782,0.014588466,0.0035340674,0.0006938296,0.15136606,0.028741438,0.026282603,0.12168677,0.29448792,0.0009828226],"about_ca_topic_score_codex":0.005389399,"about_ca_topic_score_gemma":0.0043397215,"teacher_disagreement_score":0.9794476,"about_ca_system_score_codex":0.022760635,"about_ca_system_score_gemma":0.039493117,"threshold_uncertainty_score":0.78757256},"labels":[],"label_agreement":null},{"id":"W2149719915","doi":"10.1177/1098214008325023","title":"An Assessment of the Theoretical Underpinnings of Practical Participatory Evaluation","year":2008,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":63,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Context (archaeology); Process (computing); Citizen journalism; Key (lock); Management science; Participatory action research; Knowledge management; Empirical research; Action (physics); Computer science; Epistemology; Psychology; Sociology; Engineering","score_opus":0.36844893795092876,"score_gpt":0.6248471033074683,"score_spread":0.25639816535653953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149719915","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039394006,0.02084627,0.60567015,0.069545,0.0005850194,0.0021802313,0.00011314273,0.00017504161,0.26149118],"genre_scores_gemma":[0.8407024,0.008396662,0.14255393,0.002192306,0.00028542944,0.002949103,0.000088310764,0.00006198672,0.0027697242],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.92017317,0.060504965,0.0023900876,0.0023479038,0.013051433,0.0015324331],"domain_scores_gemma":[0.75570333,0.21803387,0.0061099515,0.0091407625,0.00985541,0.0011566576],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.111405835,0.0010797334,0.0012347085,0.0057495134,0.0045651514,0.011594832,0.00305133,0.004607562,0.0061843502],"category_scores_gemma":[0.15683654,0.0010033743,0.0011251587,0.004205174,0.031364474,0.015670583,0.0080948975,0.0050195013,0.000549591],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024147635,0.00004398696,0.00069004716,0.0004410531,0.000010630439,0.000052293923,0.0024330858,0.0017399411,0.00006450642,0.96543324,0.00036061683,0.028706547],"study_design_scores_gemma":[0.000036943442,0.00010595861,0.0011639692,0.0020782205,0.000018393417,0.00018231604,0.004145402,0.009500822,0.00033604284,0.9570967,0.02529908,0.00003609853],"about_ca_topic_score_codex":0.0021969236,"about_ca_topic_score_gemma":0.0018410427,"teacher_disagreement_score":0.88859415,"about_ca_system_score_codex":0.009636865,"about_ca_system_score_gemma":0.010250106,"threshold_uncertainty_score":0.58917737},"labels":[],"label_agreement":null},{"id":"W2156859842","doi":"10.1177/1098214014535658","title":"The Ethical Tipping Points of Evaluators in Conflict Zones","year":2014,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"International Development Research Centre","funders":"","keywords":"Affect (linguistics); Section (typography); Ethical issues; Face (sociological concept); Psychology; Engineering ethics; Conflict of interest; Sociology; Social psychology; Public relations; Political science; Law; Social science; Computer science","score_opus":0.17093914735823218,"score_gpt":0.5305063413621447,"score_spread":0.3595671940039125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2156859842","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41165656,0.006306849,0.16952486,0.23575334,0.0020362749,0.000883845,0.00006600906,0.000342691,0.17342962],"genre_scores_gemma":[0.97113377,0.00069869024,0.014834713,0.009725428,0.00021401432,0.00041705027,0.000013600055,0.00012504208,0.0028376093],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.5120551,0.43410945,0.010586588,0.0077235093,0.024543902,0.010981429],"domain_scores_gemma":[0.5772742,0.3266524,0.03125395,0.017335879,0.03298106,0.014502513],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.21150197,0.00091222516,0.001324825,0.00341084,0.019745559,0.029953262,0.0037191138,0.008961262,0.0035170373],"category_scores_gemma":[0.43639562,0.0013428611,0.0011616118,0.002049129,0.06641145,0.023305988,0.022705322,0.015345817,0.0009654629],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027137413,0.0001270105,0.011402652,0.00053728087,0.00010989955,0.0012308101,0.6273545,0.0010759162,0.001323991,0.2971383,0.009341587,0.050086632],"study_design_scores_gemma":[0.000091692295,0.00018759813,0.00707584,0.002242244,0.000060149912,0.0012657103,0.46699038,0.0023521315,0.0029095975,0.42880455,0.087702155,0.0003179334],"about_ca_topic_score_codex":0.0022311627,"about_ca_topic_score_gemma":0.0027990097,"teacher_disagreement_score":0.21150197,"about_ca_system_score_codex":0.011057451,"about_ca_system_score_gemma":0.011710198,"threshold_uncertainty_score":0.97235847},"labels":[],"label_agreement":null},{"id":"W2169058265","doi":"10.1177/1098214009354774","title":"Bibliometrics as a Performance Measurement Tool for Research Evaluation: The Case of Research Funded by the National Cancer Institute of Canada","year":2010,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":157,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal; Canadian Partnership Against Cancer; Canadian Cancer Society","funders":"National Cancer Institute","keywords":"Bibliometrics; Documentation; Management science; Medical education; Medicine; Library science; Computer science; Engineering","score_opus":0.8412037386951862,"score_gpt":0.6852424986880619,"score_spread":0.1559612400071243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169058265","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28626028,0.01945312,0.14367837,0.09588692,0.0007570921,0.0030992175,0.0019050362,0.0009810205,0.44797897],"genre_scores_gemma":[0.89154816,0.004842841,0.097942874,0.0006138059,0.00013085935,0.0005940086,0.0002954428,0.00011547153,0.0039165667],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.8053959,0.114331774,0.007320051,0.0026366904,0.06422172,0.006093888],"domain_scores_gemma":[0.7692756,0.14812052,0.010351466,0.0068327137,0.061299916,0.004119823],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.10130114,0.0010093292,0.0016862112,0.022718403,0.012860315,0.019328484,0.0028948395,0.0026985905,0.0012608333],"category_scores_gemma":[0.20668185,0.00057137205,0.0009884005,0.066737324,0.010867776,0.0075008417,0.006294741,0.002863382,0.00029962833],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003033713,0.00027911915,0.064731576,0.0036071236,0.0003252834,0.002933453,0.060327683,0.018925965,0.0012621939,0.4977456,0.02436499,0.32519373],"study_design_scores_gemma":[0.0002921967,0.0006480977,0.14084741,0.007246402,0.00072993577,0.0021127213,0.14804275,0.11275716,0.0062162546,0.208373,0.37184858,0.0008854733],"about_ca_topic_score_codex":0.6881527,"about_ca_topic_score_gemma":0.70240825,"teacher_disagreement_score":0.9772816,"about_ca_system_score_codex":0.100291386,"about_ca_system_score_gemma":0.12122389,"threshold_uncertainty_score":0.72766834},"labels":[],"label_agreement":null},{"id":"W2253092986","doi":"10.1177/1098214015615230","title":"Introducing Evidence-Based Principles to Guide Collaborative Approaches to Evaluation","year":2015,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Carleton University; Queen's University","funders":"","keywords":"Variety (cybernetics); Set (abstract data type); Context (archaeology); Computer science; Management science; Knowledge management; Psychology; Data science; Engineering; Artificial intelligence","score_opus":0.7348226152021603,"score_gpt":0.5424577550404095,"score_spread":0.1923648601617508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2253092986","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":"methods","prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030161364,0.013868798,0.84041744,0.11023677,0.0022779286,0.009031439,0.00017465392,0.00041253286,0.020564375],"genre_scores_gemma":[0.029388253,0.0031635824,0.9569552,0.0042676954,0.0002445022,0.0053242915,0.000083876126,0.00005862253,0.00051402854],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.33379608,0.51675075,0.07010379,0.011648082,0.06392045,0.0037808737],"domain_scores_gemma":[0.27690932,0.5771062,0.024898369,0.030002296,0.085480236,0.005603571],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5746235,0.003932948,0.005515346,0.028634386,0.011596618,0.03210038,0.016280133,0.017812906,0.0023479213],"category_scores_gemma":[0.56535465,0.003991971,0.005193956,0.011077273,0.045044065,0.025871368,0.025334602,0.035222605,0.0017713526],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014438503,0.0005861977,0.004545994,0.014985781,0.00082370604,0.0011130382,0.042072274,0.009353304,0.00086448627,0.60322195,0.026986962,0.295302],"study_design_scores_gemma":[0.00027420907,0.00033448587,0.0017479402,0.038930602,0.00030675417,0.0006584228,0.014600765,0.0077646533,0.0016617312,0.77099955,0.16235791,0.00036295044],"about_ca_topic_score_codex":0.008942683,"about_ca_topic_score_gemma":0.014068931,"teacher_disagreement_score":0.42537647,"about_ca_system_score_codex":0.02609422,"about_ca_system_score_gemma":0.07054382,"threshold_uncertainty_score":0.524565},"labels":[],"label_agreement":null},{"id":"W2318178033","doi":"10.1177/1098214013503698","title":"Managing Tensions Between Evaluation and Research","year":2013,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université de Sherbrooke","funders":"","keywords":"Temporality; Psychological intervention; Software deployment; Process (computing); Management science; Psychology; Sociology; Engineering ethics; Computer science; Epistemology","score_opus":0.47772656074020564,"score_gpt":0.6157514516411698,"score_spread":0.1380248909009642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2318178033","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":"methods","domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042764284,0.047844715,0.27516702,0.554916,0.005339212,0.0021949986,0.00006547049,0.0005686366,0.07113967],"genre_scores_gemma":[0.7766484,0.012713482,0.14212693,0.05121756,0.0046005277,0.0065467954,0.00005497956,0.00049606065,0.005595319],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.14882767,0.74421144,0.029847687,0.014879947,0.056167223,0.0060659833],"domain_scores_gemma":[0.1042697,0.79156697,0.019076835,0.032034148,0.04314361,0.009908788],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7301786,0.0018743942,0.006284974,0.015876928,0.015486139,0.04943062,0.0074745617,0.013236254,0.003959014],"category_scores_gemma":[0.6980662,0.0031302257,0.0016730808,0.009548042,0.1006884,0.051139906,0.041770637,0.021559667,0.0011325603],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005061971,0.0003378964,0.0061459974,0.004396831,0.0003454243,0.0009904495,0.15854713,0.0012142754,0.00080636895,0.51911837,0.012312264,0.29527882],"study_design_scores_gemma":[0.00046729553,0.0008581026,0.0039156107,0.012890758,0.00018439245,0.0014293692,0.10325854,0.0044121724,0.0012162455,0.76274234,0.10820525,0.00042003102],"about_ca_topic_score_codex":0.0031638488,"about_ca_topic_score_gemma":0.0030883367,"teacher_disagreement_score":0.2698214,"about_ca_system_score_codex":0.03255483,"about_ca_system_score_gemma":0.052853774,"threshold_uncertainty_score":0.33273786},"labels":[],"label_agreement":null},{"id":"W2319861733","doi":"10.1177/1098214014542100","title":"Insights on Using Developmental Evaluation for Innovating","year":2014,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Conceptualization; Process (computing); Knowledge management; Process management; Rendering (computer graphics); Computer science; Psychology; Management science; Business; Engineering; Artificial intelligence","score_opus":0.35735194148799976,"score_gpt":0.5543062644225002,"score_spread":0.19695432293450044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2319861733","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.073958814,0.01056835,0.43589702,0.06277229,0.00042433254,0.0016820172,0.000099477365,0.00056869816,0.41402906],"genre_scores_gemma":[0.8674265,0.0029546365,0.122635596,0.0020840063,0.000091601796,0.0010165924,0.000039332004,0.0001445667,0.003607173],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.74239546,0.22711095,0.006280934,0.003314413,0.016646646,0.0042515127],"domain_scores_gemma":[0.57370156,0.37786654,0.008866041,0.016090045,0.020328177,0.0031476086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.16166079,0.0012312129,0.0010764115,0.008369868,0.005066621,0.017172499,0.0032344395,0.00354057,0.0043488354],"category_scores_gemma":[0.22253811,0.000773814,0.0010819673,0.0038426332,0.03223392,0.028180672,0.012934297,0.0040431265,0.0004947126],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001005008,0.00017868068,0.0052687284,0.0007192925,0.000028831853,0.00032535996,0.037442632,0.0015819524,0.0003999371,0.80945045,0.0021268704,0.1423767],"study_design_scores_gemma":[0.00013143585,0.0005166459,0.0052496106,0.003994573,0.00007348069,0.0011189299,0.04566437,0.008933743,0.0042129704,0.77001244,0.15988739,0.0002044635],"about_ca_topic_score_codex":0.0048516677,"about_ca_topic_score_gemma":0.004775603,"teacher_disagreement_score":0.16166079,"about_ca_system_score_codex":0.014105084,"about_ca_system_score_gemma":0.016004415,"threshold_uncertainty_score":0.8549542},"labels":[],"label_agreement":null},{"id":"W2520519417","doi":"10.1177/1098214016668401","title":"Introducing Reflexivity to Evaluation Practice","year":2016,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reflexivity; Action (physics); Psychology; Thematic analysis; Action research; Engineering ethics; Action plan; Competence (human resources); Sociology; Qualitative research; Pedagogy; Social psychology; Social science","score_opus":0.27423114310164076,"score_gpt":0.5920985262212245,"score_spread":0.31786738311958374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2520519417","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":"methods","prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015111738,0.010774797,0.70047987,0.18562904,0.0051490692,0.0044767475,0.0000986639,0.00094007415,0.07733988],"genre_scores_gemma":[0.44650272,0.0048061633,0.50606793,0.02285674,0.0019722988,0.011696885,0.000076806886,0.0006485169,0.0053719557],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.17608705,0.76140606,0.020047734,0.014837955,0.02403522,0.0035858783],"domain_scores_gemma":[0.1790014,0.70258856,0.02003249,0.05233655,0.04006831,0.0059727733],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6363066,0.0023923852,0.004010295,0.011745642,0.0107399635,0.035169918,0.008170935,0.012264353,0.005727237],"category_scores_gemma":[0.6249049,0.002773824,0.0033291471,0.0049712635,0.13846008,0.04205873,0.032209393,0.023065591,0.0015263923],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020513945,0.00018430207,0.001518221,0.0042971605,0.00026610892,0.00046610238,0.20699854,0.0014064319,0.00073608145,0.67878205,0.008746704,0.096393146],"study_design_scores_gemma":[0.00027745092,0.00027323383,0.0004969805,0.009680139,0.00008800164,0.0003592958,0.04204626,0.0023158835,0.0016286368,0.8379896,0.10467662,0.0001678459],"about_ca_topic_score_codex":0.0024364695,"about_ca_topic_score_gemma":0.0018928031,"teacher_disagreement_score":0.36369342,"about_ca_system_score_codex":0.020911653,"about_ca_system_score_gemma":0.044426695,"threshold_uncertainty_score":0.4484988},"labels":[],"label_agreement":null},{"id":"W2800401326","doi":"10.1177/1098214018765698","title":"Outcomes and Impacts of Development Interventions","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":120,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Roads University","funders":"International Fund for Agricultural Development; Consortium of International Agricultural Research Centers; Centre for International Forestry Research; Canada Research Chairs; United Nations Development Programme","keywords":"CLARITY; Psychological intervention; Consistency (knowledge bases); Outcome (game theory); Accountability; Confusion; Management science; Psychology; Computer science; Sociology; Political science; Economics","score_opus":0.2737634216383244,"score_gpt":0.5787112819025034,"score_spread":0.304947860264179,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2800401326","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15597679,0.029888991,0.16022487,0.040484436,0.0025994761,0.0040179677,0.0048787086,0.00049235177,0.60143644],"genre_scores_gemma":[0.95414966,0.0072129387,0.028365579,0.0015749361,0.00029592428,0.0030922804,0.00074148574,0.00011691944,0.004450218],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.8732661,0.090497635,0.0067988364,0.004151207,0.022188473,0.0030977228],"domain_scores_gemma":[0.88338464,0.08496131,0.013026809,0.0049871164,0.011895183,0.0017450002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.059530612,0.0015863775,0.0011873563,0.009203728,0.002047702,0.011257616,0.0015232221,0.0024513076,0.009539082],"category_scores_gemma":[0.16302256,0.00043745994,0.0018865723,0.0058415444,0.010876529,0.009573197,0.008454396,0.0034454488,0.0006577182],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005076659,0.00050790654,0.024702417,0.006027138,0.0007413821,0.0001765454,0.007871564,0.007817643,0.0005657935,0.7739328,0.0066354913,0.17051359],"study_design_scores_gemma":[0.0002985804,0.0016963194,0.07662067,0.012939941,0.0015294635,0.00038406564,0.015833171,0.007518196,0.0067863325,0.7692204,0.106936455,0.00023645234],"about_ca_topic_score_codex":0.002679595,"about_ca_topic_score_gemma":0.0020928164,"teacher_disagreement_score":0.059530612,"about_ca_system_score_codex":0.010371127,"about_ca_system_score_gemma":0.010213242,"threshold_uncertainty_score":0.31483173},"labels":[],"label_agreement":null},{"id":"W2802781340","doi":"10.1177/1098214018763553","title":"Evaluating Social Innovations","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Variety (cybernetics); Psychological intervention; Management science; Conceptual framework; Computer science; Knowledge management; Sociology; Engineering ethics; Psychology; Social science; Economics; Engineering","score_opus":0.4567176324915049,"score_gpt":0.6367359267377268,"score_spread":0.18001829424622195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2802781340","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5218914,0.08301069,0.18631394,0.0140174925,0.0028952642,0.03659081,0.0018487133,0.00046335027,0.15296832],"genre_scores_gemma":[0.82247335,0.014654728,0.14931427,0.0010993073,0.00028769183,0.009540528,0.00044634158,0.000073397634,0.0021104433],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7264995,0.23779276,0.012876938,0.0032847452,0.018092956,0.0014531526],"domain_scores_gemma":[0.55418634,0.39119163,0.017371546,0.0072125755,0.027650539,0.0023873085],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.17137894,0.0016478365,0.0020297011,0.0067424276,0.0015791466,0.006889093,0.0017992423,0.0019223016,0.0069193286],"category_scores_gemma":[0.32965532,0.0004090881,0.0019170493,0.0045275334,0.0035894576,0.0054874527,0.0039524552,0.0012649894,0.00041919132],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015708002,0.0017373563,0.022199847,0.038472757,0.0037301008,0.00019385299,0.012938147,0.008654835,0.0015797538,0.07033233,0.0062949113,0.8322953],"study_design_scores_gemma":[0.004553753,0.047618188,0.08520885,0.15583383,0.018298315,0.0007481138,0.08245593,0.04097948,0.031188464,0.31658518,0.2156029,0.0009270485],"about_ca_topic_score_codex":0.001468389,"about_ca_topic_score_gemma":0.0033836174,"teacher_disagreement_score":0.17137894,"about_ca_system_score_codex":0.00747007,"about_ca_system_score_gemma":0.010281045,"threshold_uncertainty_score":0.9063493},"labels":[],"label_agreement":null},{"id":"W2884500380","doi":"10.1177/1098214018778809","title":"The Need for Analysts in Social Impact Measurement","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":40,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Psychology; Program evaluation; Management science; Actuarial science; Applied psychology; Business; Political science; Economics; Public administration","score_opus":0.3492754177554118,"score_gpt":0.5984297125613699,"score_spread":0.24915429480595808,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2884500380","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013238663,0.011889014,0.14948523,0.7838632,0.005095208,0.00035137238,0.00018357097,0.0015800899,0.034313682],"genre_scores_gemma":[0.48784274,0.0075579593,0.37744904,0.108410686,0.0074279374,0.0016226334,0.00037601616,0.00085082,0.008462204],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7712388,0.15541768,0.012336659,0.010106022,0.046593856,0.004306987],"domain_scores_gemma":[0.22800471,0.55875325,0.022660363,0.0393338,0.12648487,0.024762994],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3016407,0.0020851912,0.0028328395,0.014607583,0.008830369,0.025082411,0.008349001,0.018122438,0.006904041],"category_scores_gemma":[0.5140064,0.002417527,0.001768144,0.005616866,0.023540622,0.056531806,0.018383043,0.03252623,0.0033408757],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006470824,0.0009874406,0.029844841,0.002298061,0.00028958378,0.00047853656,0.019673746,0.004037233,0.0028314064,0.36985832,0.13945071,0.42960307],"study_design_scores_gemma":[0.0002298537,0.00022400453,0.0067299423,0.0035160342,0.00013150575,0.00075899885,0.019200744,0.013417316,0.0017018913,0.72529286,0.22838755,0.0004093143],"about_ca_topic_score_codex":0.013192122,"about_ca_topic_score_gemma":0.01522988,"teacher_disagreement_score":0.3016407,"about_ca_system_score_codex":0.010137553,"about_ca_system_score_gemma":0.053497057,"threshold_uncertainty_score":0.86120135},"labels":[],"label_agreement":null},{"id":"W2884511286","doi":"10.1177/1098214018781506","title":"Making Space for Adaptive Learning","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"University of Ottawa","keywords":"Interrogation; Space (punctuation); Mediation; Psychology; Social learning; Process (computing); Social psychology; Computer science; Sociology; Pedagogy; Political science; Social science","score_opus":0.3892532860461554,"score_gpt":0.5842406175819069,"score_spread":0.19498733153575154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2884511286","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33781016,0.0027108402,0.293062,0.05286011,0.0006680273,0.0009353665,0.000107199696,0.00094618835,0.31090018],"genre_scores_gemma":[0.9586256,0.00040492162,0.03619913,0.00039539664,0.00005937896,0.00022487613,0.000027534157,0.0000692914,0.00399389],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.97861415,0.015354856,0.00069669145,0.0015218802,0.0025475873,0.001264846],"domain_scores_gemma":[0.9583796,0.024894474,0.003423658,0.006324415,0.0036708235,0.0033069735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018609101,0.0006270022,0.00041391485,0.001349034,0.00535308,0.010470405,0.0024374863,0.0027122472,0.00917422],"category_scores_gemma":[0.051447652,0.00044813275,0.00074243074,0.0005927907,0.022291781,0.019663515,0.018447634,0.0035564103,0.0010259433],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015040534,0.0003393677,0.010526781,0.00058991066,0.000055225122,0.0008411006,0.12398529,0.0033560738,0.004367659,0.64512163,0.0053874166,0.20527913],"study_design_scores_gemma":[0.00008943997,0.00038444548,0.0051140813,0.00082180195,0.00005288626,0.0006603855,0.10487921,0.004386424,0.006475233,0.62691265,0.25008702,0.00013638772],"about_ca_topic_score_codex":0.0010649068,"about_ca_topic_score_gemma":0.001601221,"teacher_disagreement_score":0.018609101,"about_ca_system_score_codex":0.0024015433,"about_ca_system_score_gemma":0.0070529343,"threshold_uncertainty_score":0.09841555},"labels":[],"label_agreement":null},{"id":"W2901688885","doi":"10.1177/1098214018796319","title":"Honoring Lived Experience: Life Histories as a Realist Evaluation Method","year":2018,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; University of British Columbia; Impact; St. Michael's Hospital","funders":"Tehran University of Medical Sciences and Health Services","keywords":"Lived experience; Context (archaeology); Set (abstract data type); Psychology; Indigenous; Sociology; Epistemology; Computer science; History; Psychotherapist","score_opus":0.365242125327843,"score_gpt":0.5972764607902662,"score_spread":0.2320343354624232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901688885","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5440579,0.0034417384,0.27704456,0.009697029,0.00068264385,0.031534128,0.0024353976,0.00025308374,0.13085352],"genre_scores_gemma":[0.78194183,0.0009930116,0.14837518,0.0013177673,0.00011818243,0.05970327,0.00069548824,0.0001269977,0.0067282384],"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.8772202,0.11246415,0.0027441129,0.0026584514,0.0038164142,0.0010966166],"domain_scores_gemma":[0.9042284,0.060257602,0.008371769,0.012001743,0.011641605,0.003498934],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0729803,0.0009289782,0.0007014651,0.0056962087,0.004515259,0.0058208234,0.002716025,0.0013084679,0.009001489],"category_scores_gemma":[0.095426,0.0008088898,0.0006129029,0.0038206577,0.008705724,0.007340001,0.010354905,0.0024328236,0.0008987842],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013693484,0.001826032,0.045136698,0.0029452946,0.00023407805,0.00093764096,0.43849337,0.0014206492,0.0030670618,0.083992094,0.009277587,0.41130018],"study_design_scores_gemma":[0.0008560201,0.0036975043,0.044974,0.0053405394,0.00037705118,0.0009408385,0.56413776,0.00567217,0.009462442,0.12552768,0.23866166,0.0003523042],"about_ca_topic_score_codex":0.0016361729,"about_ca_topic_score_gemma":0.004881443,"teacher_disagreement_score":0.9270197,"about_ca_system_score_codex":0.0056477003,"about_ca_system_score_gemma":0.0051672137,"threshold_uncertainty_score":0.38596135},"labels":[],"label_agreement":null},{"id":"W2936658129","doi":"10.1177/1098214019835821","title":"Research and Evaluation With Community-Based Projects: Approaches, Considerations, and Strategies","year":2019,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Public Health Agency of Canada","keywords":"Context (archaeology); Management science; Program evaluation; Engineering ethics; Psychology; Process management; Political science; Business; Engineering","score_opus":0.8556213530259436,"score_gpt":0.7057655941049885,"score_spread":0.14985575892095504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2936658129","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038240276,0.024893519,0.5219778,0.25222442,0.0017898055,0.11187637,0.00025871443,0.00072552235,0.048013557],"genre_scores_gemma":[0.22003882,0.0059999195,0.6008642,0.010291254,0.00033132383,0.16023335,0.000079133184,0.00015303236,0.002008955],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.17415603,0.7807003,0.019537423,0.0045291055,0.017629655,0.0034474605],"domain_scores_gemma":[0.33401662,0.54926145,0.022419306,0.031934842,0.04759129,0.01477639],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.65620685,0.0029484017,0.0032945666,0.009485471,0.017638793,0.030777363,0.009855736,0.012769156,0.0048268735],"category_scores_gemma":[0.46756703,0.0031528238,0.0020443304,0.0077762855,0.028191904,0.032739464,0.030373596,0.011086025,0.0012125896],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007696518,0.0040453845,0.01468589,0.021296557,0.00060868094,0.0010410498,0.1550503,0.004117558,0.0011804159,0.26756287,0.013777201,0.5158644],"study_design_scores_gemma":[0.0021849154,0.004332617,0.009628681,0.05782796,0.00064092624,0.0016764825,0.28663132,0.014004924,0.0041462705,0.4885527,0.12961112,0.00076217594],"about_ca_topic_score_codex":0.0069190436,"about_ca_topic_score_gemma":0.012974405,"teacher_disagreement_score":0.65620685,"about_ca_system_score_codex":0.02407199,"about_ca_system_score_gemma":0.09932799,"threshold_uncertainty_score":0.42395824},"labels":[],"label_agreement":null},{"id":"W2971517022","doi":"10.1177/1098214019866260","title":"Evaluations in the English-Speaking Commonwealth Caribbean Region: Lessons From the Field","year":2019,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Commonwealth; Pride; Nexus (standard); Face (sociological concept); Public relations; Political science; Sociology; Psychology; Social psychology; Social science; Law","score_opus":0.20495958925068822,"score_gpt":0.5292134531119421,"score_spread":0.3242538638612539,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2971517022","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21828444,0.100457646,0.007946042,0.43530762,0.0021551505,0.0013431242,0.00017395777,0.000104857354,0.23422715],"genre_scores_gemma":[0.935555,0.028485572,0.0058097425,0.021150038,0.00053012744,0.00051256525,0.000055583772,0.000060906394,0.00784058],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.93437046,0.053449597,0.0014258748,0.0010644874,0.004086589,0.0056029162],"domain_scores_gemma":[0.80560845,0.13135086,0.005061563,0.004038375,0.03522698,0.018713819],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.067844056,0.0004699727,0.0009769626,0.0026311884,0.0125036575,0.017225163,0.0024467222,0.0036359348,0.004706574],"category_scores_gemma":[0.092838146,0.0002976203,0.00051642285,0.0034454833,0.011936446,0.0069676386,0.006848414,0.0053801006,0.00037348914],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059638225,0.0022056794,0.024729865,0.005378432,0.00016036218,0.003991125,0.21462974,0.0022062093,0.00080004876,0.116395034,0.071286514,0.5576207],"study_design_scores_gemma":[0.00018911298,0.0007021939,0.03285742,0.018635416,0.00008382687,0.0006789462,0.5820024,0.0009592334,0.0009963022,0.044823892,0.3178438,0.00022738532],"about_ca_topic_score_codex":0.14071572,"about_ca_topic_score_gemma":0.20684077,"teacher_disagreement_score":0.14071572,"about_ca_system_score_codex":0.027530316,"about_ca_system_score_gemma":0.08780412,"threshold_uncertainty_score":0.35879797},"labels":[],"label_agreement":null},{"id":"W3035076565","doi":"10.1177/1098214020908211","title":"The Role of Intuition in Evaluative Judgment and Decision","year":2020,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Intuition; Psychology; Epistemology; Social psychology; Management science; Cognitive science","score_opus":0.117224265564639,"score_gpt":0.4903323073651327,"score_spread":0.37310804180049373,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035076565","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46568155,0.0062781414,0.3289952,0.017325044,0.00044584586,0.0005672115,0.00008502105,0.00032045064,0.18030159],"genre_scores_gemma":[0.9621582,0.00072575745,0.035000216,0.00071332365,0.000057245463,0.00011751463,0.000020741378,0.000042855332,0.0011642252],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.90719134,0.07056125,0.0034317174,0.003731327,0.011992136,0.0030921751],"domain_scores_gemma":[0.71210605,0.24901897,0.0143994745,0.009949345,0.011237207,0.003288969],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06891401,0.00067022943,0.000775956,0.0039334544,0.0034911032,0.010106097,0.0013123776,0.0022373612,0.0016288573],"category_scores_gemma":[0.17944165,0.00064755994,0.0008793494,0.001682092,0.023713844,0.0099550355,0.0057747574,0.0037824344,0.00034976855],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004013953,0.00038945212,0.032239337,0.0013142644,0.0002618623,0.0011059884,0.22393352,0.0085086515,0.0043974197,0.4956857,0.0033742094,0.22838825],"study_design_scores_gemma":[0.0000861354,0.00046902004,0.026424702,0.0018236109,0.00009275157,0.0008841656,0.04097775,0.015131037,0.0034139885,0.87082297,0.039454605,0.0004193339],"about_ca_topic_score_codex":0.0023312056,"about_ca_topic_score_gemma":0.0021223724,"teacher_disagreement_score":0.06891401,"about_ca_system_score_codex":0.0038656453,"about_ca_system_score_gemma":0.0070195254,"threshold_uncertainty_score":0.36445647},"labels":[],"label_agreement":null},{"id":"W3035271094","doi":"10.1177/1098214019899164","title":"Talking Circles: A Culturally Responsive Evaluation Practice","year":2020,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":64,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Stollery Children's Hospital","funders":"","keywords":"Privilege (computing); Indigenous; Invisibility; Sociology; Power (physics); Stakeholder; Culturally appropriate; Power structure; Psychology; Pedagogy; Social psychology; Public relations; Computer science; Ethnography; Political science; Medicine; Computer security; Artificial intelligence","score_opus":0.22132405977411476,"score_gpt":0.5437861279564394,"score_spread":0.3224620681823247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035271094","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.076629765,0.003420535,0.5459233,0.08687682,0.0024067054,0.0071182866,0.00013713646,0.0035797874,0.2739076],"genre_scores_gemma":[0.5257627,0.0024039012,0.42708117,0.012185345,0.0005464227,0.006373995,0.000101793994,0.0014353207,0.024109362],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.6821614,0.2902946,0.004625014,0.00588567,0.014154793,0.0028786024],"domain_scores_gemma":[0.82422036,0.11021578,0.0069644637,0.018455964,0.025676686,0.014466688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.18009643,0.001232699,0.0010689141,0.0052143494,0.014123045,0.017829113,0.005078589,0.00442523,0.009659098],"category_scores_gemma":[0.16168351,0.0010973728,0.0011726711,0.0027454514,0.021966891,0.014894296,0.02143405,0.006730668,0.0032978442],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029142844,0.0010458318,0.00346065,0.0015147346,0.00013240596,0.0012772575,0.39458713,0.0016247546,0.0032816485,0.15086897,0.05233152,0.38958365],"study_design_scores_gemma":[0.00022756222,0.00086954643,0.0020059508,0.004351746,0.00014402448,0.0016737361,0.2779998,0.0043149632,0.0064303963,0.17794836,0.52364033,0.0003936325],"about_ca_topic_score_codex":0.0022395349,"about_ca_topic_score_gemma":0.0053148465,"teacher_disagreement_score":0.18009643,"about_ca_system_score_codex":0.009125361,"about_ca_system_score_gemma":0.024566334,"threshold_uncertainty_score":0},"labels":[{"model":"gpt","categories":[],"domain":null,"study_design":"qualitative","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"},{"model":"grok","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"},{"model":"opus","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"split"},{"id":"W3127477413","doi":"10.1177/1098214020927785","title":"Photo-Based Evaluation: A Method for Participatory Evaluation With Adolescents","year":2021,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Participatory Visual Research Methods","field":"Social Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Participatory evaluation; Transformative learning; Visual methods; Program evaluation; Citizen journalism; Participatory action research; Health promotion; Promotion (chess); Process (computing); Psychology; Youth engagement; Applied psychology; Visual research; Positive Youth Development; Medical education; Evaluation methods; Computer science; Pedagogy; Public relations; Sociology; Public health; Medicine; Nursing; Political science; Developmental psychology; Politics; Social science; Engineering","score_opus":0.6163707543039818,"score_gpt":0.6992086103214732,"score_spread":0.08283785601749138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3127477413","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02778556,0.0012603671,0.82159394,0.0032543524,0.00090278796,0.08392409,0.00075371505,0.0011056898,0.05941948],"genre_scores_gemma":[0.06263266,0.00064111134,0.8419294,0.00061760464,0.00011447539,0.08711758,0.00011716082,0.00023811145,0.0065919654],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.8680907,0.11817503,0.002646489,0.003478522,0.0067330287,0.0008762367],"domain_scores_gemma":[0.9252458,0.055919398,0.0028177337,0.008415786,0.006408611,0.0011925779],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09034528,0.0017665405,0.0010412406,0.004708692,0.0045402288,0.0038220373,0.002680621,0.0020823237,0.019217381],"category_scores_gemma":[0.065445945,0.0011181452,0.0013782976,0.0026793808,0.005798509,0.0038191315,0.0069968877,0.0027062383,0.0029269534],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016987219,0.0022077945,0.0026526197,0.006457251,0.00014802278,0.00069458084,0.119912885,0.0017221824,0.026271332,0.052252624,0.021020954,0.764961],"study_design_scores_gemma":[0.0034048792,0.006646184,0.010385707,0.011617151,0.00042191014,0.0020294134,0.09315203,0.009386378,0.048679743,0.10109885,0.7123427,0.0008351451],"about_ca_topic_score_codex":0.00095320476,"about_ca_topic_score_gemma":0.0032162243,"teacher_disagreement_score":0.09034528,"about_ca_system_score_codex":0.0024078886,"about_ca_system_score_gemma":0.0062097316,"threshold_uncertainty_score":0.4777972},"labels":[],"label_agreement":null},{"id":"W3178131540","doi":"10.1177/1098214020940409","title":"Reviewing Health Service and Program Evaluations in Indigenous Contexts: A Systematic Review","year":2021,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Indigenous Health, Education, and Rights","field":"Social Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Addiction and Mental Health; Public Health Ontario; University of Toronto; St. Michael's Hospital","funders":"Ontario Ministry of Health and Long-Term Care","keywords":"Indigenous; Reciprocity (cultural anthropology); Public relations; Sociology; Service (business); Management science; Psychology; Political science; Business; Social science; Engineering; Marketing; Ecology","score_opus":0.07921618690179015,"score_gpt":0.4988806023287977,"score_spread":0.41966441542700755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3178131540","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018675721,0.9911841,0.0010802079,0.0010290826,0.00055620243,0.0024864785,0.0008057826,0.000035806876,0.00095473666],"genre_scores_gemma":[0.023207525,0.967424,0.0034581714,0.0010626545,0.00017485186,0.003721595,0.00065377785,0.000026975244,0.00027050867],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9321017,0.03164806,0.0202156,0.0024268145,0.012667512,0.0009403411],"domain_scores_gemma":[0.77308106,0.17216676,0.022217145,0.004975047,0.025840733,0.0017192394],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06274541,0.0020658064,0.0089894505,0.024262318,0.0019934315,0.004786278,0.003377312,0.0027017775,0.0038808363],"category_scores_gemma":[0.23473468,0.0019068759,0.005148672,0.020778084,0.002440463,0.0051097693,0.0038774954,0.002322346,0.00048611054],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000112936556,0.000028296165,0.00063571305,0.91873026,0.0029989406,0.000097315344,0.00090008945,0.00011985721,0.00015691954,0.0005820302,0.003035915,0.072601795],"study_design_scores_gemma":[0.00006778248,0.00008453326,0.0012538756,0.9652172,0.010292938,0.00008673058,0.00080310885,0.000058546884,0.0001445891,0.00033202808,0.021632174,0.000026521226],"about_ca_topic_score_codex":0.013762189,"about_ca_topic_score_gemma":0.046239395,"teacher_disagreement_score":0.9372546,"about_ca_system_score_codex":0.01085946,"about_ca_system_score_gemma":0.04949241,"threshold_uncertainty_score":0.33183342},"labels":[],"label_agreement":null},{"id":"W3183879452","doi":"10.1177/10982140211007573","title":"Understanding Evaluation Policy and Organizational Capacity for Evaluation: An Interview Study","year":2021,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Accountability; Context (archaeology); Thematic analysis; Organizational performance; Organizational learning; Capacity building; Conceptual framework; Knowledge management; Management science; Sociology; Psychology; Public relations; Political science; Qualitative research; Computer science; Social science; Economics","score_opus":0.6422351798637937,"score_gpt":0.5788854004605903,"score_spread":0.06334977940320341,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3183879452","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96356034,0.00074200385,0.009309752,0.012821319,0.00007179799,0.00035090497,0.000052615072,0.000020226133,0.013071136],"genre_scores_gemma":[0.99262434,0.0006371503,0.0033552225,0.0015068036,0.000037918246,0.00030156987,0.000024933264,0.000014045147,0.0014980818],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9435546,0.047267944,0.0017848343,0.0013275266,0.0022611383,0.0038040443],"domain_scores_gemma":[0.83649504,0.14132023,0.006470966,0.002420934,0.008016272,0.0052765524],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07192578,0.00034998704,0.0006125908,0.0027838952,0.0130604245,0.008076048,0.0016363115,0.0036292989,0.0020051864],"category_scores_gemma":[0.09034042,0.00089422025,0.00030831748,0.0028840455,0.010972221,0.009771001,0.0062859836,0.0058021005,0.00024105371],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025921247,0.00014532024,0.006112574,0.00007516429,0.0000032613,0.00029821988,0.9777541,0.0001612511,0.0005187251,0.008486974,0.0006047361,0.005813755],"study_design_scores_gemma":[0.0000065682766,0.000047760197,0.001393371,0.0001468876,0.0000029521862,0.00008193329,0.980648,0.000427126,0.0002717076,0.0014171697,0.015542857,0.000013699141],"about_ca_topic_score_codex":0.004700779,"about_ca_topic_score_gemma":0.005780962,"teacher_disagreement_score":0.92807424,"about_ca_system_score_codex":0.012348824,"about_ca_system_score_gemma":0.013575854,"threshold_uncertainty_score":0.38038445},"labels":[],"label_agreement":null},{"id":"W3202036368","doi":"10.1177/1098214021997574","title":"Cocreating an Evaluation Approach for a Healthy Relationships Program With Community Partners: Lessons Learned and Recommendations","year":2021,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Community Health and Development","field":"Health Professions","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Public Health Agency of Canada","keywords":"Community-based participatory research; Participatory action research; Flexibility (engineering); Mental health; Context (archaeology); Program evaluation; Psychology; Process (computing); Medical education; Applied psychology; Sociology; Medicine; Political science; Computer science; Psychiatry","score_opus":0.5453551764340937,"score_gpt":0.6032116774600894,"score_spread":0.05785650102599571,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202036368","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04655737,0.030780982,0.091130435,0.73864394,0.0059015783,0.04475192,0.00044523244,0.0012235575,0.04056495],"genre_scores_gemma":[0.17575255,0.032981433,0.67631125,0.05013838,0.00089355995,0.056908153,0.000338944,0.00028514877,0.0063905027],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.8784046,0.10143858,0.0051587126,0.0016944014,0.009581387,0.0037221967],"domain_scores_gemma":[0.8635868,0.087843455,0.0041694622,0.0058798227,0.022824675,0.01569581],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.130368,0.0017442877,0.001582608,0.003390837,0.009091638,0.009540884,0.0056839017,0.006038964,0.0063879336],"category_scores_gemma":[0.13502012,0.00081911683,0.0020683804,0.0024925142,0.0053675664,0.012375949,0.011642838,0.010233667,0.0008442031],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034973418,0.0068281307,0.007568955,0.008695048,0.00016335896,0.0016624171,0.030794842,0.0011237905,0.000547309,0.024545114,0.064339474,0.8533819],"study_design_scores_gemma":[0.0028912893,0.006774954,0.025603168,0.14437598,0.0012111868,0.0032959534,0.28197715,0.0064672963,0.0040899077,0.128198,0.3942743,0.0008408163],"about_ca_topic_score_codex":0.019684765,"about_ca_topic_score_gemma":0.08313926,"teacher_disagreement_score":0.130368,"about_ca_system_score_codex":0.011724273,"about_ca_system_score_gemma":0.08678647,"threshold_uncertainty_score":0.6894601},"labels":[],"label_agreement":null},{"id":"W3203032688","doi":"10.1177/10982140211009971","title":"Book Review: <i>Scaling Impact: Innovation for the Public Good</i> by Robert McLean &amp; John Gargani.","year":2021,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"University-Industry-Government Innovation Models","field":"Business, Management and Accounting","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Public Health Ontario; University of Toronto","funders":"","keywords":"Sociology; Scaling; Psychology; Management; Economics; Mathematics","score_opus":0.032573228569346045,"score_gpt":0.2951786650114992,"score_spread":0.2626054364421532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3203032688","genre_codex":"commentary","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00021862911,0.30790672,0.0022849203,0.3883763,0.14288442,0.00028185255,0.00080219225,0.00052698783,0.15671803],"genre_scores_gemma":[0.00365472,0.21131958,0.0019178633,0.13734922,0.066502996,0.00025471,0.00065506087,0.00055723975,0.57778865],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99657947,0.00045799298,0.0001386826,0.00026848738,0.0024218855,0.00013354183],"domain_scores_gemma":[0.982454,0.0067329253,0.00076295383,0.00039093551,0.008659824,0.0009993685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029914547,0.0012514241,0.0012678456,0.0032023236,0.001275042,0.0050039436,0.0017669399,0.005067677,0.045995325],"category_scores_gemma":[0.015540248,0.00054077496,0.0007137671,0.0034972099,0.002035064,0.004079864,0.0014181897,0.0061904024,0.039703317],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000017753051,0.0000024131189,0.000010456301,0.000047660546,0.0000014677892,0.0000021529227,0.0000043229884,0.0000124891585,0.000011009028,0.00054067903,0.98913693,0.010228582],"study_design_scores_gemma":[0.000006714178,0.000011317309,0.00020743752,0.00037692144,0.000005917629,0.00003112787,0.000021011732,0.000050956045,0.000049843336,0.0014911648,0.9977368,0.000010780807],"about_ca_topic_score_codex":0.014363753,"about_ca_topic_score_gemma":0.037993122,"teacher_disagreement_score":0.045995325,"about_ca_system_score_codex":0.003689216,"about_ca_system_score_gemma":0.005574681,"threshold_uncertainty_score":0.15386969},"labels":[],"label_agreement":null},{"id":"W3208939115","doi":"10.1177/1098214020936769","title":"The Use of Evaluability Assessments in Improving Future Evaluations: A Scoping Review of 10 Years of Literature (2008–2018)","year":2021,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Guelph","funders":"","keywords":"Ambiguity; Psychology; Engineering ethics; Equity (law); Relevance (law); Management science; Political science; Engineering; Computer science","score_opus":0.36142971211574515,"score_gpt":0.6078438813411042,"score_spread":0.24641416922535908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3208939115","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011183418,0.98951983,0.0029404573,0.002761074,0.00041462664,0.0009197474,0.00025210646,0.000023399221,0.0020504233],"genre_scores_gemma":[0.016012272,0.97255784,0.0075911614,0.0010326327,0.00019461851,0.0020250916,0.00030545692,0.000027453656,0.00025339136],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.8831443,0.06618196,0.028251035,0.0032737223,0.01793993,0.0012090234],"domain_scores_gemma":[0.56652033,0.34349766,0.030329645,0.0073116445,0.051085044,0.0012555915],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.17074722,0.0018487164,0.004163675,0.035633203,0.0023758665,0.007143104,0.002581965,0.0028559405,0.0031844128],"category_scores_gemma":[0.35214257,0.0016521544,0.0054588346,0.02698022,0.002985786,0.009955379,0.0046141054,0.003138896,0.0005955464],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013938113,0.000056326073,0.0016012695,0.4223826,0.0015198087,0.00014339648,0.0033501722,0.0005047263,0.00030374498,0.005759147,0.008148799,0.5560906],"study_design_scores_gemma":[0.000030854782,0.00007082749,0.0019553187,0.918065,0.0032461816,0.00014615459,0.0014680005,0.00020542809,0.00031377954,0.0023197618,0.07213477,0.000043838434],"about_ca_topic_score_codex":0.010697138,"about_ca_topic_score_gemma":0.024896147,"teacher_disagreement_score":0.8292528,"about_ca_system_score_codex":0.009703105,"about_ca_system_score_gemma":0.045269195,"threshold_uncertainty_score":0.9030084},"labels":[],"label_agreement":null},{"id":"W4235318396","doi":"10.1177/109821400002100306","title":"Planning for Community-based Evaluation","year":2000,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Program evaluation; Unit (ring theory); Management science; Set (abstract data type); Process (computing); Conflict resolution; Process management; Psychology; Computer science; Sociology; Political science; Business; Engineering; Mathematics education; Social science","score_opus":0.355370496696108,"score_gpt":0.5926022913441982,"score_spread":0.2372317946480902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4235318396","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0050453786,0.0048363693,0.6097273,0.07798723,0.0026840358,0.09517651,0.0010867572,0.002389382,0.20106702],"genre_scores_gemma":[0.028394036,0.0028802808,0.8865543,0.0044233557,0.00028315888,0.052577317,0.0009998729,0.00047655051,0.023411091],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7813758,0.17594545,0.010166298,0.0034528873,0.024011439,0.0050481725],"domain_scores_gemma":[0.75095135,0.109420575,0.009600819,0.017599674,0.0913053,0.021122318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.18714656,0.0023443387,0.0015809287,0.0076117557,0.009463544,0.013135313,0.0064269393,0.0066717872,0.036263872],"category_scores_gemma":[0.23700802,0.0016473652,0.0020859197,0.006857105,0.005060877,0.011946364,0.01481273,0.01021183,0.011912131],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016673282,0.00094585604,0.0012511022,0.0029609562,0.00007784449,0.0007993412,0.008530861,0.010675406,0.00063890475,0.13860334,0.27878612,0.55656356],"study_design_scores_gemma":[0.00033865223,0.00052098476,0.001663537,0.007570773,0.00006353058,0.00041942263,0.012530033,0.0070279604,0.0010262257,0.17505515,0.79352754,0.000256186],"about_ca_topic_score_codex":0.01629358,"about_ca_topic_score_gemma":0.031974178,"teacher_disagreement_score":0.18714656,"about_ca_system_score_codex":0.016554939,"about_ca_system_score_gemma":0.1151069,"threshold_uncertainty_score":0.98973745},"labels":[],"label_agreement":null},{"id":"W4236526761","doi":"10.1177/109821400002100304","title":"Legal and Ethical Issues in Evaluating Abortion Services","year":2000,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute for Clinical Evaluative Sciences; University of Toronto","funders":"","keywords":"Confidentiality; Abortion; Context (archaeology); Legal service; Ethical issues; Business; Public relations; Internet privacy; Engineering ethics; Law; Political science; Computer science; Engineering","score_opus":0.11986276787814307,"score_gpt":0.5591065619368589,"score_spread":0.43924379405871583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236526761","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":"methods","domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059033632,0.047476858,0.22593632,0.46175918,0.003510112,0.005903832,0.00026704872,0.00014569948,0.19596739],"genre_scores_gemma":[0.6808108,0.009554195,0.25371742,0.042190447,0.002127985,0.006799338,0.000074796604,0.00010107441,0.0046239793],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.2890046,0.6096316,0.035837226,0.005029302,0.056338347,0.0041590054],"domain_scores_gemma":[0.25162134,0.68111634,0.01999979,0.012390896,0.031631183,0.003240306],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.52892303,0.0009011449,0.0021077422,0.0037950792,0.0078002554,0.014359593,0.003619016,0.01207793,0.0021760985],"category_scores_gemma":[0.5799772,0.0008570174,0.0015951693,0.0037268642,0.031337105,0.011413996,0.007057069,0.01158108,0.00042622047],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037497137,0.000332185,0.008279344,0.0019838028,0.00025962276,0.0007179378,0.027278142,0.004787749,0.0008260608,0.7719745,0.012816525,0.17036918],"study_design_scores_gemma":[0.0003796253,0.00092429394,0.008667532,0.0110279005,0.00032653825,0.0010814281,0.019067932,0.008023757,0.0037451168,0.82719994,0.119232416,0.00032354167],"about_ca_topic_score_codex":0.004536248,"about_ca_topic_score_gemma":0.0077285725,"teacher_disagreement_score":0.52892303,"about_ca_system_score_codex":0.010501213,"about_ca_system_score_gemma":0.036662698,"threshold_uncertainty_score":0.58092177},"labels":[],"label_agreement":null},{"id":"W4238820305","doi":"10.1177/109821400402500311","title":"Commentary: Minimizing Evaluation Misuse as Principled Practice","year":2004,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Psychology; Management science; Computer science; Sociology; Economics","score_opus":0.19120246854867498,"score_gpt":0.5583508310794473,"score_spread":0.3671483625307723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4238820305","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00011183526,0.0006893316,0.00011923377,0.967313,0.030877309,0.000014190718,0.000038011443,0.000014332206,0.00082279317],"genre_scores_gemma":[0.00175677,0.00033226053,0.0003294583,0.96679795,0.02959647,0.000057922873,0.000012214169,0.000019526526,0.0010974327],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.94614846,0.01738978,0.006801644,0.007658439,0.016716627,0.0052849874],"domain_scores_gemma":[0.6635491,0.24687038,0.013734276,0.0062297396,0.05752358,0.012092939],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.037903544,0.0022346058,0.003515667,0.003667878,0.012763327,0.011331053,0.012622596,0.18531933,0.01013695],"category_scores_gemma":[0.29394376,0.0024736465,0.0036232257,0.004974994,0.020240936,0.010170992,0.007591005,0.1492979,0.007828983],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004509703,0.000014049156,0.00008137106,0.00022195737,0.000044234985,0.00012684919,0.00053047796,0.000058340916,0.00008013381,0.006620167,0.98932636,0.002850954],"study_design_scores_gemma":[0.00046891003,0.00006976515,0.0009664276,0.004228417,0.00038937843,0.0005769232,0.0020937163,0.0007245809,0.000616981,0.038295455,0.95137036,0.00019916763],"about_ca_topic_score_codex":0.0318085,"about_ca_topic_score_gemma":0.039688468,"teacher_disagreement_score":0.96209645,"about_ca_system_score_codex":0.020989256,"about_ca_system_score_gemma":0.03259691,"threshold_uncertainty_score":0.20045549},"labels":[],"label_agreement":null},{"id":"W4244203368","doi":"10.1177/109821400302400106","title":"A Comparison of Three Retrospective Self-reporting Methods of Measuring Change in Instructional Practice","year":2003,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Behavioral and Psychological Studies","field":"Psychology","cited_by":221,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Satisficing; Psychology; Recall; Retrospective cohort study; Attitude change; Cognition; Behavior change; Intervention (counseling); Social psychology; Applied psychology; Cognitive psychology; Medicine; Computer science","score_opus":0.5729309612507211,"score_gpt":0.5646686413581473,"score_spread":0.008262319892573755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244203368","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91364694,0.0018468489,0.06516715,0.00023774525,0.0003783423,0.006758137,0.0015189693,0.00044563357,0.010000169],"genre_scores_gemma":[0.83996004,0.0016625744,0.12718686,0.00028225308,0.0002616239,0.023646612,0.0025125714,0.00014184728,0.0043455614],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9545159,0.027037028,0.0055610025,0.0025936814,0.00974712,0.00054530293],"domain_scores_gemma":[0.8564786,0.08354625,0.027340233,0.013913523,0.016902333,0.0018191169],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02802865,0.0012775449,0.0011536375,0.0031896268,0.00050142896,0.0013126502,0.0012448415,0.0010569018,0.0017318776],"category_scores_gemma":[0.080324315,0.0010571138,0.0015215867,0.002469829,0.0009751674,0.0020345866,0.0016605359,0.0011836811,0.00051771937],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.01863281,0.004502828,0.58504075,0.002154378,0.0039410447,0.00007125129,0.006409683,0.001460249,0.011815872,0.0015194834,0.0017513983,0.36270025],"study_design_scores_gemma":[0.0014178925,0.018479882,0.95901555,0.0003065994,0.0010174273,0.00034461854,0.002431995,0.004467707,0.0076689664,0.0010164444,0.0035725366,0.00026040614],"about_ca_topic_score_codex":0.0012173553,"about_ca_topic_score_gemma":0.0030806358,"teacher_disagreement_score":0.97197133,"about_ca_system_score_codex":0.00071150356,"about_ca_system_score_gemma":0.0008452845,"threshold_uncertainty_score":0.14823145},"labels":[],"label_agreement":null},{"id":"W4283033908","doi":"10.1177/10982140211056913","title":"Developing Evaluation Approaches for an Anti-Human Trafficking Housing Program","year":2022,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Migration, Health and Trauma","field":"Psychology","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Formative assessment; Protocol (science); Program evaluation; Evaluation methods; Human trafficking; Public relations; Best practice; Psychology; Process management; Political science; Engineering ethics; Business; Medicine; Engineering; Public administration; Pedagogy; Alternative medicine","score_opus":0.2469876266596626,"score_gpt":0.4708132701818361,"score_spread":0.22382564352217352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283033908","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12542073,0.0019485982,0.41890186,0.01326807,0.0006817193,0.3958716,0.00047267968,0.00046653987,0.042968176],"genre_scores_gemma":[0.14178045,0.0007974623,0.63620466,0.0011377484,0.000042578784,0.218584,0.00009266843,0.000040691004,0.0013196823],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.51002836,0.44873607,0.017938966,0.0034558126,0.016712064,0.0031287537],"domain_scores_gemma":[0.63241047,0.29078585,0.0149687985,0.015146082,0.04242589,0.0042628734],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.4301727,0.0014790341,0.0014013514,0.006568038,0.006747271,0.008569424,0.0043925554,0.0030115629,0.0070258765],"category_scores_gemma":[0.32766032,0.0013763037,0.0021464569,0.0031388034,0.0076489826,0.01071438,0.008928987,0.0048930594,0.00078974594],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017084186,0.010600865,0.008788983,0.016149575,0.00036779244,0.0005086471,0.11915312,0.009049056,0.003438036,0.12581576,0.008132579,0.6962871],"study_design_scores_gemma":[0.0139369285,0.048982136,0.033368953,0.058790956,0.0018699523,0.0010180085,0.38741848,0.037286494,0.03372506,0.22173305,0.16073407,0.00113606],"about_ca_topic_score_codex":0.0031244059,"about_ca_topic_score_gemma":0.0065892446,"teacher_disagreement_score":0.4301727,"about_ca_system_score_codex":0.022225266,"about_ca_system_score_gemma":0.057266448,"threshold_uncertainty_score":0.70269847},"labels":[],"label_agreement":null},{"id":"W4308916752","doi":"10.1177/10982140211008978","title":"A Comparison of Fidelity Implementation Frameworks Used in the Field of Early Intervention","year":2022,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Family and Disability Support Research","field":"Psychology","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Fidelity; Conceptualization; Conceptual framework; Intervention (counseling); Field (mathematics); Management science; Computer science; Quality (philosophy); Knowledge management; Psychology; Sociology; Engineering; Social science","score_opus":0.11013994599000954,"score_gpt":0.5610765366306308,"score_spread":0.4509365906406213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4308916752","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46225888,0.086833864,0.34173614,0.024539199,0.0018721804,0.0151511235,0.001086756,0.0006751155,0.06584686],"genre_scores_gemma":[0.81878334,0.014187082,0.15422136,0.0016630427,0.0001652723,0.009269229,0.0005582258,0.00018757797,0.0009648705],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.7011972,0.2073304,0.030677443,0.0051598903,0.051782396,0.003852737],"domain_scores_gemma":[0.44401586,0.445656,0.02745346,0.01644228,0.06396163,0.0024707608],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.30126616,0.00093960494,0.0019247613,0.013024241,0.0039590984,0.008835308,0.002830603,0.0023779776,0.0018289813],"category_scores_gemma":[0.45814118,0.0011607646,0.00449123,0.00826134,0.0072201234,0.009585715,0.0077416417,0.0049138083,0.0002032709],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010632512,0.0004894634,0.10188077,0.011195006,0.0013456355,0.00017603349,0.084712476,0.0019568102,0.00075131294,0.11031179,0.0032849398,0.6828325],"study_design_scores_gemma":[0.00084359007,0.010530098,0.49405295,0.0822896,0.0038928052,0.0026801876,0.16086167,0.018983932,0.006874815,0.10603597,0.11143649,0.0015179018],"about_ca_topic_score_codex":0.012190056,"about_ca_topic_score_gemma":0.013626475,"teacher_disagreement_score":0.6987338,"about_ca_system_score_codex":0.024126682,"about_ca_system_score_gemma":0.025393972,"threshold_uncertainty_score":0.8616632},"labels":[],"label_agreement":null},{"id":"W4313250641","doi":"10.1177/10982140221106991","title":"Laying a Solid Foundation for the Next Generation of Evaluation Capacity Building: Findings from an Integrative Review","year":2022,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Scholarship; Foundation (evidence); Capacity building; Engineering ethics; Political science; Management science; Sociology; Public relations; Economics; Engineering; Law","score_opus":0.5281446215139854,"score_gpt":0.552257622583783,"score_spread":0.024113001069797524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313250641","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062805777,0.9670913,0.004769043,0.015600752,0.0004985606,0.00028214758,0.00014187054,0.000021444694,0.005314355],"genre_scores_gemma":[0.069871455,0.9132697,0.011479808,0.0037878943,0.0004968773,0.0005810567,0.0001709838,0.000025496138,0.00031676443],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9819375,0.009753063,0.0035608145,0.00087572174,0.0033408226,0.00053201575],"domain_scores_gemma":[0.71480715,0.24809591,0.0131496675,0.0033895727,0.018817935,0.0017397049],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04974335,0.0006450059,0.0023898953,0.013381621,0.0016264321,0.0088867005,0.0014972357,0.0019474067,0.002536958],"category_scores_gemma":[0.12700666,0.00063126505,0.0019279777,0.016620146,0.0027827015,0.01144428,0.0045151734,0.0028159395,0.00033315105],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014267431,0.00017463761,0.0057379846,0.22205646,0.0016696848,0.00032810163,0.014725041,0.0010014664,0.00072975055,0.042176202,0.011917158,0.69934076],"study_design_scores_gemma":[0.00005888685,0.0002745704,0.016572917,0.6067102,0.0056442977,0.0007676105,0.034327753,0.0010451541,0.0010406952,0.03230424,0.3010931,0.00016055725],"about_ca_topic_score_codex":0.0038509727,"about_ca_topic_score_gemma":0.010647668,"teacher_disagreement_score":0.95025665,"about_ca_system_score_codex":0.0050107194,"about_ca_system_score_gemma":0.029490652,"threshold_uncertainty_score":0.26307112},"labels":[],"label_agreement":null},{"id":"W4313466224","doi":"10.1177/10982140221079837","title":"Translating Evaluation Policy Into Practice in Government Organizations","year":2022,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Government (linguistics); Policy analysis; Program evaluation; Key (lock); Public policy; Public relations; Evaluation methods; Process management; Business; Public administration; Management science; Political science; Computer science; Economics; Engineering; Computer security","score_opus":0.09400042012462564,"score_gpt":0.5279233762509515,"score_spread":0.4339229561263259,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313466224","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48802778,0.0067803124,0.06251783,0.28361994,0.00089359866,0.0018329178,0.00048277155,0.0005820861,0.15526277],"genre_scores_gemma":[0.97040963,0.0011963552,0.02201591,0.003499558,0.00009954117,0.0005969199,0.0001134556,0.000079739395,0.0019889218],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.7110819,0.21283136,0.013724812,0.009230595,0.036780056,0.016351264],"domain_scores_gemma":[0.51682985,0.30833933,0.04067477,0.0274892,0.09412217,0.01254469],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.25782382,0.0005690552,0.0009778845,0.009021072,0.011053453,0.031750426,0.003604721,0.0052469564,0.0020156663],"category_scores_gemma":[0.40749115,0.0010126126,0.000568925,0.010895831,0.022958657,0.013119882,0.008649492,0.0047079925,0.0004134769],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003055224,0.001015883,0.08301634,0.001534473,0.00014870388,0.0005971419,0.17420341,0.012957432,0.0018600817,0.36329037,0.025169527,0.33590117],"study_design_scores_gemma":[0.00021656923,0.00064201,0.15733258,0.007399133,0.0001211,0.00017463288,0.33914143,0.0148997065,0.0049966504,0.21280825,0.2618394,0.00042850434],"about_ca_topic_score_codex":0.24196346,"about_ca_topic_score_gemma":0.19838312,"teacher_disagreement_score":0.25782382,"about_ca_system_score_codex":0.16499034,"about_ca_system_score_gemma":0.24592866,"threshold_uncertainty_score":0.96849287},"labels":[],"label_agreement":null},{"id":"W4392775921","doi":"10.1177/10982140241234841","title":"Mapping Evaluation Use: A Scoping Review of Extant Literature (2005–2022)","year":2024,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Queen's University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Extant taxon; Program evaluation; Management science; Evaluation methods; Psychology; Sociology; Political science; Engineering; Public administration","score_opus":0.39634967058603965,"score_gpt":0.6096230452322937,"score_spread":0.21327337464625407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392775921","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043426147,0.98436594,0.002433128,0.0024425602,0.00049229275,0.0018890226,0.0009830904,0.000041940402,0.0030093468],"genre_scores_gemma":[0.01913995,0.9691161,0.0064622057,0.00087309896,0.00011777095,0.0028792394,0.00089088496,0.000030788437,0.00048988604],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.96046853,0.014173153,0.013942092,0.0017882675,0.008797949,0.00083010265],"domain_scores_gemma":[0.85422087,0.09033804,0.017783279,0.0038753506,0.032643124,0.0011393226],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07300958,0.0016129101,0.0034232975,0.057254087,0.0022375027,0.0055958177,0.0020876583,0.0024270976,0.0029944074],"category_scores_gemma":[0.19321127,0.0018882252,0.003797541,0.055707593,0.0019734595,0.0071745752,0.00520608,0.0018124423,0.00072411634],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012447377,0.000048638703,0.0029488674,0.4582332,0.0012886979,0.00030688263,0.004556241,0.00040692504,0.0006188703,0.0023268482,0.012471994,0.5166683],"study_design_scores_gemma":[0.000026624515,0.00009766914,0.0059850235,0.91517305,0.0027938175,0.00026301248,0.0029338952,0.0001555871,0.00042493746,0.00086511293,0.07123532,0.000045977773],"about_ca_topic_score_codex":0.015893111,"about_ca_topic_score_gemma":0.04680869,"teacher_disagreement_score":0.9269904,"about_ca_system_score_codex":0.008754549,"about_ca_system_score_gemma":0.04775911,"threshold_uncertainty_score":0.3861162},"labels":[],"label_agreement":null},{"id":"W4401375525","doi":"10.1177/10982140241270011","title":"Book Review: Policy Evaluation in the Era of COVID-19","year":2024,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Centre for Disease Control","funders":"","keywords":"Pearl; Coronavirus disease 2019 (COVID-19); Severe acute respiratory syndrome coronavirus 2 (SARS-CoV-2); 2019-20 coronavirus outbreak; Political science; Virology; Geography; Medicine","score_opus":0.19986347767528487,"score_gpt":0.5871476618453196,"score_spread":0.38728418417003474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401375525","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00010011447,0.74044997,0.00071640935,0.14175227,0.09835622,0.00026269557,0.00026418277,0.00006424716,0.018033916],"genre_scores_gemma":[0.0026296577,0.80575776,0.0015709181,0.098989874,0.06479497,0.00078866904,0.00035999282,0.00010690753,0.025001232],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9798956,0.009790711,0.0015761484,0.0010442882,0.0071694134,0.0005238094],"domain_scores_gemma":[0.8685938,0.10393359,0.005002861,0.0015121903,0.019239927,0.0017177084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012246823,0.0017230462,0.003013502,0.006110852,0.0019451336,0.011444765,0.003307348,0.009930486,0.020036612],"category_scores_gemma":[0.08859364,0.0011100644,0.0017890502,0.008729793,0.004203593,0.0064840596,0.002433288,0.009020424,0.008775885],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017308143,0.000013963008,0.000033754004,0.004749216,0.000026120711,0.000025460473,0.000108072476,0.00011964745,0.000019753403,0.003234267,0.9561863,0.035466157],"study_design_scores_gemma":[0.000035064186,0.0000317278,0.00026454878,0.02130105,0.00004187283,0.00015198106,0.00013779826,0.000076111115,0.000037746493,0.0041282116,0.97376674,0.00002718371],"about_ca_topic_score_codex":0.009165893,"about_ca_topic_score_gemma":0.017892467,"teacher_disagreement_score":0.020036612,"about_ca_system_score_codex":0.012007848,"about_ca_system_score_gemma":0.019826882,"threshold_uncertainty_score":0.08712345},"labels":[],"label_agreement":null},{"id":"W4403692227","doi":"10.1177/10982140241287936","title":"Streamlining Complex Intervention Evaluation Through Participatory Systems Mapping and Contribution Analysis: A Comprehensive Framework for Actionable Complexity Evaluation","year":2024,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École Nationale d'Administration Publique","funders":"","keywords":"Participatory evaluation; Intervention (counseling); Citizen journalism; Computer science; Management science; Evaluation methods; Program evaluation; Process management; Risk analysis (engineering); Engineering; Sociology; Business; Political science; Psychology; Public administration; Reliability engineering; Social science","score_opus":0.5525892727434512,"score_gpt":0.5892092362444481,"score_spread":0.036619963500996944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403692227","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008273981,0.0021909778,0.9488872,0.010129921,0.00020075108,0.009500217,0.00021284985,0.00040931624,0.020194825],"genre_scores_gemma":[0.0927222,0.00089120434,0.8957034,0.00038808887,0.000041549167,0.009399041,0.00010276538,0.00008325337,0.0006685572],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.6436016,0.3174325,0.009578758,0.007576405,0.01946743,0.002343336],"domain_scores_gemma":[0.7191433,0.22260612,0.01403035,0.020755922,0.01981633,0.0036479787],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2689332,0.0038350457,0.0033518393,0.018636068,0.008576431,0.016697757,0.0055311793,0.0044948384,0.0038796898],"category_scores_gemma":[0.1951218,0.0018825774,0.0026626457,0.008783014,0.03001605,0.018692387,0.020903345,0.006251897,0.0006536679],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001749896,0.00051598507,0.0052095824,0.0066943197,0.0004139565,0.0002675993,0.06809263,0.011691395,0.0015725098,0.5300408,0.0036154722,0.37171087],"study_design_scores_gemma":[0.00030173914,0.000638507,0.003366583,0.012001891,0.00033246793,0.00026144617,0.03142241,0.029005326,0.0035785607,0.8515329,0.06725599,0.000302093],"about_ca_topic_score_codex":0.0063801715,"about_ca_topic_score_gemma":0.008800229,"teacher_disagreement_score":0.7310668,"about_ca_system_score_codex":0.02068201,"about_ca_system_score_gemma":0.060063012,"threshold_uncertainty_score":0.9015355},"labels":[],"label_agreement":null},{"id":"W4413360974","doi":"10.1177/10982140251355140","title":"Theory-Practice Connections in Collaborative Approaches to Evaluation: A Systematic Review of Practice","year":2025,"lang":"en","type":"article","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Systematic review; Management science; Psychology; Evaluation methods; Program evaluation; Peer evaluation; Computer science; Engineering ethics; MEDLINE; Political science; Higher education; Engineering","score_opus":0.31953432024649814,"score_gpt":0.545800497569362,"score_spread":0.22626617732286386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413360974","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002803836,0.9818762,0.0053096046,0.0049554147,0.00074139115,0.0028263456,0.0001644794,0.000029983448,0.0012928611],"genre_scores_gemma":[0.09164884,0.8543649,0.039138954,0.0034526272,0.00037790355,0.0103789875,0.00032489208,0.000046535333,0.00026640284],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.67903095,0.18878175,0.08098328,0.009713273,0.039541982,0.0019487278],"domain_scores_gemma":[0.31693748,0.57669544,0.039141994,0.01381052,0.05115325,0.0022613208],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.26739362,0.001927317,0.0077699563,0.03953016,0.0036751586,0.010932877,0.0044685644,0.005181111,0.002344959],"category_scores_gemma":[0.56961,0.0025362414,0.0063104015,0.03251974,0.008446326,0.016302317,0.008976991,0.004955205,0.00036912868],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023790948,0.00011823367,0.0026574163,0.64708424,0.0037899453,0.00020148374,0.010814526,0.00055952714,0.00032097954,0.007614995,0.004296316,0.32230446],"study_design_scores_gemma":[0.00014972077,0.00015165994,0.0016618223,0.962301,0.004419383,0.00022354328,0.0040040673,0.00026876872,0.00022968098,0.003956664,0.0225723,0.00006132328],"about_ca_topic_score_codex":0.009200981,"about_ca_topic_score_gemma":0.0259329,"teacher_disagreement_score":0.7326064,"about_ca_system_score_codex":0.01938085,"about_ca_system_score_gemma":0.07264046,"threshold_uncertainty_score":0.9034341},"labels":[],"label_agreement":null}]}