{"id":"W4241594092","doi":"10.6028/nist.ir.7310","title":"Evaluating reasoning systems","year":2006,"lang":"en","type":"report","venue":"","topic":"Complex Systems and Decision Making","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02506703,0.002015785,0.001578295,0.007365448,0.001312378,0.008736071,0.002264695,0.002435325,0.01019961],"category_scores_gemma":[0.1650161,0.0004813167,0.001391933,0.003589301,0.003604517,0.01028905,0.004708305,0.001730033,0.001648167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003360099,"about_ca_system_score_gemma":0.003211597,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001788097,"about_ca_topic_score_gemma":0.001513758,"domain_scores_codex":[0.953469,0.02244175,0.003776933,0.004129794,0.01501325,0.001169334],"domain_scores_gemma":[0.8824111,0.08462223,0.006930008,0.008761579,0.01485665,0.00241839],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009221697,0.0004486277,0.0224939,0.002662389,0.0009041302,0.0002676402,0.001881349,0.08846412,0.003075612,0.3576688,0.009462105,0.5117491],"study_design_scores_gemma":[0.0001939221,0.0009371073,0.005916835,0.0008922009,0.0005640202,0.0003352471,0.001459802,0.2049206,0.006073837,0.7516909,0.02688634,0.0001291676],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.1425615,0.01041012,0.7487862,0.00534382,0.0008497936,0.001979985,0.002448626,0.002365305,0.08525462],"genre_scores_gemma":[0.6626409,0.002803883,0.3262683,0.0004240685,0.0002946341,0.0006688425,0.002931993,0.0002036209,0.003763676],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.02506703,"threshold_uncertainty_score":0.1325687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5543361083301273,"score_gpt":0.5519882853164297,"score_spread":0.002347823013697514,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}