{"id":"W4241594092","doi":"10.6028/nist.ir.7310","title":"Evaluating reasoning systems","year":2006,"lang":"en","type":"report","venue":"","topic":"Complex Systems and Decision Making","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":["metaresearch","insufficient_payload"],"category_scores_codex":[0.03683748,0.000547167,0.001752579,0.001154085,0.0004372168,0.003060249,0.001796997,0.0005308797,0.002107525],"category_scores_gemma":[0.02775668,0.0003578486,0.0006528345,0.001383612,0.00005572084,0.0002427875,0.0006978346,0.0005655324,0.002300382],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000489763,"about_ca_system_score_gemma":0.001892204,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007401427,"about_ca_topic_score_gemma":0.0003346638,"domain_scores_codex":[0.9737312,0.0007613736,0.004088159,0.001759852,0.01902211,0.0006373682],"domain_scores_gemma":[0.9838445,0.004852546,0.002780766,0.002631315,0.00568825,0.0002026003],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000004397511,0.00001395555,0.001259907,0.00002838655,0.00002308303,0.00004693403,0.00002106029,0.003779764,0.00001332525,0.003985626,0.956136,0.03468751],"study_design_scores_gemma":[0.0001746032,0.00005017464,0.00183852,0.0006140358,0.00003324731,0.0003862278,0.0003037577,0.05752146,0.000001203797,0.005279223,0.9332646,0.0005329269],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.001047519,0.01134047,0.03395952,0.00001979083,0.01130841,0.00061599,0.0000421058,0.0002204446,0.9414458],"genre_scores_gemma":[0.1758347,0.00001235638,0.005039227,0.00003036599,0.004001898,0.00006222536,0.00001876071,0.00009373367,0.8149067],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1747872,"threshold_uncertainty_score":0.9998873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5543361083301273,"score_gpt":0.5519882853164297,"score_spread":0.002347823013697514,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}