{"id":"W2053901450","doi":"10.7771/1932-6246.1152","title":"On Evaluating Human Problem Solving of Computationally Hard Problems","year":2013,"lang":"en","type":"article","venue":"The Journal of Problem Solving","topic":"Multi-Criteria Decision Making","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Perception; Computational problem; Computational complexity theory; Euclidean geometry; Cognition; Computational model; Theoretical computer science; Artificial intelligence; Cognitive science; Mathematics; Algorithm; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01284532,0.001601947,0.00107049,0.003097907,0.000989033,0.00405158,0.00127355,0.002312345,0.005594075],"category_scores_gemma":[0.1488529,0.0003311097,0.0008219473,0.002264852,0.003429164,0.005120629,0.002622535,0.001590368,0.001126481],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001491747,"about_ca_system_score_gemma":0.0008032218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002239978,"about_ca_topic_score_gemma":0.001932781,"domain_scores_codex":[0.97877,0.01392874,0.001093772,0.001966391,0.003694369,0.0005466829],"domain_scores_gemma":[0.8804564,0.09754007,0.008888046,0.004506604,0.006166309,0.002442569],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002860605,0.001774481,0.1873003,0.00377293,0.00161184,0.0005815103,0.01412249,0.1512371,0.01046776,0.1034838,0.02450307,0.4982841],"study_design_scores_gemma":[0.0004625628,0.005235774,0.2460196,0.001386085,0.0003169607,0.001240326,0.01195097,0.3546579,0.007305088,0.3303195,0.04049224,0.0006129251],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7724397,0.008995779,0.1173168,0.003676564,0.000490647,0.0007949467,0.000850446,0.000541814,0.09489331],"genre_scores_gemma":[0.9573696,0.001858703,0.03623627,0.0005343871,0.000123244,0.0004668202,0.0007422291,0.0000967602,0.002571962],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01284532,"threshold_uncertainty_score":0.06793338,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2057732547937177,"score_gpt":0.4290050318615999,"score_spread":0.2232317770678822,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}