{"id":"W2079281756","doi":"10.1119/1.3602073","title":"The puzzling reliability of the Force Concept Inventory","year":2011,"lang":"en","type":"article","venue":"American Journal of Physics","topic":"Science Education and Pedagogy","field":"Social Sciences","cited_by":80,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Concordia University; Vanier College; John Abbott College","funders":"","keywords":"Reliability (semiconductor); Internal consistency; Consistency (knowledge bases); Test (biology); Physics; Statistics; Psychology; Reliability engineering; Econometrics; Psychometrics; Mathematics; Computer science; Thermodynamics; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1015308,0.0005220208,0.001030986,0.003347829,0.001069671,0.001792438,0.002217629,0.0009331609,0.00140968],"category_scores_gemma":[0.3212998,0.0007211551,0.0006706754,0.002075139,0.004361013,0.001450474,0.00212569,0.001942132,0.0008455153],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001150424,"about_ca_system_score_gemma":0.001220984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004318098,"about_ca_topic_score_gemma":0.003741426,"domain_scores_codex":[0.868227,0.07136546,0.008524807,0.01184965,0.03842546,0.001607505],"domain_scores_gemma":[0.6106368,0.2937903,0.01310539,0.03463989,0.04557689,0.002250814],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006576263,0.0002326706,0.7642898,0.0008585452,0.0009104725,0.0002890083,0.0129342,0.001742515,0.002058185,0.01323762,0.01580292,0.1869864],"study_design_scores_gemma":[0.00009676517,0.0006982984,0.9374341,0.0009307346,0.0002896217,0.001108631,0.004908008,0.008652443,0.004536765,0.01777385,0.02338393,0.0001866852],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8919583,0.006760759,0.0530577,0.007701181,0.002792574,0.0004907396,0.001399307,0.0002903395,0.03554924],"genre_scores_gemma":[0.9907153,0.0004870001,0.00603192,0.0006618273,0.0003689901,0.0002927321,0.0005231676,0.0001174845,0.0008016921],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8984692,"threshold_uncertainty_score":0.5369526,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06388540973370929,"score_gpt":0.3546067386217578,"score_spread":0.2907213288880485,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}