{"id":"W843694097","doi":"10.1007/978-4-431-66996-8_24","title":"Construct Comparability in International Assessments: Comparability of English and French versions of TIMSS","year":2003,"lang":"en","type":"book-chapter","venue":"","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Comparability; Differential item functioning; Construct (python library); Item response theory; Construct validity; Statistics; Psychology; Test (biology); International comparisons; Econometrics; Mathematics; Mathematics education; Psychometrics; Computer science; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05519048,0.000655696,0.0007051045,0.005685163,0.001249296,0.00542916,0.001109234,0.000874934,0.003106481],"category_scores_gemma":[0.1769606,0.0003955805,0.001519282,0.005656515,0.002345525,0.005849692,0.002691526,0.001577795,0.0004902179],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003291545,"about_ca_system_score_gemma":0.003365045,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01573533,"about_ca_topic_score_gemma":0.02115423,"domain_scores_codex":[0.9625626,0.02456319,0.00279012,0.002410348,0.006875897,0.0007977591],"domain_scores_gemma":[0.8767315,0.0940036,0.004996889,0.007702881,0.01587611,0.0006891272],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0005678211,0.0001367473,0.1465452,0.0007892092,0.0007663345,0.0001772925,0.03535508,0.001567326,0.002068545,0.07084356,0.01234781,0.7288351],"study_design_scores_gemma":[0.0001207224,0.0007333888,0.7872952,0.004444608,0.001431679,0.001159025,0.02523108,0.005604635,0.006669723,0.09190214,0.07507356,0.0003342765],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6715685,0.037927,0.1447651,0.009704672,0.00219253,0.0005570501,0.002719524,0.000618218,0.1299474],"genre_scores_gemma":[0.954109,0.002605832,0.03654056,0.0006270803,0.0002299383,0.0004127747,0.001485402,0.0003162313,0.003673055],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05519048,"threshold_uncertainty_score":0.2918787,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4229317062111467,"score_gpt":0.4744835525279423,"score_spread":0.0515518463167956,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}