{"id":"W4413888047","doi":"10.1101/2025.08.28.672503","title":"Separating error from bias: A new framework for facial age estimation in humans and AIs","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Face recognition and analysis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Estimation; Computer science; Psychology; Artificial intelligence; Cognitive psychology; Statistics; Mathematics; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003977531,0.0003848631,0.0005584656,0.0004565099,0.0001631919,0.0006729597,0.0006509853,0.0004770027,0.00001492694],"category_scores_gemma":[0.0006013248,0.0004370034,0.0001586854,0.0007092415,0.00004245045,0.0002842141,0.0004919051,0.0005449994,0.00001522583],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001185814,"about_ca_system_score_gemma":0.0004914285,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003599826,"about_ca_topic_score_gemma":0.00006021036,"domain_scores_codex":[0.9976941,0.0001077714,0.0005358729,0.001050417,0.0002486837,0.0003631239],"domain_scores_gemma":[0.9983187,0.0002838768,0.0003239793,0.0007681466,0.0001268331,0.0001785068],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005135067,0.004586274,0.2131305,0.01165373,0.008023432,0.001284212,0.0109003,0.0527223,0.3197985,0.3411921,0.0153466,0.02084847],"study_design_scores_gemma":[0.002500623,0.0001119418,0.07115392,0.005204923,0.0004330552,8.478443e-9,0.00002752499,0.8693285,0.03722695,0.003879567,0.00718997,0.002943017],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2177775,0.0004043928,0.7796205,0.000599562,0.0005401921,0.0005536357,0.000228145,0.0002698469,0.000006218208],"genre_scores_gemma":[0.5301083,0.00005773443,0.4690801,0.0003884781,0.0001653682,0.0001549264,0.00000141778,0.00002326714,0.00002048543],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8166062,"threshold_uncertainty_score":0.9998082,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03839404834667662,"score_gpt":0.284073950998686,"score_spread":0.2456799026520094,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}