{"id":"W4411523900","doi":"10.2478/aoj-2025-0017","title":"How reliable are AI-Assisted cephalometric programs in assessing measurements involving bilateral landmarks?","year":2025,"lang":"en","type":"article","venue":"Australasian Orthodontic Journal","topic":"Dental Radiography and Imaging","field":"Dentistry","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Cephalometric analysis; Reliability (semiconductor); Computer science; Landmark; Orthodontics; Gonial angle; Artificial intelligence; Medicine; Radiography; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0187184,0.0006101502,0.000729747,0.002311219,0.0002518727,0.001510973,0.00105507,0.0008026807,0.001362815],"category_scores_gemma":[0.07404103,0.0004729762,0.0003384512,0.002049995,0.0007877756,0.001505516,0.0008303021,0.000394043,0.0007716704],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000206752,"about_ca_system_score_gemma":0.0004305634,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001072157,"about_ca_topic_score_gemma":0.002008927,"domain_scores_codex":[0.9877108,0.006292813,0.001159593,0.001120681,0.003485431,0.0002306436],"domain_scores_gemma":[0.9335093,0.03770491,0.007457145,0.004865668,0.01596409,0.0004988589],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001507915,0.0001226508,0.4627028,0.0008620746,0.0004396371,0.0001962383,0.001029416,0.003951909,0.02551633,0.0008637913,0.001813637,0.5009935],"study_design_scores_gemma":[0.0001860942,0.001831635,0.8715478,0.0005329648,0.0006820853,0.002517033,0.00201734,0.08322974,0.02624286,0.002229565,0.008756357,0.0002264714],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8506286,0.004866472,0.1352496,0.0006211072,0.0002214258,0.0003288461,0.0007274828,0.0009231137,0.00643328],"genre_scores_gemma":[0.9425989,0.0005964309,0.05592697,0.00006470464,0.00006328175,0.0000917018,0.0002142702,0.00006456496,0.0003792016],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9812816,"threshold_uncertainty_score":0.09899348,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06700353739115038,"score_gpt":0.3150510142281558,"score_spread":0.2480474768370054,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}