{"id":"W4411523900","doi":"10.2478/aoj-2025-0017","title":"How reliable are AI-Assisted cephalometric programs in assessing measurements involving bilateral landmarks?","year":2025,"lang":"en","type":"article","venue":"Australasian Orthodontic Journal","topic":"Dental Radiography and Imaging","field":"Dentistry","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Cephalometric analysis; Reliability (semiconductor); Computer science; Landmark; Orthodontics; Gonial angle; Artificial intelligence; Medicine; Radiography; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001379814,0.0003659501,0.0005426236,0.002012086,0.0003930376,0.003252746,0.0004688348,0.0001997126,0.00006569111],"category_scores_gemma":[0.0003152496,0.0003424523,0.0003611879,0.00349273,0.00009339541,0.002027978,0.00008777764,0.001277893,0.0000157448],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003502479,"about_ca_system_score_gemma":0.0001435361,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009384305,"about_ca_topic_score_gemma":0.0003191039,"domain_scores_codex":[0.9967771,0.0003162238,0.0007695425,0.000474132,0.0007183484,0.0009447188],"domain_scores_gemma":[0.9986876,0.0000604825,0.0004489224,0.0003439017,0.00021211,0.0002469404],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00003019028,0.0002563576,0.9614645,0.0001141627,0.0001633341,0.001147178,0.00006826877,0.00001117572,0.001123497,0.00002991804,0.001551601,0.0340398],"study_design_scores_gemma":[0.001933928,0.00004134175,0.9913249,0.001610733,0.0001262893,0.001189345,0.001333092,0.00009759296,0.0003080866,0.0003146448,0.001397625,0.0003224146],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9895825,0.001352536,0.002401442,0.0009938078,0.002488879,0.0003258527,0.000002720262,0.0001057185,0.002746514],"genre_scores_gemma":[0.9925467,0.00004617112,0.003757456,0.0001699796,0.0001635593,0.00001275372,0.00001236994,0.00003836981,0.003252607],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03371738,"threshold_uncertainty_score":0.9999027,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06700353739115038,"score_gpt":0.3150510142281558,"score_spread":0.2480474768370054,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}