{"id":"W4412877152","doi":"10.1145/3711896.3736572","title":"The Hitchhikers Guide to Production-ready Trustworthy Foundation Model Powered Software (FMware)","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University; Queen's University; Huawei Technologies (Canada)","funders":"","keywords":"Trustworthiness; Foundation (evidence); Production (economics); Computer science; Software; Engineering; Computer security; Political science; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005589949,0.001459566,0.0005207647,0.003547447,0.001346283,0.007610593,0.002394265,0.004464998,0.02189667],"category_scores_gemma":[0.01381261,0.001840655,0.0008400349,0.002911191,0.004634892,0.009893182,0.003873236,0.006306145,0.01882426],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003257868,"about_ca_system_score_gemma":0.004262402,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007112769,"about_ca_topic_score_gemma":0.01003005,"domain_scores_codex":[0.9955363,0.00116487,0.0004405025,0.0004309808,0.002200359,0.0002269462],"domain_scores_gemma":[0.992804,0.004052154,0.0002415885,0.001071096,0.001296084,0.0005350825],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000306357,0.00007748832,0.0003399126,0.000555938,0.00002640856,0.0003253054,0.001311732,0.004429216,0.001187587,0.3384457,0.4038748,0.2493952],"study_design_scores_gemma":[0.000005675816,0.00001721205,0.00009361191,0.0002850773,0.000004045112,0.0002019364,0.0001342796,0.001582386,0.0002732003,0.07127767,0.9261092,0.00001571957],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.001659347,0.05615669,0.6782011,0.06117723,0.003606818,0.0003193731,0.001282747,0.007874066,0.1897227],"genre_scores_gemma":[0.02297759,0.06640001,0.7216474,0.01019543,0.002234778,0.0005579849,0.002352306,0.004780316,0.1688542],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.02189667,"threshold_uncertainty_score":0.07325161,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1455940437326464,"score_gpt":0.4601947726157387,"score_spread":0.3146007288830923,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}