{"id":"W6979286200","doi":"","title":"Towards a Probabilistic Framework for Analyzing and Improving LLM-Enabled Software","year":2025,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Agencia Nacional de Promoción Científica y Tecnológica; Consejo Nacional de Investigaciones Científicas y Técnicas; International Development Research Centre; Universidade Federal do Amazonas; Secretaría de Ciencia y Técnica, Universidad de Buenos Aires; Agencia Nacional de Investigación e Innovación","keywords":"Probabilistic logic; Identification (biology); Documentation; Reliability (semiconductor); Software; Strengths and weaknesses; Natural language; Software system","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02271205,0.00188528,0.001479105,0.006475952,0.001761945,0.005128094,0.004302582,0.002044063,0.001652935],"category_scores_gemma":[0.06533794,0.002363841,0.003923206,0.003044095,0.007884376,0.01097493,0.00615887,0.004985305,0.0006006729],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00448729,"about_ca_system_score_gemma":0.007167371,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007491486,"about_ca_topic_score_gemma":0.009684881,"domain_scores_codex":[0.9825494,0.008237717,0.001100667,0.002275256,0.005195619,0.0006413729],"domain_scores_gemma":[0.953029,0.02791421,0.00472512,0.00939913,0.004411388,0.0005211137],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006555369,0.0001168244,0.003468239,0.0003104004,0.0001638217,0.000227399,0.0006667362,0.4228732,0.005515244,0.4943254,0.001190878,0.07107641],"study_design_scores_gemma":[0.00001290928,0.00004262393,0.0002877548,0.00006453722,0.00003823357,0.00007718579,0.00007769626,0.6026022,0.003216127,0.3908023,0.002740278,0.00003804226],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001573092,0.00006556079,0.9974658,0.0002275409,0.000004179267,0.00002844274,0.00003494848,0.000346754,0.0002536996],"genre_scores_gemma":[0.0830006,0.0002361092,0.9153812,0.000175795,0.00004816208,0.0002402305,0.0001937444,0.0003129292,0.0004112759],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02271205,"threshold_uncertainty_score":0.1201143,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02669073692214376,"score_gpt":0.2107906475326801,"score_spread":0.1840999106105364,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}