{"id":"W4413953529","doi":"10.14778/3742728.3742753","title":"Robust Plan Evaluation Based on Approximate Probabilistic Machine Learning","year":2025,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Rough Sets and Fuzzy Logic","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"IBM (Canada); University of Ottawa","funders":"","keywords":"Plan (archaeology); Probabilistic logic; Computer science; Machine learning; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000937417,0.0001398076,0.0001567323,0.0001087256,0.0001823877,0.0001126564,0.0008293201,0.00003877117,0.000009998591],"category_scores_gemma":[0.0002640941,0.0000884228,0.00007394229,0.0004442264,0.0000392972,0.0001203574,0.0002588639,0.0001730506,0.000003689221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001329829,"about_ca_system_score_gemma":0.00005971695,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001921889,"about_ca_topic_score_gemma":0.000001391306,"domain_scores_codex":[0.9986692,0.00002326601,0.0002352647,0.0003326748,0.0005307074,0.0002088931],"domain_scores_gemma":[0.9993613,0.00006084307,0.0001824651,0.0001926639,0.0001711918,0.00003158894],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002361324,0.001405008,0.0196474,0.001361338,0.000144598,0.000001418306,0.001897573,0.346534,0.007741678,0.5112867,0.005420078,0.1043241],"study_design_scores_gemma":[0.0005806446,0.0001326776,0.001548099,0.000181503,0.00002764999,0.000001054592,0.000025618,0.9785664,0.003952552,0.01430354,0.000581029,0.00009920854],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1076751,0.0009820121,0.1215602,0.03778073,0.003242617,0.01080081,0.00002404674,0.001242009,0.7166924],"genre_scores_gemma":[0.9874206,0.000005779293,0.01182327,0.0004324458,0.00002004963,0.0001186391,0.000002198334,0.000005969255,0.00017111],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8797454,"threshold_uncertainty_score":0.3605777,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03655889814260427,"score_gpt":0.2401569082094122,"score_spread":0.2035980100668079,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}