{"id":"W2774512527","doi":"","title":"MaxSAT Evaluation 2017: Solver and Benchmark Descriptions","year":2017,"lang":"en","type":"article","venue":"","topic":"Constraint Satisfaction and Optimization","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Benchmark (surveying); Computer science; Maximum satisfiability problem; Solver; Artificial intelligence; Programming language; Algorithm; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002729553,0.00005363638,0.00004692345,0.00004681583,0.0005130339,0.000503996,0.0002108696,0.00003079386,0.000399901],"category_scores_gemma":[0.00007671781,0.00004895019,0.00001750678,0.00002794294,0.00005250582,0.001042812,0.0001052272,0.00003857474,0.00004241187],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002134012,"about_ca_system_score_gemma":0.00004356368,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004400402,"about_ca_topic_score_gemma":0.0001252623,"domain_scores_codex":[0.9994624,0.00002315394,0.00008328077,0.0001790093,0.0001647022,0.00008743876],"domain_scores_gemma":[0.9993357,0.00001455905,0.00006182595,0.0004356747,0.0000994528,0.00005275907],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002297571,0.00002663776,0.02304978,0.00000456312,0.00001645807,0.000002085686,0.0004146184,0.0002746634,0.0007996447,0.1935968,0.01160035,0.7702121],"study_design_scores_gemma":[0.0003699631,0.0000146171,0.4170848,0.000006741622,0.00001009404,0.00001852258,0.00002551771,0.5728976,0.0001482008,0.006607396,0.002694164,0.0001224067],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009527037,0.00003260977,0.8923795,0.003114423,0.0005500226,0.0001713074,8.750106e-7,0.0000775436,0.09414674],"genre_scores_gemma":[0.9408242,0.00003579559,0.05779776,0.0001817816,0.00002543727,0.000009168411,0.000002673872,0.000001958916,0.001121207],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9312972,"threshold_uncertainty_score":0.4860046,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05351257202851611,"score_gpt":0.3038087786782595,"score_spread":0.2502962066497433,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}