{"id":"W2086886584","doi":"10.1145/1538902.1538906","title":"Empirical hardness models","year":2009,"lang":"en","type":"article","venue":"Journal of the ACM","topic":"Auction Theory and Applications","field":"Decision Sciences","cited_by":110,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Division of Information and Intelligent Systems; Defense Advanced Research Projects Agency","keywords":"Computer science; Benchmark (surveying); Suite; Machine learning; Artificial intelligence; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01056809,0.002053768,0.002355908,0.003657723,0.001293972,0.004810638,0.00563639,0.003521065,0.01499702],"category_scores_gemma":[0.08689384,0.001194237,0.002693808,0.002631382,0.0039288,0.01149124,0.002391045,0.006918444,0.00259283],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003586657,"about_ca_system_score_gemma":0.001700116,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002436,"about_ca_topic_score_gemma":0.00288833,"domain_scores_codex":[0.9917671,0.003030064,0.0004864,0.002175618,0.001670032,0.0008707102],"domain_scores_gemma":[0.9032117,0.07423607,0.006155157,0.01124255,0.003711027,0.001443628],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002749367,0.0003150503,0.01477155,0.0004659286,0.0002629574,0.0001707315,0.0003694289,0.556791,0.0004170619,0.361701,0.02289889,0.04156154],"study_design_scores_gemma":[0.00004347104,0.00007389132,0.002034805,0.0000644387,0.000038374,0.0001291977,0.00006297457,0.7289229,0.0001927343,0.2641589,0.004243835,0.00003440768],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1006465,0.002323694,0.8442644,0.007641329,0.0003691627,0.0004483114,0.004835372,0.001980562,0.03749061],"genre_scores_gemma":[0.8649176,0.001830989,0.1083161,0.001865554,0.0009481991,0.001610045,0.006648656,0.0007323038,0.01313047],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01499702,"threshold_uncertainty_score":0.05589008,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3077122164162031,"score_gpt":0.4750898193824556,"score_spread":0.1673776029662526,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}