{"id":"W4382239581","doi":"10.1609/aaai.v37i4.25527","title":"Graphs, Constraints, and Search for the Abstraction and Reasoning Corpus","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Theoretical computer science; Artificial intelligence; Graph; Beam search; Machine learning; Search algorithm; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003282489,0.000955604,0.0008924128,0.00189111,0.001301293,0.003210424,0.002370655,0.001873345,0.006751785],"category_scores_gemma":[0.01829238,0.0008312973,0.001496776,0.002814303,0.003316479,0.009216117,0.003176423,0.00364196,0.001077611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001992779,"about_ca_system_score_gemma":0.003097641,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007037593,"about_ca_topic_score_gemma":0.009937043,"domain_scores_codex":[0.9967046,0.001614511,0.0001676364,0.0006339666,0.0007236386,0.0001556798],"domain_scores_gemma":[0.9906887,0.006287751,0.0004790243,0.001845896,0.0004817722,0.0002169379],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005172952,0.000280062,0.002737389,0.001074189,0.0001231676,0.0003984194,0.001257505,0.2303223,0.00746648,0.4396184,0.02472809,0.2914768],"study_design_scores_gemma":[0.00009687468,0.00009016951,0.0005709148,0.00009137808,0.00004320538,0.0001463139,0.00044955,0.5234337,0.007652839,0.4444719,0.02290511,0.0000480185],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06371859,0.0008383889,0.9167995,0.002259314,0.00007119458,0.0003535136,0.001834972,0.006127964,0.007996583],"genre_scores_gemma":[0.2552053,0.0004149398,0.737051,0.0003619097,0.00002621735,0.0002600364,0.003710669,0.0008476023,0.002122478],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007037593,"threshold_uncertainty_score":0.02258694,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1176371003688176,"score_gpt":0.3206746685680134,"score_spread":0.2030375681991958,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}