{"id":"W6949154600","doi":"10.5281/zenodo.15285528","title":"Bongard in Wonderland: Visual Puzzles that Still Make AI Go Mad?","year":2024,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Feature (linguistics); Visualization; Perspective (graphical); Set (abstract data type); Focus (optics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002513823,0.0005983912,0.00041592,0.001075207,0.006164179,0.0131959,0.001015027,0.004023428,0.04412422],"category_scores_gemma":[0.01151672,0.0003757748,0.0003943242,0.0008747614,0.01605179,0.01527505,0.004759939,0.005320708,0.003986019],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002998712,"about_ca_system_score_gemma":0.002235803,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01150655,"about_ca_topic_score_gemma":0.0175177,"domain_scores_codex":[0.998413,0.0009885558,0.00002165308,0.0001363724,0.0002036641,0.0002367219],"domain_scores_gemma":[0.9974068,0.001440907,0.0001339901,0.0003581281,0.000252908,0.000407314],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001549974,0.0000246725,0.0005872346,0.00008296396,0.00001410015,0.0003203577,0.02314245,0.0005628072,0.0003733753,0.7999261,0.1487195,0.02609145],"study_design_scores_gemma":[0.00002984974,0.00001460508,0.0004276795,0.0001878037,0.00001151768,0.0002634613,0.02599031,0.0008479566,0.0003172203,0.5717931,0.4000814,0.00003524408],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.03688599,0.006968134,0.01539588,0.2907996,0.005088257,0.00002306554,0.0003322409,0.0008009553,0.6437058],"genre_scores_gemma":[0.8859279,0.002994751,0.006514809,0.01561741,0.0007425737,0.0000497123,0.0002139122,0.001412284,0.08652667],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04412422,"threshold_uncertainty_score":0.1476102,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06833901653629182,"score_gpt":0.3557452636361113,"score_spread":0.2874062470998194,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}