{"id":"W7106815409","doi":"10.48448/v7rn-9t93","title":"STRICT: Stress-Test of Rendering Image Containing Text","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute; University of Toronto","funders":"","keywords":"Rendering (computer graphics); Correctness; Legibility; Locality; Consistency (knowledge bases); Benchmark (surveying)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001942344,0.001372063,0.0005377771,0.0007302757,0.0005467445,0.002053424,0.002755795,0.001733913,0.009048057],"category_scores_gemma":[0.01743771,0.0004589031,0.0007920959,0.0004674472,0.001139597,0.002219759,0.002032247,0.002019256,0.0020599],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001064937,"about_ca_system_score_gemma":0.0008512666,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00525547,"about_ca_topic_score_gemma":0.004864012,"domain_scores_codex":[0.9982554,0.0004032479,0.0001143604,0.0003486503,0.0007257452,0.000152533],"domain_scores_gemma":[0.9920472,0.004888095,0.000265861,0.001568125,0.0008919946,0.0003386802],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002939999,0.001994602,0.009332311,0.002207689,0.0003832662,0.001086127,0.001682451,0.4896096,0.09093507,0.03666253,0.07206755,0.2910988],"study_design_scores_gemma":[0.0001788372,0.0008311544,0.001460741,0.0001019832,0.00004590716,0.0002042921,0.0002426088,0.9180458,0.05852124,0.009185409,0.01110951,0.00007243641],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6544175,0.002073604,0.235677,0.00168346,0.001852802,0.0007462148,0.005213343,0.04903805,0.04929801],"genre_scores_gemma":[0.8878525,0.0004093996,0.09267861,0.0005436295,0.00010318,0.00029129,0.004739841,0.006274501,0.007107114],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.009048057,"threshold_uncertainty_score":0.03026873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03748428658425577,"score_gpt":0.3275957515859995,"score_spread":0.2901114650017437,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}