{"id":"W4415169288","doi":"10.4204/eptcs.432","title":"Proceedings of the International Workshop on Verification of Scientific Software","year":2025,"lang":"en","type":"paratext","venue":"Electronic Proceedings in Theoretical Computer Science","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Correctness; Software; Software verification; Snapshot (computer storage); Set (abstract data type); Verification and validation","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008471057,0.001596284,0.001414796,0.001937912,0.001522811,0.007413326,0.002436514,0.002504777,0.06701934],"category_scores_gemma":[0.01562621,0.0008992305,0.002334806,0.001560827,0.002598453,0.006260284,0.00383991,0.005890901,0.023348],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002904871,"about_ca_system_score_gemma":0.003600365,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002568755,"about_ca_topic_score_gemma":0.002570847,"domain_scores_codex":[0.99174,0.002935917,0.0006996829,0.001224875,0.002886078,0.0005134537],"domain_scores_gemma":[0.9908064,0.004380358,0.0002150283,0.002013232,0.002064422,0.000520523],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003336219,0.0001491096,0.0005157395,0.0006965811,0.0001377453,0.0003505268,0.0006854733,0.006355649,0.003564557,0.2218357,0.4833759,0.2819994],"study_design_scores_gemma":[0.00004096132,0.00005680455,0.0003210504,0.0003235043,0.00003140991,0.0002489744,0.000091706,0.005752954,0.002114543,0.09506473,0.8959243,0.00002907203],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.005901679,0.03322426,0.5946811,0.03717592,0.04740887,0.0006385489,0.003677663,0.00568671,0.2716053],"genre_scores_gemma":[0.103298,0.03386152,0.3715591,0.007892246,0.02132304,0.001317402,0.0177197,0.007043476,0.4359854],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9915289,"threshold_uncertainty_score":0.224202,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02769949629412516,"score_gpt":0.3292721216118527,"score_spread":0.3015726253177275,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}