{"id":"W4412887682","doi":"10.18653/v1/2025.findings-acl.1134","title":"Small Encoders Can Rival Large Decoders in Detecting Groundedness","year":2025,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Encoder; Computer science; Decoding methods; Algorithm","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005027584,0.001769398,0.001093758,0.001356972,0.0006137468,0.002323531,0.002202496,0.001574421,0.005323889],"category_scores_gemma":[0.02430785,0.001186313,0.001009867,0.001063357,0.001156246,0.007238635,0.003222714,0.003420665,0.003793203],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001451743,"about_ca_system_score_gemma":0.002252158,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007939496,"about_ca_topic_score_gemma":0.01927111,"domain_scores_codex":[0.9978907,0.0008497755,0.0001439168,0.0005588284,0.0004021556,0.0001546515],"domain_scores_gemma":[0.9841144,0.01200342,0.0003833188,0.002285594,0.0008858016,0.000327442],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001513184,0.000469529,0.01107667,0.001241785,0.0004536865,0.0003890621,0.0009272963,0.06440128,0.053726,0.01759442,0.02315408,0.825053],"study_design_scores_gemma":[0.0001222908,0.0002238481,0.001690804,0.00009812898,0.0002214881,0.0002130008,0.0002320624,0.925849,0.02586378,0.03649314,0.008933618,0.00005890072],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09585993,0.004063456,0.8493887,0.001829976,0.0002747466,0.0003840899,0.001957598,0.04150813,0.004733485],"genre_scores_gemma":[0.587005,0.001337409,0.3994875,0.0009313372,0.0001939633,0.0004004929,0.004531214,0.001688086,0.004425092],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007939496,"threshold_uncertainty_score":0.02658868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02872723235680232,"score_gpt":0.2571890027726212,"score_spread":0.2284617704158189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}