{"id":"W6967018607","doi":"10.48448/hs80-1a56","title":"GAIA Search: Hugging Face and Pyserini Interoperability for NLP Training Data Exploration","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Interoperability; Leverage (statistics); Face (sociological concept); Analytics; Scripting language; Operationalization; Domain (mathematical analysis); Scalability; Data curation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.005330571,0.0003972003,0.0004440658,0.0008763716,0.0006529023,0.0004587406,0.002445708,0.0001112552,0.001563309],"category_scores_gemma":[0.0009942012,0.000386405,0.00003730307,0.001118618,0.001701077,0.001473952,0.002496377,0.000512068,0.0001273732],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004259177,"about_ca_system_score_gemma":0.0009665128,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003093585,"about_ca_topic_score_gemma":0.001478151,"domain_scores_codex":[0.9957339,0.0001778574,0.0004288247,0.001921021,0.0009503447,0.0007881232],"domain_scores_gemma":[0.9972755,0.0001804978,0.0002611649,0.00190924,0.0001271602,0.0002464376],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003694741,0.0009225715,0.001936795,0.001383379,0.0003095399,0.00003140228,0.04853812,0.003255608,0.03654705,0.007317349,0.4820925,0.4172962],"study_design_scores_gemma":[0.0008972069,0.0003178944,0.00007419563,0.0001764003,0.00005609651,0.00002092564,0.02376749,0.323417,0.000128134,0.0005064916,0.6497589,0.0008792686],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.05036613,0.004694075,0.3326131,0.009871865,0.009848339,0.02985298,0.04877647,0.01392728,0.5000498],"genre_scores_gemma":[0.6779847,0.0002548838,0.155341,0.00100061,0.002053335,0.0007836212,0.01288913,0.004768508,0.1449243],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6276186,"threshold_uncertainty_score":0.9998588,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3765175985623099,"score_gpt":0.4075187840282223,"score_spread":0.03100118546591246,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}