{"id":"W2115215982","doi":"10.14778/1453856.1453883","title":"Hashed samples","year":2008,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":68,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; York University","funders":"","keywords":"Similarity (geometry); Estimator; A priori and a posteriori; Computer science; Overhead (engineering); Set (abstract data type); Sampling (signal processing); Cosine similarity; Algorithm; Data mining; Pattern recognition (psychology); Mathematics; Artificial intelligence; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004565466,0.0006726794,0.001362233,0.002230134,0.000921086,0.002051435,0.002230327,0.001123271,0.004500926],"category_scores_gemma":[0.03776337,0.0005905165,0.000822903,0.002566817,0.001185464,0.005785041,0.002667209,0.001325861,0.001692427],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001066717,"about_ca_system_score_gemma":0.001060048,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008847294,"about_ca_topic_score_gemma":0.0009028299,"domain_scores_codex":[0.9945505,0.001136465,0.000413102,0.001123357,0.002431354,0.0003453371],"domain_scores_gemma":[0.9784055,0.01173603,0.001616584,0.005762353,0.002085104,0.0003944561],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001852968,0.000370091,0.01649695,0.0005989676,0.0002148063,0.0003696934,0.0008046987,0.1180419,0.02576906,0.1614239,0.009754835,0.6643022],"study_design_scores_gemma":[0.0001393787,0.0004997972,0.001761964,0.00004850404,0.00005759515,0.0008079214,0.0002721543,0.8574864,0.02977708,0.09955601,0.009540191,0.00005301665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02584472,0.000289877,0.970735,0.0002005079,0.00005993759,0.0001841746,0.0003463727,0.00119614,0.001143298],"genre_scores_gemma":[0.443255,0.0003471438,0.5510418,0.0002990266,0.0002459524,0.0004468014,0.001448263,0.0002319907,0.002684038],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004565466,"threshold_uncertainty_score":0.02414477,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2910039101624485,"score_gpt":0.3717421828373335,"score_spread":0.08073827267488504,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}