{"id":"W3010661194","doi":"10.1109/tifs.2020.2980190","title":"<i>CPA</i>: Accurate <i>C</i>ross-<i>P</i>latform Binary <i>A</i>uthorship Characterization Using LDA","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Information Forensics and Security","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"United Arab Emirates University","keywords":"Computer science; Binary number; Code refactoring; Binary code; Robustness (evolution); Compiler; Code (set theory); Coding (social sciences); Artificial intelligence; Natural language processing; Information retrieval; Set (abstract data type); Programming language; Software; Arithmetic","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002076284,0.002013617,0.001126508,0.006799058,0.001630282,0.004685813,0.001624027,0.001800301,0.02318537],"category_scores_gemma":[0.01739396,0.0007797427,0.00132302,0.003265583,0.0006934402,0.003962092,0.003103446,0.002207838,0.03754347],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001280588,"about_ca_system_score_gemma":0.001795459,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006615321,"about_ca_topic_score_gemma":0.01144674,"domain_scores_codex":[0.9971885,0.0003343083,0.000268152,0.0009327357,0.00104859,0.0002276604],"domain_scores_gemma":[0.9895064,0.001846532,0.001001803,0.003793131,0.003503148,0.0003490154],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004560386,0.0001419889,0.01957804,0.0004150428,0.0001308681,0.0002762356,0.000233326,0.006866023,0.01509351,0.007380615,0.3620257,0.5874026],"study_design_scores_gemma":[0.00007134172,0.00008934801,0.01917148,0.000179446,0.00006312382,0.0009410254,0.0003913664,0.6685562,0.071611,0.02883255,0.2099006,0.0001926087],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04023239,0.0009140787,0.6420715,0.001629206,0.001194564,0.000683579,0.0480631,0.237927,0.02728453],"genre_scores_gemma":[0.3078514,0.0005873502,0.5543268,0.0007856212,0.0005191537,0.0009580235,0.08911522,0.01250446,0.03335195],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02318537,"threshold_uncertainty_score":0.07756281,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02852679818523693,"score_gpt":0.2492858496141107,"score_spread":0.2207590514288738,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}