【发布时间】:2022-06-10 20:29:24
【问题描述】:
我这样训练了一个 xgb 模型:
candidates_var_train <- model.matrix(job_change ~ 0 + ., data = candidates_train)
candidates_train_xgb <- xgb.DMatrix(data = candidates_var_train,
label = ifelse(candidates_train$job_change == "Interested", 1, 0))
candidates_var_test <- model.matrix(job_change ~ 0 + ., data = candidates_test)
candidates_test_xgb <- xgb.DMatrix(data = candidates_var_test,
label = ifelse(candidates_test$job_change == "Interested", 1, 0))
获得了不错的 AUC,并希望将其应用于我的新数据集。新数据保存为数据框,并且与测试/训练数据具有相同的列,但目标变量“job_change”除外。我试图将它转换成这样的稀疏矩阵:
candidates_predict_sparse <- as(as.matrix(candidates_predict), "sparseMatrix")
candidates_predict_xgb <- xgb.DMatrix(data = candidates_predict_sparse)
但是在稀疏矩阵中引入了 NA,当我尝试使用 predict() 进行预测时,会出现以下错误:
Error in predict.xgb.Booster(xgb_model, newdata = candidates_predict_sparse, :
Feature names stored in `object` and `newdata` are different!
编辑:可重现的示例
最小数据集:
candidates_predict(我想要预测的数据集)
structure(list(enrollee_id = c(23427, 17605, 20912, 13948, 15205,
15140, 21736, 19800, 23755, 12148), city_development_index = c(0.698,
0.896, 0.754, 0.926, 0.92, 0.878, 0.926, 0.767, 0.689, 0.92),
gender = structure(c(4L, 4L, 4L, 2L, 2L, 2L, 2L, 2L, 2L,
2L), levels = c("Female", "Male", "Other", "keine Angabe"
), class = "factor"), enrolled_university = structure(c(4L,
2L, 1L, 2L, 1L, 3L, 3L, 2L, 2L, 2L), levels = c("Full time course",
"no_enrollment", "Part time course", "keine Angabe"), class = "factor"),
company_size = structure(c(9L, 9L, 9L, 5L, 3L, 9L, 3L, 6L,
2L, 9L), levels = c("<10", "10/49", "100-500", "1000-4999",
"10000+", "50-99", "500-999", "5000-9999", "keine Angabe"
), class = "factor"), company_type = structure(c(7L, 7L,
7L, 6L, 6L, 7L, 6L, 6L, 6L, 7L), levels = c("Early Stage Startup",
"Funded Startup", "NGO", "Other", "Public Sector", "Pvt Ltd",
"keine Angabe"), class = "factor"), last_new_job = structure(c(6L,
6L, 6L, 1L, 1L, 1L, 1L, 1L, 5L, 5L), levels = c("1", "2",
"3", "4", ">4", "never", "keine Angabe"), class = "factor"),
training_hours = c(63, 10, 46, 18, 55, 4, 324, 26, 140, 158
), education_detail = structure(c(8L, 7L, 7L, 21L, 8L, 22L,
7L, 7L, 7L, 19L), levels = c("Graduate Arts", "Graduate Business Degree",
"Graduate Humanities", "Graduate No Major", "Graduate no major discipline",
"Graduate Other", "Graduate STEM", "High School", "keine Angabe",
"Masters Arts", "Masters Business Degree", "Masters Humanities",
"Masters No Major", "Masters no major discipline", "Masters Other",
"Masters STEM", "Phd Arts", "Phd Business Degree", "Phd Humanities",
"Phd Other", "Phd STEM", "Primary School"), class = "factor"),
experience_detail = structure(c(23L, 23L, 23L, 23L, 23L,
21L, 23L, 17L, 10L, 23L), levels = c("<1", ">20", "1", "10",
"11", "12", "13", "14", "15", "16", "17", "18", "19", "2",
"20", "3", "4", "5", "6", "7", "8", "9", "no relevant experience"
), class = "factor")), row.names = c(NA, -10L), class = c("tbl_df",
"tbl", "data.frame"))
candidates_train(我用来训练 xgboost 模型的数据集)
structure(list(enrollee_id = c(26270, 3166, 20087, 8518, 8899,
25403, 14514, 3300, 10364, 5220), city_development_index = c(0.92,
0.887, 0.698, 0.92, 0.92, 0.92, 0.624, 0.84, 0.926, 0.754), gender = structure(c(1L,
2L, 2L, 2L, 4L, 2L, 2L, 4L, 4L, 2L), levels = c("Female", "Male",
"Other", "keine Angabe"), class = "factor"), enrolled_university = structure(c(2L,
2L, 2L, 2L, 2L, 2L, 1L, 2L, 2L, 2L), levels = c("Full time course",
"no_enrollment", "Part time course", "keine Angabe"), class = "factor"),
company_size = structure(c(7L, 9L, 1L, 9L, 9L, 3L, 9L, 2L,
5L, 9L), levels = c("<10", "10/49", "100-500", "1000-4999",
"10000+", "50-99", "500-999", "5000-9999", "keine Angabe"
), class = "factor"), company_type = structure(c(2L, 7L,
2L, 7L, 7L, 6L, 7L, 6L, 4L, 7L), levels = c("Early Stage Startup",
"Funded Startup", "NGO", "Other", "Public Sector", "Pvt Ltd",
"keine Angabe"), class = "factor"), last_new_job = structure(c(3L,
1L, 1L, 1L, 6L, 1L, 6L, 3L, 5L, 4L), levels = c("1", "2",
"3", "4", ">4", "never", "keine Angabe"), class = "factor"),
training_hours = c(127, 36, 7, 39, 53, 168, 111, 52, 107,
46), job_change = c("Interested", "Not interested", "Not interested",
"Not interested", "Not interested", "Not interested", "Not interested",
"Not interested", "Not interested", "Not interested"), education_detail = structure(c(3L,
7L, 16L, 22L, 22L, 3L, 8L, 7L, 8L, 6L), levels = c("Graduate Arts",
"Graduate Business Degree", "Graduate Humanities", "Graduate No Major",
"Graduate no major discipline", "Graduate Other", "Graduate STEM",
"High School", "keine Angabe", "Masters Arts", "Masters Business Degree",
"Masters Humanities", "Masters No Major", "Masters no major discipline",
"Masters Other", "Masters STEM", "Phd Arts", "Phd Business Degree",
"Phd Humanities", "Phd Other", "Phd STEM", "Primary School"
), class = "factor"), experience_detail = structure(c(17L,
5L, 18L, 23L, 23L, 14L, 23L, 8L, 5L, 2L), levels = c("<1",
">20", "1", "10", "11", "12", "13", "14", "15", "16", "17",
"18", "19", "2", "20", "3", "4", "5", "6", "7", "8", "9",
"no relevant experience"), class = "factor")), row.names = c(NA,
-10L), class = c("tbl_df", "tbl", "data.frame"), na.action = structure(c(`505` = 505L,
`688` = 688L, `1355` = 1355L, `1498` = 1498L, `1594` = 1594L,
`3607` = 3607L, `4897` = 4897L, `5743` = 5743L, `5863` = 5863L,
`5908` = 5908L, `6377` = 6377L, `7449` = 7449L, `7578` = 7578L
), class = "omit"))
candidates_test(我用来测试 xgboost 模型的数据集)
structure(list(enrollee_id = c(402, 27107, 8722, 6588, 4167,
19061, 17139, 14928, 10164, 8612), city_development_index = c(0.762,
0.92, 0.624, 0.926, 0.92, 0.926, 0.624, 0.92, 0.926, 0.92), gender = structure(c(2L,
2L, 4L, 2L, 4L, 2L, 4L, 2L, 2L, 4L), levels = c("Female", "Male",
"Other", "keine Angabe"), class = "factor"), enrolled_university = structure(c(2L,
2L, 1L, 2L, 2L, 2L, 3L, 2L, 2L, 2L), levels = c("Full time course",
"no_enrollment", "Part time course", "keine Angabe"), class = "factor"),
company_size = structure(c(1L, 6L, 9L, 2L, 6L, 3L, 7L, 3L,
3L, 9L), levels = c("<10", "10/49", "100-500", "1000-4999",
"10000+", "50-99", "500-999", "5000-9999", "keine Angabe"
), class = "factor"), company_type = structure(c(6L, 6L,
7L, 6L, 6L, 6L, 6L, 6L, 6L, 7L), levels = c("Early Stage Startup",
"Funded Startup", "NGO", "Other", "Public Sector", "Pvt Ltd",
"keine Angabe"), class = "factor"), last_new_job = structure(c(5L,
1L, 6L, 5L, 6L, 2L, 1L, 3L, 4L, 4L), levels = c("1", "2",
"3", "4", ">4", "never", "keine Angabe"), class = "factor"),
training_hours = c(18, 46, 26, 18, 106, 50, 148, 40, 42,
50), job_change = c("Interested", "Interested", "Not interested",
"Not interested", "Not interested", "Not interested", "Interested",
"Not interested", "Interested", "Not interested"), education_detail = structure(c(7L,
7L, 8L, 7L, 7L, 16L, 7L, 7L, 21L, 7L), levels = c("Graduate Arts",
"Graduate Business Degree", "Graduate Humanities", "Graduate No Major",
"Graduate no major discipline", "Graduate Other", "Graduate STEM",
"High School", "keine Angabe", "Masters Arts", "Masters Business Degree",
"Masters Humanities", "Masters No Major", "Masters no major discipline",
"Masters Other", "Masters STEM", "Phd Arts", "Phd Business Degree",
"Phd Humanities", "Phd Other", "Phd STEM", "Primary School"
), class = "factor"), experience_detail = structure(c(7L,
20L, 23L, 10L, 3L, 5L, 8L, 2L, 2L, 23L), levels = c("<1",
">20", "1", "10", "11", "12", "13", "14", "15", "16", "17",
"18", "19", "2", "20", "3", "4", "5", "6", "7", "8", "9",
"no relevant experience"), class = "factor")), row.names = c(NA,
-10L), class = c("tbl_df", "tbl", "data.frame"), na.action = structure(c(`531` = 531L,
`615` = 615L, `715` = 715L, `1000` = 1000L, `1148` = 1148L, `1318` = 1318L,
`1416` = 1416L), class = "omit"))
使用的库
library(Matrix)
library(xgboost)
library(dplyr)
library(readr)
【问题讨论】:
-
你能提供一个可重现的例子吗? stackoverflow.com/questions/5963269/…
-
@tavdp 我尽可能在问题中添加了一个可重现的示例,请告诉我是否缺少任何内容
-
您的 xgb_model / 获取方式丢失,因此无法重现。我怀疑问题是您在“candidates_train_xgb”上进行训练,这会导致 xgb_model 由于该模型而需要 73 个特征。矩阵将因子扩展为一组虚拟变量(数据集中每个唯一条目对应一列),但是“ Candidates_predict_sparse" 只有 10 个,因为这些特征没有被虚拟化。
-
我会把它制定成答案
标签: r machine-learning xgboost