资讯动态

机器学习6:集成学习:集成学习思想、随机森林算法、Adaboost算法、GBDT、XGBoost

发布时间:2026/10/5 10:23:40 来源:尧图企业网站定制
集成学习思想随机森林算法 案例集成学习算法之Bagging思想 随机森林算法演示 集成学习 概述把多个弱学习器组成一个强学习器的过程-集成学习 思想 Bagging思想 1.有放回的随机抽取样本 2.平权投票 3.可以并行执行 Boosting思想 1.每次训练都会使用全部样本 2.加权投票-预测正确权重降低预测错误权重增加 3.只能串行执行 Bagging思想代表 随机森林算法 随机森林算法: 1.每个弱学习器都是CART树必须是二叉树 2.有放回的随机抽样平权投票并行执行 #导包 import numpy as np import pandas as pd from sklearn.model_selection import train_test_split#划分数据集 from sklearn.ensemble import RandomForestClassifier#随机森林分类器 from sklearn.tree import DecisionTreeClassifier#决策树分类器 from sklearn.model_selection import GridSearchCV#网格搜索 #1.加载数据 datapd.read_csv(./datas/train.csv) data.info() #2.数据预处理 #2.1抽取特征和标签 xdata[[Pclass,Sex,Age]].copy() ydata[Survived] #2.2空值处理用Age列的平均值填充缺失值 x[Age]x[Age].fillna(x[Age].mean()) #2.3热编码处理 xpd.get_dummies(x) #2.4划分训练集和测试集 x_train,x_test,y_train,y_testtrain_test_split(x,y,test_size0.2,random_state10) #3.特征工程 #4.模型训练预测评估 #场景1单一决策树 #4.1创建决策树对象演示单一的决策树效果 estimator1DecisionTreeClassifier() #4.2模型训练 estimator1.fit(x_train,y_train) #4.3模型预测 y_pre1estimator1.predict(x_test) print(f模型1预测结果{y_pre1}) #4.4模型评估 print(f决策树模型的评估准确率为{estimator1.score(x_test,y_test)}) print(-*23) #场景2随机森林算法-采用默认参数 #4.1创建随机森林对象演示多个的决策树(Bagging思想效果 estimator2RandomForestClassifier()#n_estimators100表示有100个决策树,max_depth10表示最大深度为10,绘制决策树时最多10层 #4.2模型训练 estimator2.fit(x_train,y_train) #4.3模型预测 y_pre2estimator2.predict(x_test) print(f模型2预测结果{y_pre2}) #4.4模型评估 print(f随机森林模型的评估准确率为{estimator2.score(x_test,y_test)}) print(-*23) #场景3随机森林算法-网格搜索 #4.1创建随机森林对象演示多个的决策树(Bagging思想效果 estimator3RandomForestClassifier()#n_estimators100表示有100个决策树,max_depth10表示最大深度为10,绘制决策树时最多10层 #4.2参数准备 params{n_estimators:[30,50,60,90,100],max_depth:[2,3,5,7]} #4.3创建网格搜索对象结合交叉验证 gs_estimatorGridSearchCV(estimator3,param_gridparams,cv5) #4.4模型训练 gs_estimator.fit(x_train,y_train) #4.5模型预测 y_pre3gs_estimator.predict(x_test) print(f模型3预测结果{y_pre3}) #4.6模型评估 print(f随机森林模型的评估准确率为{gs_estimator.score(x_test,y_test)}) #4.7查看最佳参数 print(f最佳参数为{gs_estimator.best_params_})Adaboost算法PS:下面的以5.5来分时有6个样本分类错误下面写错了案例AdaBoost实战葡萄酒数据Adaptive Boost自适应提升逐步调整权重-对下降错上升 案例 演示AdaBoost算法之葡萄酒案例 AdaBoost算法介绍 它属于Boosting思想即串行执行每次使用全部样本最后加权投票 原理 1.使用全部样本通过决策树模型第一个弱分类器进行训练获取结果 思路预测正确-权重下降预测错误-权重上升 2.把第1个弱分类器的处理结果交给第2个弱分类器进行训练获取结果 思路预测正确-权重下降预测错误-权重上升 3.重复以上步骤直到所有弱分类器训练完成 4.最后根据所有弱分类器的投票结果得到最终分类结果 思路投票数最多的类别就是最终分类结果 #导包 import pandas as pd from sklearn.preprocessing import LabelEncoder#标签编码器 from sklearn.model_selection import train_test_split#训练集、测试集分割 from sklearn.tree import DecisionTreeClassifier#决策树分类器 from sklearn.ensemble import AdaBoostClassifier#AdaBoost分类器,集成Boosting思想 from sklearn.metrics import accuracy_score#模型评估-正确率 #1.获取数据集 df_winepd.read_csv(./datas/wine0501.csv) # df_wine.info() # print(df_wine[Class label].unique())#[1 2 3]有3种类别但是决策树只能识别二叉树 #2.数据预处理 #2.1从标签列(Class label)中过滤掉1类别剩下2,3,类别 df_winedf_wine[df_wine[Class label]!1] # print(df_wine[Class label].unique())#[2 3]只有2,3,类别 #2.2获取特征列和标签列 xdf_wine[[Alcohol,Hue]]#酒精和色泽 ydf_wine[Class label]#标签列 #2.3打印数据 # print(x[:5]) # print(y[:5]) #2.4通过标签编码器把标签列转换为数值列 leLabelEncoder() yle.fit_transform(y) # print(y)#[2,3]-[0,1] #2.5分割数据集 #参数1特征列参数2标签列参数3测试集占比参数4随机种子参数5按类别比例分割 x_train,x_test,y_train,y_testtrain_test_split(x,y,test_size0.2,random_state42,stratifyy) #3.特征工程 #4.模型训练 #场景1单一决策树-充当弱分类器 #4.1创建模型对象 estimator1DecisionTreeClassifier() #4.2训练模型 estimator1.fit(x_train,y_train) #4.3模型预测 y_pred1estimator1.predict(x_test) print(f单一决策树模型预测结果{y_pred1}) #4.4模型评估 print(f单一决策树模型正确率{accuracy_score(y_test,y_pred1)}) #场景2AdaBoost分类器-集成Boosting思想,CART树200棵 #4.1创建模型对象 estimator2AdaBoostClassifier(estimatorestimator1,n_estimators200,learning_rate0.5,algorithmSAMME) #4.2训练模型 estimator2.fit(x_train,y_train) #4.3模型预测 y_pred2estimator2.predict(x_test) print(fAdaBoost分类器模型预测结果{y_pred2}) #4.4模型评估 print(fAdaBoost分类器模型正确率{accuracy_score(y_test,y_pred2)})GBDT根据划分左右子树后左右的目标值均值作为预测值进行迭代计算依次类推这里算出来的负梯度会被当成下一轮的目标值进行迭代计算这个预测值是根据上次划分来计算目标值的均值作为预测值用平方损失找本棵树的切分点需要把负梯度传给下一个作为真实值把切分点传给下一个作为划分点求均值作为预测值相减得到残差(负梯度、下棵树的真实值)还需要找第二棵树的切分点用于下棵树求均值 案例 演示Boosting思想之GBDTGradient Boosting Decision Tree,梯度提升树处理泰坦尼克号数据集 GBDT梯度提升树解释 概述通过拟合付梯度来获取一个强学习器 流程 1.采用所有目标值的均值作为第一个弱学习器的预测值 2.目标值-预测值负梯度残差该列的值作为第2个弱学习器的目标值 3.针对第一个弱学习器一次计算每个分割点的最小平方和找到最佳分割点至此第一个弱学习器搭建完毕 4.把上述的分割点带入第2个弱学习器计算它的预测值以此分割点为界目标值的均值即为该部分数据的预测值 5.计算第2个弱学习器的付梯度最佳分割点至此第二个弱学习器搭建完毕 6.以此类推直至程序结束 #导包 import numpy as np import pandas as pd from sklearn.model_selection import train_test_split#分割数据集为训练集和测试集 from sklearn.tree import DecisionTreeClassifier#决策树分类器 from sklearn.ensemble import GradientBoostingClassifier#梯度提升树分类器 from sklearn.metrics import accuracy_score#准确率评估 from sklearn.model_selection import GridSearchCV # 网格搜索 #1.读取数据集 dfpd.read_csv(./datas/train.csv) # df.info() #2.数据预处理 #2.1提取特征和标签 xdf[[Pclass,Sex,Age]].copy() ydf[Survived].copy() #2.2处理Age列的缺失值用该列的均值填充 # x[Age].fillna(x[Age].mean(),inplaceTrue) x[Age]x[Age].fillna(x[Age].mean()) #2.3热编码处理字符串类型 xpd.get_dummies(x) #2.4分割数据集为训练集和测试集 x_train,x_test,y_train,y_testtrain_test_split(x,y,test_size0.2,random_state22) #3.特征工程 #4.模型训练预测评估 #场景1决策树分类器 #4.1创建模型对象 estimatorDecisionTreeClassifier() #4.2训练模型 estimator.fit(x_train,y_train) #4.3模型预测 y_predestimator.predict(x_test) print(f单个决策树对象的预测结果{y_pred}) #4.4模型评估 print(f决策树分类器的准确率为{accuracy_score(y_test,y_pred)})#0.7821229050279329 #场景2梯度提升树分类器 #4.1创建模型对象 estimator2GradientBoostingClassifier() #4.2训练模型 estimator2.fit(x_train,y_train) #4.3模型预测 y_predestimator2.predict(x_test) #4.4模型评估 print(f梯度提升树分类器的准确率为{accuracy_score(y_test,y_pred)})#0.7541899441340782 #场景3网格搜索和交叉验证 #4.1创建模型对象 estimator3GradientBoostingClassifier() #4.2定义参数网格 params{ n_estimators:[100,200,300],#弱学习器数量 learning_rate:[0.1,0.2,0.3],#学习率 max_depth: [3, 5] # 树最大深度 } #4.3创建网格搜索对象 grid_searchGridSearchCV(estimator3,params,cv5) #4.4训练模型 grid_search.fit(x_train,y_train) #4.5模型评估 print(f网格搜索的准确率为{grid_search.best_score_})#0.827331823106471 print(f网格搜索后的模型为{grid_search.best_estimator_})XGBoost最重要的是将这个打分函数记住并且上面的4句话可以陈述出具体是干什么的PS最后一个判断一个树是否要进行分类是依据打分函数的。XGBoost算法APIbst XGBClassifier(n_estimators, max_depth, learning_rate, objective)xgb案例红酒品质分类 案例通过XGBoost极限梯度提升树 完成 红酒品质分类案例 回顾XGBoost极限梯度提升树 概述 Extreme Gradient Boosting Tree,底层采用打分函数决定是否分支 原理 Gain值分枝前的打分-分枝后的左子树的打分分枝后的右子树的打分 如果Gain值大于0说明分枝有收益否则不分枝 #导包 import xgboost as xgb import joblib#保存和加载模型 import numpy as np import pandas as pd from collections import Counter#统计元素出现次数 from sklearn.model_selection import train_test_split, GridSearchCV # 划分数据集 from sklearn.metrics import accuracy_score,classification_report#评估模型,分类报告 from sklearn.model_selection import StratifiedKFold#分层K折交叉验证,类似网格搜索时cvk折数 from sklearn.utils import class_weight#计算类别权重 #1.定义函数对红酒品质分类元数据-拆分成训练集和测试集并存储到csv文件中 def dm01_data_split(): #1.加载数据集 dfpd.read_csv(./datas/红酒品质分类.csv) #2.查看数据集 # df.info() #3.抽取特征数据和标签数据 xdf.iloc[:,:-1] ydf.iloc[:,-1]-3 #最后一列是品质(标签)品质从3开始所以减3 #4.查看数据集 # print(x[:5]) # print(y[:5]) # print(f查看标签结果的分布情况{Counter(y)}) #5.切分数据集为训练集和测试集 #参1特征数据参2标签数据参3测试集占比参4随机种子参5参考数据集的标签分布 x_train,x_test,y_train,y_testtrain_test_split(x,y,test_size0.2,random_state42,stratifyy) #6.把上述的训练集特征和标签数据拼接到一起测试集特征和标签数据也拼接到一起最后写到csv文件中 pd.concat([x_train,y_train],axis1).to_csv(./datas/红酒品质分类_训练集.csv,indexFalse) pd.concat([x_test,y_test],axis1).to_csv(./datas/红酒品质分类_测试集.csv,indexFalse)#忽略索引 #2.定义函数训练模型并保存模型 def dm02_train_model(): # 1.读取训练集和测试集 train_datapd.read_csv(./datas/红酒品质分类_训练集.csv) test_datapd.read_csv(./datas/红酒品质分类_测试集.csv) #2.提取训练集和测试集的特征数据和标签数据 x_traintrain_data.iloc[:,:-1]#除了最后一列都是特征 y_traintrain_data.iloc[:,-1]#最后一列是标签 x_testtest_data.iloc[:,:-1] y_testtest_data.iloc[:,-1] #3.创建模型对象 estimatorxgb.XGBClassifier( n_estimators100, learning_rate0.1, max_depth3, random_state42, objectivemulti:softmax,#多分类问题,使用多分类模型 ) #加入平衡权重因为数据集是样本不均衡的 #参1平衡权重参2标签数据(即参考标签数据分布平衡权重) # class_weight.compute_class_weight(balanced,y_train) weights class_weight.compute_class_weight(balanced, classesnp.unique(y_train), yy_train) #4.模型训练 estimator.fit(x_train,y_train) #5.模型评估 print(f准确率{estimator.score(x_test,y_test)}) #6.保存模型 joblib.dump(estimator,./model/红酒品质分类_模型.pkl)#后缀名也可以为.pth都是pickle文件格式 print(模型保存成功) #3.定义函数测试模型 def dm03_use_model(): # 1.读取训练集和测试集 train_data pd.read_csv(./datas/红酒品质分类_训练集.csv) test_data pd.read_csv(./datas/红酒品质分类_测试集.csv) # 2.提取训练集和测试集的特征数据和标签数据 x_train train_data.iloc[:, :-1] # 除了最后一列都是特征 y_train train_data.iloc[:, -1] # 最后一列是标签 x_test test_data.iloc[:, :-1] y_test test_data.iloc[:, -1] #3.加载模型 estimatorjoblib.load(./model/红酒品质分类_模型.pkl) #4.创建网格搜索交叉验证结合分层采样数据找模型最优参数组合 #4.1定义变量记录参数组合 param_dict{max_depth:[3,4,5,6,7],learning_rate:[0.2,0.6,1.0,1.3],n_estimators:[70,100,200,250]} #4.2创建分层采样对象 #参1K折数参2随机种子参3是否打乱数据 kfStratifiedKFold(n_splits5,random_state42,shuffleTrue) #4.3创建网格搜索交叉验证对象 #参1模型对象参2参数组合参3分层采样对象 gs_estimatorGridSearchCV(estimator,param_gridparam_dict,cvkf) #5.模型训练 gs_estimator.fit(x_train,y_train) #6.模型预测 y_predgs_estimator.predict(x_test) print(f预测值为{y_pred}) #7.打印模型评估系数 print(f最优估计器对象组合{gs_estimator.best_params_}) print(f最优评分{gs_estimator.best_score_}) print(f准确率{accuracy_score(y_test,y_pred)}) #4.测试 if __name____main__: # dm01_data_split() # dm02_train_model() dm03_use_model()

读完文章,也想定制专属网站?

尧图设计师 24 小时内与您沟通定制方案

免费获取报价 →
↑