Gradient Boosting starts with a rough prediction, then repeatedly adds small trees that correct what the current model still gets wrong. Many small, targeted corrections accumulate into a flexible predictor.
# training datasetX=np.linspace(-10,10,500)[:,np.newaxis]noise=np.random.rand(X.shape[0])*10# targety=((np.sin(X).ravel()+np.cos(4*X).ravel())*10+10+np.linspace(-10,10,500)+noise)# train gradient boostingreg=GradientBoostingRegressor(n_estimators=50,learning_rate=0.5,)reg.fit(X,y)y_pred=reg.predict(X)# Check the fitting to the training dataplt.figure(figsize=(10,5))plt.scatter(X,y,c="k",marker="x",label="training data")plt.plot(X,y_pred,c="r",label="Predictions for the final created model",linewidth=1)plt.xlabel("x")plt.ylabel("y")plt.title("Degree of fitting to training dataset")plt.legend()plt.show()
Check how fitting to the training data changes when loss is changed to [“squared_error”, “absolute_error”, “huber”, “quantile”]." It is expected that “absolute_error” and “huber” will not go on to predict outliers since the penalty for outliers is not as large as the squared error.
# training dataX=np.linspace(-10,10,500)[:,np.newaxis]# prepare outliersnoise=np.random.rand(X.shape[0])*10fori,niinenumerate(noise):ifi%80==0:noise[i]=70+np.random.randint(-10,10)# targety=((np.sin(X).ravel()+np.cos(4*X).ravel())*10+10+np.linspace(-10,10,500)+noise)forlossin["squared_error","absolute_error","huber","quantile"]:# train gradient boostingreg=GradientBoostingRegressor(n_estimators=50,learning_rate=0.5,loss=loss,)reg.fit(X,y)y_pred=reg.predict(X)# Check the fitting to the training data.plt.figure(figsize=(10,5))plt.scatter(X,y,c="k",marker="x",label="training dataset")plt.plot(X,y_pred,c="r",label="Predictions for the final created model",linewidth=1)plt.xlabel("x")plt.ylabel("y")plt.title(f"Degree of fitting to training data, loss={loss}",fontsize=18)plt.legend()plt.show()
fromsklearn.metricsimportmean_squared_errorasMSE# trainig datasetX=np.linspace(-10,10,500)[:,np.newaxis]noise=np.random.rand(X.shape[0])*10# targety=((np.sin(X).ravel()+np.cos(4*X).ravel())*10+10+np.linspace(-10,10,500)+noise)# Try to create a model with different n_estimatorsn_estimators_list=[(i+1)*5foriinrange(20)]mses=[]forn_estimatorsinn_estimators_list:reg=GradientBoostingRegressor(n_estimators=n_estimators,learning_rate=0.3,)reg.fit(X,y)y_pred=reg.predict(X)mses.append(MSE(y,y_pred))# Plotting mean_squared_error for different n_estimatorsplt.figure(figsize=(10,5))plt.plot(n_estimators_list,mses,"x")plt.xlabel("n_estimators")plt.ylabel("Mean Squared Error(training data)")plt.grid()plt.show()
# Try to create a model with different n_estimatorslearning_rate_list=[np.round(0.1*(i+1),1)foriinrange(20)]mses=[]forlearning_rateinlearning_rate_list:reg=GradientBoostingRegressor(n_estimators=30,learning_rate=learning_rate,)reg.fit(X,y)y_pred=reg.predict(X)mses.append(np.log(MSE(y,y_pred)))# Plotting mean_squared_error for different n_estimatorsplt.figure(figsize=(10,5))plt_index=[iforiinrange(len(learning_rate_list))]plt.plot(plt_index,mses,"x")plt.xticks(plt_index,learning_rate_list,rotation=90)plt.xlabel("learning_rate",fontsize=15)plt.ylabel("log(Mean Squared Error) (training data)",fontsize=15)plt.grid()plt.show()