一、理论知识及其原理

关于原理,可以参照https://blog.csdn.net/u011734144/article/details/79717470和笔记即可
笔记截图:
image.png
image.png

二、线性逻辑回归

1.梯度下降法的应用

  1. import matplotlib.pyplot as plt
  2. import numpy as np
  3. from sklearn.metrics import classification_report
  4. from sklearn import preprocessing
  5. # 数据是否需要标准化
  6. scale = False
  7. # 载入数据
  8. data = np.genfromtxt(r"C:\Users\小新\Desktop\课程pdf\py\机器学习\逻辑回归\LR-testSet.csv", delimiter=",")
  9. x_data = data[:,:-1] #切分除了最后一列的数据都是x
  10. y_data = data[:,-1] #切分最后一列为y,这里的y_data是分类类别(取值0和1)
  11. def plot(): #定义这个函数是用来画出下面的图的
  12. x0 = [] #分类类别对应的值
  13. x1 = [] #分类类别对应的值
  14. y0 = [] #分类类别为0
  15. y1 = [] #分类类别为1
  16. # 切分不同类别的数据
  17. for i in range(len(x_data)): #数据切分
  18. if y_data[i] == 0: #说明这是第0个类别
  19. x0.append(x_data[i,0]) #存放第0个特征
  20. y0.append(x_data[i,1]) #存放第1个特征
  21. else:
  22. x1.append(x_data[i,0])
  23. y1.append(x_data[i,1])
  24. # 画图
  25. scatter0 = plt.scatter(x0, y0, c='b', marker='o') #0类别的散点图
  26. scatter1 = plt.scatter(x1, y1, c='r', marker='x') #1类别的散点图
  27. #画图例
  28. plt.legend(handles=[scatter0,scatter1],labels=['label0','label1'],loc='best') #loc是选择最佳位置
  29. plot()
  30. plt.show()
  31. #注:这里可以用以下代码快速画图:
  32. x_data = data[:,:-1]
  33. y_data = data[:,-1]
  34. plt.scatter(x.data[:,0],x_data[:,1],color = y_data) #直接利用y_data来进行颜色赋值(y_data本来是二分类变量)
  35. plt.show()
  36. # 数据处理,添加偏置项
  37. x_data = data[:,:-1]
  38. y_data = data[:,-1,np.newaxis] #增加维度
  39. print(np.mat(x_data).shape) #100行2列
  40. print(np.mat(y_data).shape) #100行1列(相当于列向量)
  41. # 给样本添加偏置项
  42. X_data = np.concatenate((np.ones((100,1)),x_data),axis=1)
  43. print(X_data.shape) #100行3列
  44. #开始进行逻辑回归
  45. def sigmoid(x): #定义sigmoid函数
  46. return 1.0/(1+np.exp(-x))
  47. def cost(xMat, yMat, ws): #定义损失函数,xmat是训练数据,ymat是标签,ws是权值(都是矩阵形式)
  48. left = np.multiply(yMat, np.log(sigmoid(xMat*ws))) #损失函数左边的部分,np.multiply是对应位置元素相乘,yMat和np.log(sigmoid(xMat*ws)都是100*1的
  49. right = np.multiply(1 - yMat, np.log(1 - sigmoid(xMat*ws))) #损失函数的右边部分
  50. return np.sum(left + right) / -(len(xMat)) #得到最终的损失函数
  51. #注:这里损失函数用矩阵乘法会不会成功?可以试试看
  52. def gradAscent(xArr, yArr):
  53. if scale == True:
  54. xArr = preprocessing.scale(xArr) #对数据进行标准化
  55. xMat = np.mat(xArr)
  56. yMat = np.mat(yArr)
  57. lr = 0.001 #定义学习率
  58. epochs = 10000 #定义迭代次数
  59. costList = [] #保存cost值,每次迭代都会产生一个新的cost
  60. # 计算数据行列数
  61. # 行代表数据个数,列代表权值个数
  62. m,n = np.shape(xMat) #m代表行,n代表列
  63. # 初始化权值
  64. ws = np.mat(np.ones((n,1))) #都初始化为1
  65. for i in range(epochs+1):
  66. # xMat和weights矩阵相乘
  67. h = sigmoid(xMat*ws) #损失函数中hθ(x),100行1列
  68. # 计算误差
  69. ws_grad = xMat.T*(h - yMat)/m #计算梯度,矩阵乘法,3行100列乘以100行1列,得到3行1列,即系数值,公式中的求和项相当于在矩阵里进行了
  70. ws = ws - lr*ws_grad
  71. if i % 50 == 0:
  72. costList.append(cost(xMat,yMat,ws)) #每隔50次保存一次cost值
  73. return ws,costList
  74. #注:
  75. #1,2
  76. #3,4
  77. #4,3
  78. #2,1
  79. #得到矩阵为
  80. ##4,6
  81. ##6,4
  82. # 训练模型,得到权值和cost值的变化
  83. ws,costList = gradAscent(X_data, y_data)
  84. print(ws)
  85. if scale == False: #非标准化的数据,标准化后的数据数值改变后画图会很奇怪,于是标准化后的图可以不用画
  86. # 画图决策边界
  87. plot()
  88. x_test = [[-4],[3]] #规定x的取值
  89. y_test = (-ws[0] - x_test*ws[1])/ws[2] #这里令w0+xw1+yw2 = 0(sigmoid分界点→0.5)求解y(边界)
  90. plt.plot(x_test, y_test, 'k')
  91. plt.show()
  92. # 画图 loss值的变化
  93. x = np.linspace(0,10000,201) #从0到10000画200个点
  94. plt.plot(x, costList, c='r')
  95. plt.title('Train')
  96. plt.xlabel('Epochs')
  97. plt.ylabel('Cost')
  98. plt.show()
  99. # 预测
  100. def predict(x_data, ws):
  101. if scale == True:
  102. x_data = preprocessing.scale(x_data)
  103. xMat = np.mat(x_data)
  104. ws = np.mat(ws)
  105. return [1 if x >= 0.5 else 0 for x in sigmoid(xMat*ws)] #如果x大于等于0.5,返回类别1,小于0.5,返回类别0
  106. predictions = predict(X_data, ws)
  107. print(classification_report(y_data, predictions)) #正确率,召回率的查看情况

2.sklearn库的方法

  1. import matplotlib.pyplot as plt
  2. import numpy as np
  3. from sklearn.metrics import classification_report
  4. from sklearn import preprocessing
  5. from sklearn import linear_model
  6. # 数据是否需要标准化
  7. scale = False
  8. # 载入数据
  9. data = np.genfromtxt(r"C:\Users\小新\Desktop\课程pdf\py\机器学习\逻辑回归\LR-testSet.csv", delimiter=",")
  10. x_data = data[:,:-1]
  11. y_data = data[:,-1]
  12. def plot():
  13. x0 = []
  14. x1 = []
  15. y0 = []
  16. y1 = []
  17. # 切分不同类别的数据
  18. for i in range(len(x_data)):
  19. if y_data[i]==0:
  20. x0.append(x_data[i,0])
  21. y0.append(x_data[i,1])
  22. else:
  23. x1.append(x_data[i,0])
  24. y1.append(x_data[i,1])
  25. # 画图
  26. scatter0 = plt.scatter(x0, y0, c='b', marker='o')
  27. scatter1 = plt.scatter(x1, y1, c='r', marker='x')
  28. #画图例
  29. plt.legend(handles=[scatter0,scatter1],labels=['label0','label1'],loc='best')
  30. plot()
  31. plt.show()
  32. logistic = linear_model.LogisticRegression() #利用sklearn库进行逻辑回归
  33. logistic.fit(x_data, y_data)
  34. #其余相关的参数如截距、回归系数的代码跟回归分析类似,在这里不过多赘述
  35. if scale == False:
  36. # 画图决策边界,注意这里是二维的,多维的情况不宜画图
  37. plot()
  38. x_test = np.array([[-4],[3]])
  39. y_test = (-logistic.intercept_ - x_test*logistic.coef_[0][0])/logistic.coef_[0][1] #logistic.coef_是一个1*2的矩阵形式,这里的[0]是取第一行的两个数,转化成一个向量形式,第二个[0],[1]就是取这两个数了
  40. plt.plot(x_test, y_test, 'k')
  41. plt.show()
  42. predictions = logistic.predict(x_data)
  43. print(classification_report(y_data, predictions)) #计算正确率、召回率和F1值

三、非线性逻辑回归

1.梯度下降法的应用

  1. import matplotlib.pyplot as plt
  2. import numpy as np
  3. from sklearn.metrics import classification_report
  4. from sklearn import preprocessing
  5. from sklearn.preprocessing import PolynomialFeatures
  6. # 数据是否需要标准化
  7. scale = False
  8. # 载入数据
  9. data = np.genfromtxt(r"C:\Users\小新\Desktop\课程pdf\py\机器学习\逻辑回归\LR-testSet2.txt", delimiter=",")
  10. x_data = data[:,:-1]
  11. y_data = data[:,-1,np.newaxis]
  12. def plot():
  13. x0 = []
  14. x1 = []
  15. y0 = []
  16. y1 = []
  17. # 切分不同类别的数据
  18. for i in range(len(x_data)):
  19. if y_data[i]==0:
  20. x0.append(x_data[i,0])
  21. y0.append(x_data[i,1])
  22. else:
  23. x1.append(x_data[i,0])
  24. y1.append(x_data[i,1])
  25. # 画图
  26. scatter0 = plt.scatter(x0, y0, c='b', marker='o')
  27. scatter1 = plt.scatter(x1, y1, c='r', marker='x')
  28. #画图例
  29. plt.legend(handles=[scatter0,scatter1],labels=['label0','label1'],loc='best')
  30. plot()
  31. plt.show()
  32. # 定义多项式回归,degree的值可以调节多项式的特征,自己可以多次尝试
  33. poly_reg = PolynomialFeatures(degree=3) #做数据多项式处理
  34. # 特征处理
  35. x_poly = poly_reg.fit_transform(x_data)
  36. ##注:特征处理我们来举个例子:
  37. ##test = [[2,3]]
  38. # 定义多项式回归,degree的值可以调节多项式的特征
  39. #poly_reg = PolynomialFeatures(degree=3)
  40. # 特征处理
  41. #x_poly = poly_reg.fit_transform(test) #构造出非线性项,包括交互项
  42. #x_poly #这里给出的第一个是x的偏置项,第二第三个是x1,x2,第四个是x1^2,第五个是x1*x2,第六个是x2^2,第七个是x2^3,第八个是x1^2*x2,第九个是x1*x2^2,第十个是x2^3
  43. #output情况:array([[ 1., 2., 3., 4., 6., 9., 8., 12., 18., 27.]])
  44. def sigmoid(x):
  45. return 1.0/(1+np.exp(-x))
  46. def cost(xMat, yMat, ws):
  47. left = np.multiply(yMat, np.log(sigmoid(xMat*ws)))
  48. right = np.multiply(1 - yMat, np.log(1 - sigmoid(xMat*ws)))
  49. return np.sum(left + right) / -(len(xMat))
  50. def gradAscent(xArr, yArr):
  51. if scale == True:
  52. xArr = preprocessing.scale(xArr)
  53. xMat = np.mat(xArr)
  54. yMat = np.mat(yArr)
  55. lr = 0.03
  56. epochs = 50000
  57. costList = []
  58. # 计算数据列数,有几列就有几个权值
  59. m,n = np.shape(xMat)
  60. # 初始化权值
  61. ws = np.mat(np.ones((n,1)))
  62. for i in range(epochs+1):
  63. # xMat和weights矩阵相乘
  64. h = sigmoid(xMat*ws)
  65. # 计算误差
  66. ws_grad = xMat.T*(h - yMat)/m
  67. ws = ws - lr*ws_grad
  68. if i % 50 == 0:
  69. costList.append(cost(xMat,yMat,ws))
  70. return ws,costList
  71. # 训练模型,得到权值和cost值的变化
  72. ws,costList = gradAscent(x_poly, y_data)
  73. print(ws)
  74. #画决策边界情况(手操较难)
  75. # 获取数据值所在的范围【决定数据下图数据框的大小】
  76. x_min, x_max = x_data[:, 0].min() - 1, x_data[:, 0].max() + 1
  77. y_min, y_max = x_data[:, 1].min() - 1, x_data[:, 1].max() + 1
  78. # 生成网格矩阵
  79. xx, yy = np.meshgrid(np.arange(x_min, x_max, 0.02),
  80. np.arange(y_min, y_max, 0.02))
  81. # np.r_按row来组合array,
  82. # np.c_按colunm来组合array
  83. # >>> a = np.array([1,2,3])
  84. # >>> b = np.array([5,2,5])
  85. # >>> np.r_[a,b]
  86. # array([1, 2, 3, 5, 2, 5])
  87. # >>> np.c_[a,b]
  88. # array([[1, 5],
  89. # [2, 2],
  90. # [3, 5]])
  91. # >>> np.c_[a,[0,0,0],b]
  92. # array([[1, 0, 5],
  93. # [2, 0, 2],
  94. # [3, 0, 5]])
  95. z = sigmoid(poly_reg.fit_transform(np.c_[xx.ravel(), yy.ravel()]).dot(np.array(ws)))# ravel与flatten类似,多维数据转一维(使之扁平化)。flatten不会改变原始数据,ravel会改变原始数据
  96. #z函数相当于把网格矩阵的点给做一个预测,是1就是一类,是0就是一类,利用密集点填充的方式来强行构造出一个边界
  97. for i in range(len(z)):
  98. if z[i] > 0.5:
  99. z[i] = 1
  100. else:
  101. z[i] = 0
  102. z = z.reshape(xx.shape)
  103. # 等高线图
  104. cs = plt.contourf(xx, yy, z)
  105. plot()
  106. plt.show()
  107. # 预测
  108. def predict(x_data, ws):
  109. # if scale == True:
  110. # x_data = preprocessing.scale(x_data)
  111. xMat = np.mat(x_data)
  112. ws = np.mat(ws)
  113. return [1 if x >= 0.5 else 0 for x in sigmoid(xMat*ws)]
  114. predictions = predict(x_poly, ws)
  115. print(classification_report(y_data, predictions))

2.sklearn库的方法

  1. import numpy as np
  2. import matplotlib.pyplot as plt
  3. from sklearn import linear_model
  4. from sklearn.datasets import make_gaussian_quantiles
  5. from sklearn.preprocessing import PolynomialFeatures
  6. # 生成2维正态分布,生成的数据按分位数分为两类,500个样本,2个样本特征
  7. # 可以生成两类或多类数据
  8. x_data, y_data = make_gaussian_quantiles(n_samples=500, n_features=2,n_classes=2)
  9. plt.scatter(x_data[:, 0], x_data[:, 1], c=y_data)
  10. plt.show()
  11. logistic = linear_model.LogisticRegression()
  12. logistic.fit(x_data, y_data)
  13. # 定义多项式回归,degree的值可以调节多项式的特征
  14. poly_reg = PolynomialFeatures(degree=5)
  15. # 特征处理
  16. x_poly = poly_reg.fit_transform(x_data)
  17. # 定义逻辑回归模型
  18. logistic = linear_model.LogisticRegression()
  19. # 训练模型
  20. logistic.fit(x_poly, y_data) #不指定x值会出现线性分类情况,所以需要自己去定义多项式部分!注意!
  21. # 获取数据值所在的范围
  22. x_min, x_max = x_data[:, 0].min() - 1, x_data[:, 0].max() + 1
  23. y_min, y_max = x_data[:, 1].min() - 1, x_data[:, 1].max() + 1
  24. # 生成网格矩阵
  25. xx, yy = np.meshgrid(np.arange(x_min, x_max, 0.02),
  26. np.arange(y_min, y_max, 0.02))
  27. # np.r_按row来组合array,
  28. # np.c_按colunm来组合array
  29. # >>> a = np.array([1,2,3])
  30. # >>> b = np.array([5,2,5])
  31. # >>> np.r_[a,b]
  32. # array([1, 2, 3, 5, 2, 5])
  33. # >>> np.c_[a,b]
  34. # array([[1, 5],
  35. # [2, 2],
  36. # [3, 5]])
  37. # >>> np.c_[a,[0,0,0],b]
  38. # array([[1, 0, 5],
  39. # [2, 0, 2],
  40. # [3, 0, 5]])
  41. z = logistic.predict(poly_reg.fit_transform(np.c_[xx.ravel(), yy.ravel()]))# ravel与flatten类似,多维数据转一维。flatten不会改变原始数据,ravel会改变原始数据
  42. z = z.reshape(xx.shape)
  43. # 等高线图
  44. cs = plt.contourf(xx, yy, z)
  45. # 样本散点图
  46. plt.scatter(x_data[:, 0], x_data[:, 1], c=y_data)
  47. plt.show()
  48. print('score:',logistic.score(x_poly,y_data)) #输出R方值