import numpy as np
import pandas as pd
import matplotlib.pyplot as plt

path =  'linear_regression.txt'
data = pd.read_csv(path, header=None, names=['Population', 'Profit'])
data.head()

data['Population'].describe()

count    97.000000
mean      8.159800
std       3.869884
min       5.026900
25%       5.707700
50%       6.589400
75%       8.578100
max      22.203000
Name: Population, dtype: float64

data.plot(kind='scatter', x='Population', y='Profit', figsize=(12,8))
#plt.figure(figsize=(12,8))
#plt.scatter(data['Population'],data['Profit'])
#lt.xlabel('Population')
#plt.ylabel('Porfit')
plt.show()

def computeCost(X, y, theta):
    inner = np.power(((X * theta.T) - y), 2)
    return np.sum(inner) / (2 * len(X))

data.insert(0, 'Ones', 1)

data.head()

# set X (training data) and y (target variable)
cols = data.shape[1]
X = data.iloc[:,0:cols-1]#X是所有行，去掉最后一列
y = data.iloc[:,cols-1:cols]#X是所有行，最后一列

X.head()#head()是观察前5行

y.head()

X = np.matrix(X.values)
y = np.matrix(y.values)
theta = np.matrix(np.array([0,0]))

theta

matrix([[0, 0]])

X.shape, theta.shape, y.shape

((97, 2), (1, 2), (97, 1))

computeCost(X, y, theta)

32.072733877455676

theta_1_list=np.linspace(-0.5,2.5,50)
cost_list=[]
for i in range(len(theta_1_list)):
    cost_list.append(computeCost(X,y,np.matrix(np.array([0,theta_1_list[i]]))))
plt.figure(figsize=(12,8))
plt.plot(theta_1_list,cost_list)
plt.xlabel('theta_1')
plt.ylabel('Cost(theta_1)')
plt.axvline(x=0,ls="--",c="red")
plt.show()

def gradientDescent(X, y, theta, alpha, iters):
    temp = np.matrix(np.zeros(theta.shape))
    parameters = int(theta.ravel().shape[1])
    cost = np.zeros(iters)
    
    for i in range(iters):
        error = (X * theta.T) - y
        
        for j in range(parameters):
            term = np.multiply(error, X[:,j])
            temp[0,j] = theta[0,j] - ((alpha / len(X)) * np.sum(term))
            
        theta = temp
        cost[i] = computeCost(X, y, theta)
        
    return theta, cost

alpha = 0.01
iters = 1500

g, cost = gradientDescent(X, y, theta, alpha, iters)
g

matrix([[-3.63029144,  1.16636235]])

computeCost(X, y, g)

4.483388256587726

x = np.linspace(data.Population.min(), data.Population.max(), 100)
f = g[0, 0] + (g[0, 1] * x)

fig, ax = plt.subplots(figsize=(12,8))
ax.plot(x, f, 'r', label='Prediction')
ax.scatter(data.Population, data.Profit, label='Traning Data')
ax.legend(loc=2)
ax.set_xlabel('Population')
ax.set_ylabel('Profit')
ax.set_title('Predicted Profit vs. Population Size')
plt.show()

fig, ax = plt.subplots(figsize=(12,8))
ax.plot(np.arange(iters), cost, 'r')
ax.set_xlabel('Iterations')
ax.set_ylabel('Cost')
ax.set_title('Error vs. Training Epoch')
plt.show()

# set X (training data) and y (target variable)
path =  'linear_regression.txt'
data = pd.read_csv(path, header=None, names=['Population', 'Profit'])

cols = data.shape[1]
X = data.iloc[:,:cols-1]#X是所有行，去掉最后一列
y = data.iloc[:,cols-1:cols]#X是所有行，最后一列

from sklearn import linear_model
model = linear_model.LinearRegression()
model.fit(X, y)
# 输出回归系数
print("回归系数：", model.coef_)
print("截距：", model.intercept_)

回归系数： [[1.19303364]]
截距： [-3.89578088]

print(X.shape)
print(y.shape)

(97, 2)
(97, 1)

from sklearn.metrics import mean_squared_error, r2_score
y_pred = model.predict(X)
# 评估模型性能
mse = mean_squared_error(y, y_pred)
r2 = r2_score(y, y_pred)
print("均方误差(MSE)：", mse)
print("R平方值：", r2)

# 绘制训练集和测试集的回归线
plt.figure(figsize=(10, 6))
plt.scatter(X, y, color='blue', label='Traning Data')
plt.plot(X, model.predict(X), color='red', linewidth=2, label='Prediction')
plt.xlabel('Population')
plt.ylabel('Profit')
plt.title('Predicted Profit vs. Population Size')
plt.legend()
plt.grid(True)
plt.show()

均方误差(MSE)： 8.953942751950358
R平方值： 0.7020315537841397

	Population	Profit
0	6.1101	17.5920
1	5.5277	9.1302
2	8.5186	13.6620
3	7.0032	11.8540
4	5.8598	6.8233

	Ones	Population	Profit
0	1	6.1101	17.5920
1	1	5.5277	9.1302
2	1	8.5186	13.6620
3	1	7.0032	11.8540
4	1	5.8598	6.8233

	Ones	Population
0	1	6.1101
1	1	5.5277
2	1	8.5186
3	1	7.0032
4	1	5.8598

	Profit
0	17.5920
1	9.1302
2	13.6620
3	11.8540
4	6.8233

单变量线性回归¶