理解机器学习的各种算法

所有算法

回归和分类作为两种类型的算法,都出现回归(线性回归和逻辑回归)的算法,如何理解。

线性回归和逻辑回归的区别

线性回归:一个参数一个值(参数不断的变化,值不断的变化)。逻辑回归是基于先线性回归的基础上加一个sigmod函数的判断,得出最终的一个分类结果(0/1). 这就是线性回归和逻辑回归的区别。

线性回归例子:

复制代码
import numpy as np
from sklearn.linear_model import LinearRegression

# 准备数据
X = np.array([[50], [60], [70], [80], [90], [100]])  # 面积
y = np.array([150, 180, 210, 240, 270, 300])          # 房价

# 创建并训练模型
model = LinearRegression()
model.fit(X, y)

# 查看学到的参数
print("权重 w =", model.coef_[0])      # 3.0
print("偏置 b =", model.intercept_)    # 0.0

# 预测 75 平米
pred = model.predict([[75]])
print("75 平米预测房价 =", pred[0], "万元")  # 225.0

逻辑回归例子:

复制代码
import numpy as np
from sklearn.linear_model import LogisticRegression

# 准备数据
X = np.array([[1], [2], [3], [4], [5], [6], [7], [8]])  # 学习时长
y = np.array([0, 0, 0, 1, 1, 1, 1, 1])                    # 是否通过

# 创建并训练模型
model = LogisticRegression()
model.fit(X, y)

# 查看学到的参数
print("权重 w =", model.coef_[0][0])
print("偏置 b =", model.intercept_[0])

# 预测 3.5 小时
prob = model.predict_proba([[3.5]])[0][1]
pred = model.predict([[3.5]])[0]
print(f"3.5 小时通过概率 = {prob:.3f}")
print(f"预测类别 = {pred}")

如何理解参数中的 \*\*

prob和pred分别代表什么意思,为什么prob有两个值?

岭回归:

复制代码
import numpy as np
from sklearn.linear_model import Ridge, LinearRegression

# 准备数据
X = np.array([[50, 1],
              [60, 2],
              [70, 3],
              [80, 4],
              [90, 5],
              [100, 6]])  # 面积, 房龄
y = np.array([150, 180, 210, 240, 270, 300])  # 房价

# 普通线性回归
ols = LinearRegression()
ols.fit(X, y)
print("普通线性回归:")
print("  权重 =", ols.coef_)
print("  偏置 =", ols.intercept_)

# 岭回归
ridge = Ridge(alpha=1.0)  # alpha 就是 λ
ridge.fit(X, y)
print("\n岭回归 (alpha=1.0):")
print("  权重 =", ridge.coef_)
print("  偏置 =", ridge.intercept_)

X_new = np.array([[75, 3]])

print("普通线性回归预测:", ols.predict(X_new)[0])
print("岭回归预测:", ridge.predict(X_new)[0])

Lasso回归:

复制代码
import numpy as np
from sklearn.linear_model import LinearRegression, Ridge, Lasso

# 数据
X = np.array([[50, 2, 1, 3, 10],
              [60, 2, 2, 5, 9],
              [70, 3, 3, 8, 8],
              [80, 3, 4, 10, 7],
              [90, 4, 5, 12, 6],
              [100, 4, 6, 15, 5],
              [110, 5, 7, 18, 4],
              [120, 5, 8, 20, 3],
              [130, 6, 9, 22, 2],
              [140, 6, 10, 25, 1]])
y = np.array([150, 180, 210, 240, 270, 300, 330, 360, 390, 420])

# 普通线性回归
ols = LinearRegression()
ols.fit(X, y)
print("普通线性回归权重:", np.round(ols.coef_, 2))

# 岭回归
ridge = Ridge(alpha=1.0)
ridge.fit(X, y)
print("岭回归权重:      ", np.round(ridge.coef_, 2))

# Lasso 回归
lasso = Lasso(alpha=0.1)
lasso.fit(X, y)
print("Lasso 权重:      ", np.round(lasso.coef_, 2))

删除冗余特征的例子:

复制代码
import numpy as np
import pandas as pd
from sklearn.linear_model import Lasso, LassoCV, LinearRegression
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_squared_error

np.random.seed(42)

# ========== 1. 生成数据 ==========
n = 200
X = pd.DataFrame({
    "面积": np.random.rand(n) * 100 + 50,
    "房间数": np.random.randint(1, 6, n),
    "房龄": np.random.randint(0, 30, n),
    "楼层": np.random.randint(1, 30, n),
    "距离地铁": np.random.rand(n) * 10,
    "朝向": np.random.randint(0, 4, n),
    "装修年限": np.random.randint(0, 10, n),
    "绿化率": np.random.rand(n),
    "物业费": np.random.rand(n) * 5,
    "编号": np.arange(n),
})
y = (3.0 * X["面积"] + 5.0 * X["房间数"] - 1.0 * X["房龄"]
     + 0.5 * X["楼层"] - 2.0 * X["距离地铁"]
     + np.random.randn(n) * 5)

print("=" * 50)
print("原始特征数:", X.shape[1])

# ========== 2. 手动粗筛:10 -> 8 ==========
X_8 = X.drop(columns=["编号", "朝向"])
print("粗筛后特征数:", X_8.shape[1])

# ========== 3. Lasso 细筛 ==========
X_train, X_test, y_train, y_test = train_test_split(
    X_8, y, test_size=0.2, random_state=42
)
lasso_cv = LassoCV(alphas=np.logspace(-3, 2, 50), cv=5, random_state=42)
lasso_cv.fit(X_train, y_train)
lasso = Lasso(alpha=lasso_cv.alpha_).fit(X_train, y_train)

print("=" * 50)
print("Lasso 最优 alpha:", lasso_cv.alpha_)
print(pd.DataFrame({"特征": X_8.columns, "权重": lasso.coef_}))

# ========== 4. 手动删除权重为 0 的特征 ==========
selected_idx = np.where(lasso.coef_ != 0)[0]
selected_features = X_8.columns[selected_idx]
X_selected = X_8[selected_features]
print("=" * 50)
print("Lasso 保留特征:", list(selected_features))
print("最终特征数:", X_selected.shape[1])

# ========== 5. 重新训练最终模型 ==========
X_train_sel, X_test_sel, y_train_sel, y_test_sel = train_test_split(
    X_selected, y, test_size=0.2, random_state=42
)
final_model = LinearRegression().fit(X_train_sel, y_train_sel)

print("=" * 50)
print("最终模型权重:")
print(pd.DataFrame({"特征": selected_features, "权重": final_model.coef_}))
print("偏置:", final_model.intercept_)

# ========== 6. 评估 ==========
mse_final = mean_squared_error(y_test_sel, final_model.predict(X_test_sel))
print("=" * 50)
print(f"最终模型测试集 MSE: {mse_final:.4f}")
相关推荐
mtouch3331 小时前
技术解析|无人机倾斜摄影3D模型全天候动态光照融合方案
人工智能·无人机·虚拟现实·电子沙盘·数字沙盘·无人机倾斜摄影
小范的技术工坊1 小时前
大模型学习操作文档(13个核心概念串讲)
人工智能·学习·算法·大模型
DP DPharness1 小时前
按下撤回却没变化:dsh-recall-plugin 故障速查与排错清单
人工智能·dpharness
MetaEnchanter1 小时前
Word 提取图片:不用逐张截图,Excel、PPT 也能取
人工智能·ai·百万工具
腾渊信息科技公司1 小时前
腾渊科技重磅出品——工业软件智能化落地指南:从AI辅助到AI原生的架构演进
人工智能·科技·ai-native
2401_865261631 小时前
亦唐科技(YIKTONG):国产贴片机行业的技术先锋与市场领军者
人工智能·科技
一个低调的青年1 小时前
电池健康状态估计(一)
经验分享·笔记·其他·算法·模型
longlongzihan1 小时前
【LeetCode 204. 计数质数】从暴力枚举到埃拉托斯特尼筛法
算法·筛法
Dream-Y.ocean1 小时前
懂你的情绪,记得你在意的事 - 知暖如何用端到端具身交互智能重构 AI 情感陪伴
人工智能·重构·交互