catboost回归自动调参

import os

import time

import optuna

import pandas as pd

from catboost import CatBoostRegressor

from sklearn.metrics import r2_score, mean_squared_error

from sklearn.model_selection import train_test_split

X_train = data.drop('label', 'b1', 'b2', axis=1).values

y_train = data'label'.values

X_train, X_test, y_train, y_test = train_test_split(X_train, y_train, test_size=0.2, random_state=42)

def epoch_time(start_time, end_time):

elapsed_secs = end_time - start_time

elapsed_mins = elapsed_secs / 60

return elapsed_mins, elapsed_secs

def objective(trial):

自定义的参数空间

depth = trial.suggest_int('depth', 1, 16)

border_count = trial.suggest_int('border_count', 1, 222)

l2_leaf_reg = trial.suggest_int('l2_leaf_reg', 1, 222)

learning_rate = trial.suggest_uniform('learning_rate', 0.001, 0.9)

iterations = trial.suggest_int('iterations', 1, 100)

estimator = CatBoostRegressor(loss_function='RMSE', random_seed=22, learning_rate=learning_rate,

iterations=iterations, l2_leaf_reg=l2_leaf_reg,

border_count=border_count,

depth=depth, verbose=0)

estimator.fit(X_train, y_train)

val_pred = estimator.predict(X_test)

mse = mean_squared_error(y_test, val_pred)

return mse

""" Run optimize.

Set n_trials and/or timeout (in sec) for optimization by Optuna

"""

study = optuna.create_study(sampler=optuna.samplers.TPESampler(), direction='minimize')

study = optuna.create_study(sampler=optuna.samplers.RandomSampler(), direction='minimize')

start_time = time.time()

study.optimize(objective, n_trials=10)

end_time = time.time()

elapsed_mins, elapsed_secs = epoch_time(start_time, end_time)

print('elapsed_secs:', elapsed_secs)

print('Best value:', study.best_trial.value)

import os

import time

import pandas as pd

from catboost import CatBoostRegressor

from hyperopt import fmin, hp, partial, Trials, tpe,rand

from sklearn.metrics import r2_score, mean_squared_error

from sklearn.model_selection import train_test_split

自定义hyperopt的参数空间

space = {"iterations": hp.choice("iterations", range(1, 100)),

"depth": hp.randint("depth", 16),

"l2_leaf_reg": hp.randint("l2_leaf_reg", 222),

"border_count": hp.randint("border_count", 222),

'learning_rate': hp.uniform('learning_rate', 0.001, 0.9),

}

X_train = data.drop('label', 'b1', 'b2', axis=1).values

y_train = data'label'.values

X_train, X_test, y_train, y_test = train_test_split(X_train, y_train, test_size=0.2, random_state=42)

def epoch_time(start_time, end_time):

elapsed_secs = end_time - start_time

elapsed_mins = elapsed_secs / 60

return elapsed_mins, elapsed_secs

自动化调参并训练

def cat_factory(argsDict):

estimator = CatBoostRegressor(loss_function='RMSE', random_seed=22, learning_rate=argsDict'learning_rate',

iterations=argsDict'iterations', l2_leaf_reg=argsDict'l2_leaf_reg',

border_count=argsDict'border_count',

depth=argsDict'depth', verbose=0)

estimator.fit(X_train, y_train)

val_pred = estimator.predict(X_test)

mse = mean_squared_error(y_test, val_pred)

return mse

算法选择 tpe

algo = partial(tpe.suggest)

随机搜索

algo = partial(rand.suggest)

初始化每次尝试

trials = Trials()

开始自动参数寻优

start_time = time.time()

best = fmin(cat_factory, space, algo=algo, max_evals=10, trials=trials)

end_time = time.time()

elapsed_mins, elapsed_secs = epoch_time(start_time, end_time)

print('elapsed_secs:', elapsed_secs)

all = \[\]

遍历每一次的寻参结果

for one in trials:

str_re = str(one)

argsDict = one'misc''vals'

value = one'result''loss'

learning_rate = argsDict"learning_rate"0

iterations = argsDict"iterations"0

depth = argsDict"depth"0

l2_leaf_reg = argsDict"l2_leaf_reg"0

border_count = argsDict"border_count"0

finish = value, learning_rate, iterations, depth, l2_leaf_reg, border_count

all.append(finish)

parameters = pd.DataFrame(all, columns='value', 'learning_rate', 'iterations', 'depth', 'l2_leaf_reg', 'border_count')

从寻参结果中找到r2最大的

best = parameters.locabs(parameters\['value').idxmin()]

print("best: {}".format(best))

相关推荐
春日见几秒前
五分钟入门强化学习DDPG
大数据·人工智能·算法·机器学习·计算机视觉
jeffer_liu9 分钟前
Spring AI 生产级实战:记忆管理
java·人工智能·后端·spring·语言模型
土星云SaturnCloud13 分钟前
基于边缘计算的商场智慧运营架构设计与AI落地实践
服务器·人工智能·ai·边缘计算
vivo互联网技术13 分钟前
ICLR 2026 | LiveMoments 用参考图引导的扩散模型提升重选封面帧画质
人工智能·算法·aigc技术探索
Wonderful U14 分钟前
Python+Django实战|个人博客内容管理系统:搭建轻量化、高自由度的个人动态博客CMS系统
人工智能·python·django
懂AI的老郑14 分钟前
词元:AI理解语言的秘密钥匙
人工智能
落羽的落羽15 分钟前
【算法札记】练习 | Week5
linux·服务器·c++·人工智能·计算机网络·算法·哈希算法
RFID舜识物联网19 分钟前
耐高温RFID:让喷涂线从“数据断点”走向“全链贯通”
大数据·人工智能·嵌入式硬件·物联网·汽车
红宝村村长19 分钟前
loss.backward() 和 梯度累积
深度学习