Python GUI 数据科学工具:数据预处理、模型训练和预测

本工具使用 Python Tkinter 库构建了一个用户友好的 GUI,方便用户进行数据预处理、模型训练和预测,无需编写复杂的代码。

功能:

  1. 数据集导入: 设置 CSV 等格式文件导入按钮,按后可导入相关数据集(四个字段)。
  2. 数据预处理: 设置缺失值处理、数值型数据标准化、类别型数据编码等按钮。
  3. 训练集测试集划分: 设置按钮将数据集划分为训练集和测试集。
  4. 模型选择: 设置下拉菜单,可选择相应模型(例如逻辑回归、决策树等)。
  5. 模型评估: 设置下拉菜单,可选择相应评价指标(例如准确率、精确率、召回率、F1 分数等)。选择后,可进行相应评估。
  6. 训练模型: 按后可训练选择的模型并输出训练结果(如准确率、损失值等)。
  7. 预测: 按后可输入待预测数据并输出预测结果。

示例代码:

import tkinter as tk
from tkinter import filedialog
import pandas as pd
from sklearn.preprocessing import StandardScaler, LabelEncoder
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LogisticRegression
from sklearn.tree import DecisionTreeClassifier
from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score

class DataPreprocessing:
    def __init__(self):
        self.df = None
        self.x = None
        self.y = None
        self.x_train = None
        self.x_test = None
        self.y_train = None
        self.y_test = None

    def load_data(self):
        file_path = filedialog.askopenfilename(filetypes=[('CSV Files', '*.csv')])
        if file_path:
            self.df = pd.read_csv(file_path)

    def handle_missing_values(self):
        self.df.fillna(self.df.mean(), inplace=True)

    def encode_categorical_data(self):
        le = LabelEncoder()
        self.df['category'] = le.fit_transform(self.df['category'])

    def scale_numeric_data(self):
        scaler = StandardScaler()
        self.df[['numeric1', 'numeric2']] = scaler.fit_transform(self.df[['numeric1', 'numeric2']])

    def split_train_test(self):
        self.x = self.df[['numeric1', 'numeric2', 'category']]
        self.y = self.df['label']
        self.x_train, self.x_test, self.y_train, self.y_test = train_test_split(self.x, self.y, test_size=0.2)

class ModelTraining:
    def __init__(self):
        self.model = None
        self.metrics = None

    def train_model(self, model_name):
        if model_name == 'Logistic Regression':
            self.model = LogisticRegression()
        elif model_name == 'Decision Tree':
            self.model = DecisionTreeClassifier()
        else:
            return False

        self.model.fit(data_preprocessing.x_train, data_preprocessing.y_train)
        return True

    def evaluate_model(self, metric):
        if metric == 'Accuracy':
            self.metrics = accuracy_score(data_preprocessing.y_test, self.model.predict(data_preprocessing.x_test))
        elif metric == 'Precision':
            self.metrics = precision_score(data_preprocessing.y_test, self.model.predict(data_preprocessing.x_test))
        elif metric == 'Recall':
            self.metrics = recall_score(data_preprocessing.y_test, self.model.predict(data_preprocessing.x_test))
        elif metric == 'F1 Score':
            self.metrics = f1_score(data_preprocessing.y_test, self.model.predict(data_preprocessing.x_test))

class Application:
    def __init__(self, master):
        self.master = master
        self.master.title('Data Science Toolkit')
        self.master.geometry('400x300')

        self.data_preprocessing = DataPreprocessing()
        self.model_training = ModelTraining()

        # Load Data Button
        self.load_data_button = tk.Button(self.master, text='Load Data', command=self.load_data)
        self.load_data_button.pack(pady=10)

        # Preprocessing Options
        self.preprocessing_options_frame = tk.Frame(self.master)
        self.preprocessing_options_frame.pack(pady=10)

        self.handle_missing_values_button = tk.Button(self.preprocessing_options_frame, text='Handle Missing Values',
                                                      command=self.handle_missing_values)
        self.handle_missing_values_button.pack(side=tk.LEFT, padx=5)

        self.encode_categorical_data_button = tk.Button(self.preprocessing_options_frame, text='Encode Categorical Data',
                                                         command=self.encode_categorical_data)
        self.encode_categorical_data_button.pack(side=tk.LEFT, padx=5)

        self.scale_numeric_data_button = tk.Button(self.preprocessing_options_frame, text='Scale Numeric Data',
                                                   command=self.scale_numeric_data)
        self.scale_numeric_data_button.pack(side=tk.LEFT, padx=5)

        # Train Test Split Button
        self.train_test_split_button = tk.Button(self.master, text='Train Test Split', command=self.train_test_split)
        self.train_test_split_button.pack(pady=10)

        # Model Selection Dropdown
        self.model_options = ['Logistic Regression', 'Decision Tree']
        self.model_selection_var = tk.StringVar(self.master)
        self.model_selection_var.set(self.model_options[0])
        self.model_selection_dropdown = tk.OptionMenu(self.master, self.model_selection_var, *self.model_options)
        self.model_selection_dropdown.pack(pady=10)

        # Model Evaluation Dropdown
        self.metric_options = ['Accuracy', 'Precision', 'Recall', 'F1 Score']
        self.metric_selection_var = tk.StringVar(self.master)
        self.metric_selection_var.set(self.metric_options[0])
        self.metric_selection_dropdown = tk.OptionMenu(self.master, self.metric_selection_var, *self.metric_options)
        self.metric_selection_dropdown.pack(pady=10)

        # Train Model Button
        self.train_model_button = tk.Button(self.master, text='Train Model', command=self.train_model)
        self.train_model_button.pack(pady=10)

        # Prediction Entry and Button
        self.prediction_entry = tk.Entry(self.master)
        self.prediction_entry.pack(pady=10)

        self.prediction_button = tk.Button(self.master, text='Predict', command=self.predict)
        self.prediction_button.pack(pady=10)

    def load_data(self):
        self.data_preprocessing.load_data()

    def handle_missing_values(self):
        self.data_preprocessing.handle_missing_values()

    def encode_categorical_data(self):
        self.data_preprocessing.encode_categorical_data()

    def scale_numeric_data(self):
        self.data_preprocessing.scale_numeric_data()

    def train_test_split(self):
        self.data_preprocessing.split_train_test()

    def train_model(self):
        if self.model_training.train_model(self.model_selection_var.get()):
            print('Training Result:', self.model_training.metrics)
        else:
            print('Invalid Model Selection')

    def predict(self):
        prediction = self.model_training.model.predict([[float(x) for x in self.prediction_entry.get().split()]])
        print('Prediction Result:', prediction)

root = tk.Tk()
app = Application(root)
root.mainloop()

使用说明:

  1. 确保已安装必要的库:pandas, scikit-learn, tkinter.
  2. 运行代码,将会弹出一个窗口。
  3. 点击 'Load Data' 按钮选择您的 CSV 数据集文件。
  4. 根据需要选择数据预处理选项。
  5. 点击 'Train Test Split' 按钮将数据划分为训练集和测试集。
  6. 从模型选择下拉菜单中选择您要使用的模型。
  7. 从模型评估下拉菜单中选择您要使用的评价指标。
  8. 点击 'Train Model' 按钮训练模型并查看结果。
  9. 在预测输入框中输入待预测数据,并点击 'Predict' 按钮查看预测结果。

注意:

  • 您的 CSV 数据集文件应包含四个字段:numeric1, numeric2, category, label。
  • 代码中的示例使用了简单的缺失值处理、数据标准化和类别型数据编码方法,您可以根据您的具体需求修改。
  • 您可以添加更多模型和评价指标到下拉菜单中。
  • 您也可以扩展 GUI 以显示更多信息,例如训练过程中的日志、模型评估结果等。

希望本工具能够帮助您轻松地进行数据科学任务。

Python GUI 数据科学工具:数据预处理、模型训练和预测

原文地址: https://www.cveoy.top/t/topic/ocsV 著作权归作者所有。请勿转载和采集!

免费AI点我,无需注册和登录