Velar/AI/train/training.ipynb

228 lines
4.4 KiB
Text
Raw Normal View History

2025-08-21 20:55:35 -07:00
{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"import pandas as pd\n",
"\n",
"df = pd.read_csv(\"transcations.csv\")\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "a074a45e",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"df = df[['Description', 'Category']].head()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "fe26b20f",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"df[['Description', 'Category']].head()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "44c3995d",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"df['Description'] = df['Description'].str.lower().str.replace('[^a-z\\s]', '', regex=True)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "82245456",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"df['Category'] = df['Category'].str.lower().str.replace('[^a-z\\s]', '', regex=True)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "620ef271",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"from sklearn.model_selection import train_test_split\n",
"\n",
"X = df['Description']\n",
"y = df['Category']\n",
"\n",
"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "3eebf352",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"from sklearn.feature_extraction.text import TfidfVectorizer\n",
"\n",
"vectorizer = TfidfVectorizer()\n",
"X_train_vec = vectorizer.fit_transform(X_train)\n",
"X_test_vec = vectorizer.transform(X_test)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "75a141a8",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"from sklearn.naive_bayes import MultinomialNB\n",
"\n",
"model = MultinomialNB()\n",
"model.fit(X_train_vec, y_train)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "184354db",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"def predict_category(text):\n",
" text = [text.lower()]\n",
" text_vec = vectorizer.transform(text)\n",
" return model.predict(text_vec)[0]\n",
"\n",
"# Example\n",
"print(predict_category(\"Netflix\"))\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c30b9f27",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"from sklearn.metrics import classification_report\n",
"y_pred = model.predict(X_test_vec)\n",
"print(classification_report(y_test, y_pred))\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "109a12d5",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"import pickle\n",
"\n",
"# Save model\n",
"with open('model.pkl', 'wb') as f:\n",
" pickle.dump(model, f)\n",
"\n",
"# Save vectorizer\n",
"with open('vectorizer.pkl', 'wb') as f:\n",
" pickle.dump(vectorizer, f)\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "bcb76b6a",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"import joblib\n",
"\n",
"joblib.dump(model, 'model.pkl')\n",
"joblib.dump(vectorizer, 'vectorizer.pkl')\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f16d6da5",
"metadata": {
"vscode": {
"languageId": "plaintext"
}
},
"outputs": [],
"source": [
"from google.colab import files\n",
"files.download('model.pkl')\n",
"files.download('vectorizer.pkl')\n"
]
}
],
"metadata": {
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 5
}