{ "cells": [ { "cell_type": "markdown", "metadata": {}, "source": [ "# Multiclass classification\n", "\n", "This example fits a three-class model, prints its tree representation, and renders the same tree with Graphviz." ] }, { "cell_type": "code", "execution_count": 1, "id": "4859319e", "metadata": {}, "outputs": [], "source": [ "from sklearn.datasets import make_classification\n", "from sklearn.model_selection import train_test_split\n", "from sklearn.metrics import accuracy_score\n", "from pybrush import BrushClassifier\n", "import graphviz\n", "\n", "X, y = make_classification(\n", " n_samples=180, n_features=6, n_informative=5, n_redundant=0,\n", " n_classes=3, n_clusters_per_class=1, random_state=42,\n", ")\n", "X_train, X_test, y_train, y_test = train_test_split(\n", " X, y, test_size=0.3, stratify=y, random_state=42\n", ")" ] }, { "cell_type": "code", "execution_count": 2, "id": "b3d1aef8", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Completed 100% [====================]\n", "Test accuracy: 0.7962962962962963\n", "Probability rows sum to: [1. 1. 1.]\n" ] } ], "source": [ "model = BrushClassifier(\n", " pop_size=100, max_gens=100, max_size=40, max_depth=10,\n", " num_islands=1, random_state=42, verbosity=1,\n", ")\n", "model.fit(X_train, y_train)\n", "\n", "print('Test accuracy:', accuracy_score(y_test, model.predict(X_test)))\n", "print('Probability rows sum to:', model.predict_proba(X_test)[:3].sum(axis=1))" ] }, { "cell_type": "code", "execution_count": 3, "id": "56a9995f", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Softmax\n", "|- x_0\n", "|- Mean\n", "| |- 12.73*x_2\n", "| |- Min\n", "| | |- 55.80*x_5\n", "| | |- -7.67\n", "| | |- -2.40\n", "| | |- 1.00\n", "| |- -14.89*x_4\n", "| |- 20.88*x_3\n", "|- Sum\n", "| |- -1.35*x_3\n", "| |- -0.76*x_2\n", "| |- 0.51*x_0\n" ] } ], "source": [ "print(model.best_estimator_.program.get_model('tree'))" ] }, { "cell_type": "code", "execution_count": 4, "id": "a46f2e0d", "metadata": {}, "outputs": [ { "data": { "image/svg+xml": [ "\n", "\n", "\n", "\n", "\n", "\n", "G\n", "\n", "^ split feature fixed, * split threshold fixed\n", "\n", "\n", "1641b0670\n", "\n", "Softmax\n", "\n", "\n", "\n", "x_0\n", "\n", "x_0\n", "\n", "\n", "\n", "1641b0670->x_0\n", "\n", "\n", "class 0\n", "\n", "\n", "\n", "1641353b0\n", "\n", "Mean\n", "\n", "\n", "\n", "1641b0670->1641353b0\n", "\n", "\n", "class 1\n", "\n", "\n", "\n", "164121200\n", "\n", "Sum\n", "\n", "\n", "\n", "1641b0670->164121200\n", "\n", "\n", "class 2\n", "\n", "\n", "\n", "x_2\n", "\n", "x_2\n", "\n", "\n", "\n", "1641353b0->x_2\n", "\n", "\n", "12.73\n", "\n", "\n", "\n", "164454d70\n", "\n", "Min\n", "\n", "\n", "\n", "1641353b0->164454d70\n", "\n", "\n", "\n", "\n", "\n", "x_4\n", "\n", "x_4\n", "\n", "\n", "\n", "1641353b0->x_4\n", "\n", "\n", "-14.89\n", "\n", "\n", "\n", "x_3\n", "\n", "x_3\n", "\n", "\n", "\n", "1641353b0->x_3\n", "\n", "\n", "20.88\n", "\n", "\n", "\n", "164121200->x_0\n", "\n", "\n", "0.51\n", "\n", "\n", "\n", "164121200->x_2\n", "\n", "\n", "-0.76\n", "\n", "\n", "\n", "164121200->x_3\n", "\n", "\n", "-1.35\n", "\n", "\n", "\n", "x_5\n", "\n", "x_5\n", "\n", "\n", "\n", "164454d70->x_5\n", "\n", "\n", "55.80\n", "\n", "\n", "\n", "164154f90\n", "\n", "-7.67\n", "\n", "\n", "\n", "164454d70->164154f90\n", "\n", "\n", "\n", "\n", "\n", "1641c5680\n", "\n", "-2.40\n", "\n", "\n", "\n", "164454d70->1641c5680\n", "\n", "\n", "\n", "\n", "\n", "16411a550\n", "\n", "1.00\n", "\n", "\n", "\n", "164454d70->16411a550\n", "\n", "\n", "\n", "\n", "\n" ], "text/plain": [ "" ] }, "execution_count": 4, "metadata": {}, "output_type": "execute_result" } ], "source": [ "graphviz.Source(model.best_estimator_.program.get_model('dot'))" ] }, { "cell_type": "markdown", "id": "a5c111df", "metadata": {}, "source": [ "## Multiclass decision trees only\n", "\n", "Set `start_from_decision_trees=True` to restrict the initial model population to split-based class-logit branches. The Softmax root still combines one branch per class." ] }, { "cell_type": "code", "execution_count": 14, "id": "99bd6400", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Completed 15% [=== ]Split-only test accuracy: 0.7777777777777778\n", "Softmax\n", "|- 0.30*x_0\n", "|- If(x_4>=-0.47)\n", "| |- -1.53*x_4\n", "| |- 3.39\n", "|- If(x_4>=-0.47)\n", "| |- 0.14*Add\n", "| | |- -8.83*x_3\n", "| | |- -5.89*x_2\n", "| |- -0.43*x_2\n" ] } ], "source": [ "split_model = BrushClassifier(\n", " pop_size=100, max_gens=100, max_size=40, max_depth=5, max_stall=10,\n", " start_from_decision_trees=True,\n", " functions=['SplitOn', 'SplitBest', 'Add', 'Mul'],\n", " num_islands=1, random_state=7, verbosity=1,\n", ")\n", "split_model.fit(X_train, y_train)\n", "\n", "print('Split-only test accuracy:', accuracy_score(y_test, split_model.predict(X_test)))\n", "print(split_model.best_estimator_.program.get_model('tree'))" ] }, { "cell_type": "code", "execution_count": 15, "id": "3f700939", "metadata": {}, "outputs": [ { "data": { "image/svg+xml": [ "\n", "\n", "\n", "\n", "\n", "\n", "G\n", "\n", "^ split feature fixed, * split threshold fixed\n", "\n", "\n", "1638d9460\n", "\n", "Softmax\n", "\n", "\n", "\n", "x_0\n", "\n", "x_0\n", "\n", "\n", "\n", "1638d9460->x_0\n", "\n", "\n", "class 0\n", "\n", "\n", "\n", "1638cf8c0\n", "\n", "x_4 >= -0.47?\n", "\n", "\n", "\n", "1638d9460->1638cf8c0\n", "\n", "\n", "class 1\n", "\n", "\n", "\n", "1638ccc20\n", "\n", "x_4 >= -0.47?\n", "\n", "\n", "\n", "1638d9460->1638ccc20\n", "\n", "\n", "class 2\n", "\n", "\n", "\n", "x_4\n", "\n", "x_4\n", "\n", "\n", "\n", "1638cf8c0->x_4\n", "\n", "\n", "-1.53\n", "Y\n", "\n", "\n", "\n", "16384d160\n", "\n", "3.39\n", "\n", "\n", "\n", "1638cf8c0->16384d160\n", "\n", "\n", "N\n", "\n", "\n", "\n", "1638ca800\n", "\n", "Add\n", "\n", "\n", "\n", "1638ccc20->1638ca800\n", "\n", "\n", "0.14\n", "Y\n", "\n", "\n", "\n", "x_2\n", "\n", "x_2\n", "\n", "\n", "\n", "1638ccc20->x_2\n", "\n", "\n", "-0.43\n", "N\n", "\n", "\n", "\n", "1638ca800->x_2\n", "\n", "\n", "-5.89\n", "\n", "\n", "\n", "x_3\n", "\n", "x_3\n", "\n", "\n", "\n", "1638ca800->x_3\n", "\n", "\n", "-8.83\n", "\n", "\n", "\n" ], "text/plain": [ "" ] }, "execution_count": 15, "metadata": {}, "output_type": "execute_result" } ], "source": [ "graphviz.Source(split_model.best_estimator_.program.get_model('dot'))" ] } ], "metadata": { "kernelspec": { "display_name": "brush", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.13.14" } }, "nbformat": 4, "nbformat_minor": 5 }