{ "cells": [ { "cell_type": "code", "execution_count": 1, "metadata": {}, "outputs": [], "source": [ "import pandas as pd\n", "import numpy as np\n", "\n", "import matplotlib.pyplot as plt\n", "\n", "from sklearn.model_selection import train_test_split\n", "from sklearn.preprocessing import OrdinalEncoder\n", "\n", "from feature_engine.wrappers import SklearnTransformerWrapper\n", "from feature_engine.encoding import RareLabelEncoder" ] }, { "cell_type": "code", "execution_count": 2, "metadata": {}, "outputs": [ { "data": { "text/html": [ "
\n", " | Id | \n", "MSSubClass | \n", "MSZoning | \n", "LotFrontage | \n", "LotArea | \n", "Street | \n", "Alley | \n", "LotShape | \n", "LandContour | \n", "Utilities | \n", "... | \n", "PoolArea | \n", "PoolQC | \n", "Fence | \n", "MiscFeature | \n", "MiscVal | \n", "MoSold | \n", "YrSold | \n", "SaleType | \n", "SaleCondition | \n", "SalePrice | \n", "
---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
0 | \n", "1 | \n", "60 | \n", "RL | \n", "65.0 | \n", "8450 | \n", "Pave | \n", "NaN | \n", "Reg | \n", "Lvl | \n", "AllPub | \n", "... | \n", "0 | \n", "NaN | \n", "NaN | \n", "NaN | \n", "0 | \n", "2 | \n", "2008 | \n", "WD | \n", "Normal | \n", "208500 | \n", "
1 | \n", "2 | \n", "20 | \n", "RL | \n", "80.0 | \n", "9600 | \n", "Pave | \n", "NaN | \n", "Reg | \n", "Lvl | \n", "AllPub | \n", "... | \n", "0 | \n", "NaN | \n", "NaN | \n", "NaN | \n", "0 | \n", "5 | \n", "2007 | \n", "WD | \n", "Normal | \n", "181500 | \n", "
2 | \n", "3 | \n", "60 | \n", "RL | \n", "68.0 | \n", "11250 | \n", "Pave | \n", "NaN | \n", "IR1 | \n", "Lvl | \n", "AllPub | \n", "... | \n", "0 | \n", "NaN | \n", "NaN | \n", "NaN | \n", "0 | \n", "9 | \n", "2008 | \n", "WD | \n", "Normal | \n", "223500 | \n", "
3 | \n", "4 | \n", "70 | \n", "RL | \n", "60.0 | \n", "9550 | \n", "Pave | \n", "NaN | \n", "IR1 | \n", "Lvl | \n", "AllPub | \n", "... | \n", "0 | \n", "NaN | \n", "NaN | \n", "NaN | \n", "0 | \n", "2 | \n", "2006 | \n", "WD | \n", "Abnorml | \n", "140000 | \n", "
4 | \n", "5 | \n", "60 | \n", "RL | \n", "84.0 | \n", "14260 | \n", "Pave | \n", "NaN | \n", "IR1 | \n", "Lvl | \n", "AllPub | \n", "... | \n", "0 | \n", "NaN | \n", "NaN | \n", "NaN | \n", "0 | \n", "12 | \n", "2008 | \n", "WD | \n", "Normal | \n", "250000 | \n", "
5 rows × 81 columns
\n", "\n", " | Alley | \n", "MasVnrType | \n", "BsmtQual | \n", "BsmtCond | \n", "BsmtExposure | \n", "BsmtFinType1 | \n", "BsmtFinType2 | \n", "Electrical | \n", "FireplaceQu | \n", "GarageType | \n", "GarageFinish | \n", "GarageQual | \n", "
---|---|---|---|---|---|---|---|---|---|---|---|---|
529 | \n", "0.0 | \n", "2.0 | \n", "3.0 | \n", "1.0 | \n", "3.0 | \n", "4.0 | \n", "1.0 | \n", "2.0 | \n", "3.0 | \n", "0.0 | \n", "2.0 | \n", "2.0 | \n", "
491 | \n", "0.0 | \n", "1.0 | \n", "3.0 | \n", "1.0 | \n", "3.0 | \n", "1.0 | \n", "0.0 | \n", "0.0 | \n", "3.0 | \n", "0.0 | \n", "3.0 | \n", "2.0 | \n", "
459 | \n", "0.0 | \n", "2.0 | \n", "3.0 | \n", "1.0 | \n", "3.0 | \n", "3.0 | \n", "1.0 | \n", "2.0 | \n", "3.0 | \n", "2.0 | \n", "3.0 | \n", "2.0 | \n", "
279 | \n", "0.0 | \n", "0.0 | \n", "1.0 | \n", "1.0 | \n", "3.0 | \n", "1.0 | \n", "1.0 | \n", "2.0 | \n", "3.0 | \n", "0.0 | \n", "0.0 | \n", "2.0 | \n", "
655 | \n", "0.0 | \n", "0.0 | \n", "3.0 | \n", "1.0 | \n", "3.0 | \n", "5.0 | \n", "1.0 | \n", "2.0 | \n", "1.0 | \n", "2.0 | \n", "3.0 | \n", "2.0 | \n", "