| FazBrowse GitHub Viewer | Trending | | Home |
| Tools: [Download Repo ZIP] [Original HTTPS Page] |
2 files changed
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -346,7 +346,8 @@ RUN pip install --upgrade cython && \ | |||
| 346 | 346 | pip install fasttext && \ | |
| 347 | 347 | apt-get install -y libhunspell-dev && pip install hunspell && \ | |
| 348 | 348 | pip install annoy && \ | |
| 349 | - pip install category_encoders && \ | ||
| 349 | + # Need to use CountEncoder from category_encoders before it's officially released | ||
| 350 | + pip install git+https://github.com/scikit-learn-contrib/categorical-encoding.git && \ | ||
| 350 | 351 | # Newer version crashes (latest = 1.14.0) when running tensorflow. | |
| 351 | 352 | # python -c "from google.cloud import bigquery; import tensorflow". This flow is common because bigquery is imported in kaggle_gcp.py | |
| 352 | 353 | # which is loaded at startup. | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -0,0 +1,16 @@ | |||
| 1 | + import unittest | ||
| 2 | + | ||
| 3 | + ## Need to make sure we have CountEncoder available from the category_encoders library | ||
| 4 | + class TestCategoryEncoders(unittest.TestCase): | ||
| 5 | + def test_count_encoder(self): | ||
| 6 | + | ||
| 7 | + from category_encoders import CountEncoder | ||
| 8 | + import pandas as pd | ||
| 9 | + | ||
| 10 | + encoder = CountEncoder(cols="data") | ||
| 11 | + | ||
| 12 | + data = pd.DataFrame([1, 2, 3, 1, 4, 5, 3, 1], columns=["data"]) | ||
| 13 | + | ||
| 14 | + encoded = encoder.fit_transform(data) | ||
| 15 | + self.assertTrue((encoded.data == [3, 1, 2, 3, 1, 1, 2, 3]).all()) | ||
| 16 | + | ||
| Back | FazBrowse Home | New Git URL |
0 commit comments