Pandas Categorical Type
Categorical is a data type in Pandas used for handling finite category values, particularly suitable for processing enumerated data such as gender, education, level, etc. The categorical type can significantly reduce memory usage and improve computational performance.
Creating Categorical Data
Creating from Series
Example
import pandas as pd
# Create categorical data
s = pd.Series(["Male", "Female", "Male", "Female", "Male"], dtype="category")
print("Basic categorical data:")
print(s)
print(f"Type: {s.dtype}")
print()
# Specify category order
s2 = pd.Series(
["Low", "Medium", "High", "Medium", "Low"],
dtype=pd.CategoricalDtype(categories=["Low", "Medium", "High"], ordered=True)
)
print("Ordered categories:")
print(s2)
print(f"Category order: {s2.dtype.categories.tolist()}")
# Create categorical data
s = pd.Series(["Male", "Female", "Male", "Female", "Male"], dtype="category")
print("Basic categorical data:")
print(s)
print(f"Type: {s.dtype}")
print()
# Specify category order
s2 = pd.Series(
["Low", "Medium", "High", "Medium", "Low"],
dtype=pd.CategoricalDtype(categories=["Low", "Medium", "High"], ordered=True)
)
print("Ordered categories:")
print(s2)
print(f"Category order: {s2.dtype.categories.tolist()}")
Creating from a List
Example
import pandas as pd
# Use pd.Categorical
categories = pd.Categorical(
["A", "B", "A", "C", "B"],
categories=["A", "B", "C"]
)
s = pd.Series(categories)
print("Creating from Categorical:")
print(s)
# Use pd.Categorical
categories = pd.Categorical(
["A", "B", "A", "C", "B"],
categories=["A", "B", "C"]
)
s = pd.Series(categories)
print("Creating from Categorical:")
print(s)
Category Attributes and Methods
Accessing Category Information
Example
import pandas as pd
s = pd.Series(
["Low", "Medium", "High", "Medium", "Low"],
dtype=pd.CategoricalDtype(categories=["Low", "Medium", "High"], ordered=True)
)
# Category attributes
print(f"Categories: {s.dtype.categories.tolist()}")
print(f"Ordered: {s.dtype.ordered}")
print()
# Count the number of each category
print("Category statistics:")
print(s.value_counts())
s = pd.Series(
["Low", "Medium", "High", "Medium", "Low"],
dtype=pd.CategoricalDtype(categories=["Low", "Medium", "High"], ordered=True)
)
# Category attributes
print(f"Categories: {s.dtype.categories.tolist()}")
print(f"Ordered: {s.dtype.ordered}")
print()
# Count the number of each category
print("Category statistics:")
print(s.value_counts())
Modifying Categories
Example
import pandas as pd
s = pd.Series(["A", "B", "C", "A", "B"], dtype="category")
print("Modify category names:")
s2 = s.cat.rename_categories({"A": "Excellent", "B": "Good", "C": "Pass"})
print(s2)
print()
# Add a new category
print("Add category:")
s3 = s.cat.add_categories(["D"])
print(s3.dtype.categories.tolist())
print()
# Remove category
print("Remove category:")
s4 = s.cat.remove_categories(["C"])
print(s4.dtype.categories.tolist())
s = pd.Series(["A", "B", "C", "A", "B"], dtype="category")
print("Modify category names:")
s2 = s.cat.rename_categories({"A": "Excellent", "B": "Good", "C": "Pass"})
print(s2)
print()
# Add a new category
print("Add category:")
s3 = s.cat.add_categories(["D"])
print(s3.dtype.categories.tolist())
print()
# Remove category
print("Remove category:")
s4 = s.cat.remove_categories(["C"])
print(s4.dtype.categories.tolist())
Ordered Categories
Ordered categories support comparison operations and are suitable for categorical data with a size order.
Example
import pandas as pd
# Create ordered categories
s = pd.Series(
["Low", "Medium", "High", "High", "Low", "Medium"],
dtype=pd.CategoricalDtype(categories=["Low", "Medium", "High"], ordered=True)
)
# Comparison operations
print("High > Medium:", s > "Medium")
print("Medium > Low:", s > "Low")
print()
# Sorting
print("After sorting:")
print(s.sort_values())
# Create ordered categories
s = pd.Series(
["Low", "Medium", "High", "High", "Low", "Medium"],
dtype=pd.CategoricalDtype(categories=["Low", "Medium", "High"], ordered=True)
)
# Comparison operations
print("High > Medium:", s > "Medium")
print("Medium > Low:", s > "Low")
print()
# Sorting
print("After sorting:")
print(s.sort_values())
Practice: Data Grouping
Example
import pandas as pd
# Simulate user level data
users = pd.DataFrame({
"User ID": range(1, 1001),
"Level": pd.Categorical(
["VIP"] * 100 + ["Gold Card"] * 300 + ["Silver Card"] * 400 + ["Regular"] * 200,
categories=["Regular", "Silver Card", "Gold Card", "VIP"],
ordered=True
),
"Consumption Amount": [1000, 500, 100, 50] * 250
})
print("User level distribution:")
print(users["Level"].value_counts().sort_index())
print()
# Group statistics by level
print("Statistics by level:")
print(users.groupby("Level", observed=True)["Consumption Amount"].sum())
# Simulate user level data
users = pd.DataFrame({
"User ID": range(1, 1001),
"Level": pd.Categorical(
["VIP"] * 100 + ["Gold Card"] * 300 + ["Silver Card"] * 400 + ["Regular"] * 200,
categories=["Regular", "Silver Card", "Gold Card", "VIP"],
ordered=True
),
"Consumption Amount": [1000, 500, 100, 50] * 250
})
print("User level distribution:")
print(users["Level"].value_counts().sort_index())
print()
# Group statistics by level
print("Statistics by level:")
print(users.groupby("Level", observed=True)["Consumption Amount"].sum())
Memory Optimization
The categorical type can greatly reduce the memory usage of string data.
Example
import pandas as pd
import numpy as np
# Create a large amount of repeated string data
np.random.seed(42)
s_str = pd.Series(np.random.choice(["Beijing", "Shanghai", "Guangzhou", "Shenzhen"], 1000000))
# Convert to categorical type
s_cat = s_str.astype("category")
# Memory comparison
print(f"String type memory: {s_str.memory_usage(deep=True) / 1024 / 1024:.2f} MB")
print(f"Categorical type memory: {s_cat.memory_usage(deep=True) / 1024 / 1024:.2f} MB")
print(f"Saved: {(1 - s_cat.memory_usage(deep=True) / s_str.memory_usage(deep=True)) * 100:.1f}%")
import numpy as np
# Create a large amount of repeated string data
np.random.seed(42)
s_str = pd.Series(np.random.choice(["Beijing", "Shanghai", "Guangzhou", "Shenzhen"], 1000000))
# Convert to categorical type
s_cat = s_str.astype("category")
# Memory comparison
print(f"String type memory: {s_str.memory_usage(deep=True) / 1024 / 1024:.2f} MB")
print(f"Categorical type memory: {s_cat.memory_usage(deep=True) / 1024 / 1024:.2f} MB")
print(f"Saved: {(1 - s_cat.memory_usage(deep=True) / s_str.memory_usage(deep=True)) * 100:.1f}%")
Other ExtensionsWhen there are many repeated string values in the data, using the categorical type can significantly reduce memory usage, making it especially suitable for large datasets.