Pandas Sampling and Random Data

Pandas provides rich random sampling functionality, allowing you to randomly select samples from a dataset as needed, and also supports generating random data.


Random Sampling

sample Method

Example

import pandas as pd
import numpy as np

# Create a large dataset
df = pd.DataFrame({
    "ID": range(1, 1001),
    "Value": np.random.randn(1000)
})

# Randomly sample 5 rows
print("Randomly sample 5 rows:")
print(df.sample(5))
print()

# Sample 30% of the data
print("Randomly sample 30%:")
print(df.sample(frac=0.3))
print()

# Sampling with replacement
print("Sample 5 rows with replacement:")
print(df.sample(5, replace=True))

Setting Random Seed

Example

import pandas as pd
import numpy as np

# Set the random seed to ensure reproducible results
np.random.seed(42)

df = pd.DataFrame({
    "ID": range(1, 11),
    "Value": np.random.randn(10)
})

print("Using random seed 42:")
print(df.sample(3))

# Run with the same seed again to get the same results
print("\nRunning with the same seed again:")
np.random.seed(42)
print(df.sample(3))

Random Data Generation

Example

import pandas as pd
import numpy as np

# Generate random integers
print("Random integers [0, 10):")
print(pd.Series(np.random.randint(0, 10, 5)))
print()

# Generate random floats
print("Random floats [0, 1):")
print(pd.Series(np.random.random(5)))
print()

# Generate normally distributed random numbers
print("Normal distribution N(0, 1):")
print(pd.Series(np.random.randn(5)))
print()

# Specify mean and standard deviation
print("Normal distribution N(10, 2):")
print(pd.Series(np.random.normal(10, 2, 5)))
print()

# Random choice
choices = ["A", "B", "C", "D"]
print("Random choice:")
print(pd.Series(np.random.choice(choices, 10)))

Train-Test Split

Example

import pandas as pd
import numpy as np

# Simulate a dataset
df = pd.DataFrame({
    "Feature1": np.random.randn(100),
    "Feature2": np.random.randn(100),
    "Target": np.random.choice([0, 1], 100)
})

# Split into training and test sets (80% / 20%)
train = df.sample(frac=0.8, random_state=42)
test = df.drop(train.index)

print(f"Training set: {len(train)} rows")
print(f"Test set: {len(test)} rows")
print()

# Stratified sampling (preserve target variable proportion)
print("Stratified sampling:")
from sklearn.model_selection import train_test_split
# Requires scikit-learn: pip install scikit-learn
# train, test = train_test_split(df, test_size=0.2, stratify=df["Target"])
Other Extensions