apply() lets you run any Python function row-by-row or column-by-column.
Use it when built-in pandas methods aren't enough — custom logic, multi-column
calculations, or complex transformations.
import pandas as pd
df = pd.DataFrame({
'name': ['Alice Smith', 'Bob Jones', 'Carol White'],
'major': ['CS', 'Math', 'CS'],
'gpa': [3.9, 2.8, 3.5],
'credits': [120, 95, 110],
'us_citizen': ['Y', 'N', 'Y'],
'pell': ['N', 'Y', 'Y']
})# Apply Python built-ins
df['name'].apply(len) # length of each name
df['name'].apply(str.upper) # uppercase each name
df['gpa'].apply(round) # round each GPA# Define a function, then apply it
def gpa_label(gpa):
if gpa >= 3.7:
return 'Honors'
elif gpa >= 3.0:
return 'Good Standing'
elif gpa >= 2.0:
return 'Satisfactory'
else:
return 'At Risk'
df['standing'] = df['gpa'].apply(gpa_label)Result:
| name | gpa | standing |
|---|---|---|
| Alice Smith | 3.9 | Honors |
| Bob Jones | 2.8 | Satisfactory |
| Carol White | 3.5 | Good Standing |
# Same as above but inline
df['standing'] = df['gpa'].apply(lambda x: 'Honors' if x >= 3.7 else 'Other')
# Scale GPA to 10-point scale
df['gpa_10'] = df['gpa'].apply(lambda x: round(x * 2.5, 1))
# Extract first name from full name
df['first_name'] = df['name'].apply(lambda x: x.split()[0])Use this when your function needs multiple columns from the same row.
# Function receives the entire row as a Series
def eligibility(row):
if row['gpa'] >= 3.0 and row['credits'] >= 100:
return 'Eligible'
else:
return 'Not Eligible'
df['scholarship'] = df.apply(eligibility, axis=1)Result:
| name | gpa | credits | scholarship |
|---|---|---|---|
| Alice Smith | 3.9 | 120 | Eligible |
| Bob Jones | 2.8 | 95 | Not Eligible |
| Carol White | 3.5 | 110 | Eligible |
# Combine two columns into a label
df['summary'] = df.apply(
lambda row: f"{row['name']} ({row['major']}) — GPA: {row['gpa']}",
axis=1
)
# Multi-column condition
df['flag'] = df.apply(
lambda row: 'Y' if row['gpa'] > 3.5 and row['us_citizen'] == 'Y' else 'N',
axis=1
)# Get the max value of each numeric column
df[['gpa', 'credits']].apply(max)
# Get data type of each column
df.apply(lambda col: col.dtype)
# Count non-null values per column
df.apply(lambda col: col.notna().sum())# Split a column into multiple new columns
def parse_name(full_name):
parts = full_name.split()
return pd.Series({
'first': parts[0],
'last': parts[1] if len(parts) > 1 else ''
})
df[['first', 'last']] = df['name'].apply(parse_name)Result:
| name | first | last |
|---|---|---|
| Alice Smith | Alice | Smith |
| Bob Jones | Bob | Jones |
| Carol White | Carol | White |
def scale_gpa(gpa, scale):
return round(gpa * scale, 2)
# Pass extra argument via args=
df['gpa_10'] = df['gpa'].apply(scale_gpa, args=(2.5,))
# Or via a lambda
df['gpa_10'] = df['gpa'].apply(lambda x: scale_gpa(x, scale=2.5))# Apply a custom function to each group
def top_student(group):
return group.nlargest(1, 'gpa')
df.groupby('major').apply(top_student).reset_index(drop=True)# ❌ Slow — avoid apply for simple math
df['gpa_10'] = df['gpa'].apply(lambda x: x * 2.5)
# ✅ Fast — use vectorized operations instead
df['gpa_10'] = df['gpa'] * 2.5
# ❌ Slow — avoid apply for simple string ops
df['upper_name'] = df['name'].apply(str.upper)
# ✅ Fast — use .str accessor instead
df['upper_name'] = df['name'].str.upper()
# ❌ Slow — avoid apply for simple conditions
df['flag'] = df['gpa'].apply(lambda x: 1 if x > 3.0 else 0)
# ✅ Fast — use np.where or boolean casting
import numpy as np
df['flag'] = np.where(df['gpa'] > 3.0, 1, 0)Rule of thumb: If pandas or numpy has a built-in for it, use that. Reserve
.apply()for logic that genuinely can't be vectorized.
| Goal | Method |
|---|---|
| Transform one column | df['col'].apply(func) |
| Inline simple logic | df['col'].apply(lambda x: ...) |
| Use multiple columns | df.apply(func, axis=1) |
| Split into multiple columns | return pd.Series({...}) inside apply |
| Pass extra arguments | apply(func, args=(val,)) |
| Apply per group | groupby().apply(func) |
| Simple math/strings | Use vectorized ops — skip apply |
axis=0 → function runs DOWN each column (default)
axis=1 → function runs ACROSS each row (use when you need multiple columns)
Think of axis=1 as: "for each row, do something with its values"