Combining Pandas with NumPy
Pandas is built on NumPy, and the two are tightly integrated. Understanding their interaction can make data processing and scientific computing more efficient.
Interconversion
Converting between DataFrame/Series and NumPy
Example
import pandas as pd
import numpy as np
# Convert DataFrame to NumPy array
df = pd.DataFrame({
"A": [1, 2, 3],
"B": [4, 5, 6]
})
arr = df.to_numpy()
print("DataFrame to array:")
print(arr)
print(f"Type: {type(arr)}")
print()
# Convert Series to array
s = pd.Series([1, 2, 3])
arr = s.values # Or s.to_numpy()
print("Series to array:")
print(arr)
print()
# Convert NumPy to DataFrame
arr = np.array([[1, 2], [3, 4], [5, 6]])
df = pd.DataFrame(arr, columns=["A", "B"])
print("Array to DataFrame:")
print(df)
import numpy as np
# Convert DataFrame to NumPy array
df = pd.DataFrame({
"A": [1, 2, 3],
"B": [4, 5, 6]
})
arr = df.to_numpy()
print("DataFrame to array:")
print(arr)
print(f"Type: {type(arr)}")
print()
# Convert Series to array
s = pd.Series([1, 2, 3])
arr = s.values # Or s.to_numpy()
print("Series to array:")
print(arr)
print()
# Convert NumPy to DataFrame
arr = np.array([[1, 2], [3, 4], [5, 6]])
df = pd.DataFrame(arr, columns=["A", "B"])
print("Array to DataFrame:")
print(df)
Using NumPy Functions in Pandas
Example
import pandas as pd
import numpy as np
s = pd.Series([1, 2, 3, 4, 5])
# Use NumPy functions
print("Absolute value:", np.abs(s))
print("Square root:", np.sqrt(s))
print("Exponential:", np.exp(s))
print("Logarithm:", np.log(s))
print()
# Conditional filtering
print("Values greater than 3:")
print(s[s > 3])
import numpy as np
s = pd.Series([1, 2, 3, 4, 5])
# Use NumPy functions
print("Absolute value:", np.abs(s))
print("Square root:", np.sqrt(s))
print("Exponential:", np.exp(s))
print("Logarithm:", np.log(s))
print()
# Conditional filtering
print("Values greater than 3:")
print(s[s > 3])
Vectorized Operations
Example
import pandas as pd
import numpy as np
df = pd.DataFrame({"A": [1, 2, 3], "B": [4, 5, 6]})
# Vectorized operations
print("A + B:", (df["A"] + df["B"]).tolist())
print("A * B:", (df["A"] * df["B"]).tolist())
print("A > B:", (df["A"] > df["B"]).tolist())
print()
# Use apply for element-wise computation
result = df.apply(lambda x: np.multiply(x, 2))
print("Each element * 2:")
print(result)
import numpy as np
df = pd.DataFrame({"A": [1, 2, 3], "B": [4, 5, 6]})
# Vectorized operations
print("A + B:", (df["A"] + df["B"]).tolist())
print("A * B:", (df["A"] * df["B"]).tolist())
print("A > B:", (df["A"] > df["B"]).tolist())
print()
# Use apply for element-wise computation
result = df.apply(lambda x: np.multiply(x, 2))
print("Each element * 2:")
print(result)
Numerical Computation Techniques
Example
import pandas as pd
import numpy as np
df = pd.DataFrame({
"A": [1, 2, 3, 4, 5],
"B": [10, 20, 30, 40, 50]
})
# Calculate the percentage change of A relative to B
df["Rate of change"] = np.divide(df["A"], df["B"], where=df["B"] != 0) * 100
print("Percentage change:")
print(df)
print()
# Use np.where for conditional assignment
df["Label"] = np.where(df["A"] > 3, "Large", "Small")
print("Conditional label:")
print(df)
import numpy as np
df = pd.DataFrame({
"A": [1, 2, 3, 4, 5],
"B": [10, 20, 30, 40, 50]
})
# Calculate the percentage change of A relative to B
df["Rate of change"] = np.divide(df["A"], df["B"], where=df["B"] != 0) * 100
print("Percentage change:")
print(df)
print()
# Use np.where for conditional assignment
df["Label"] = np.where(df["A"] > 3, "Large", "Small")
print("Conditional label:")
print(df)
Statistical Computation
Example
import pandas as pd
import numpy as np
s = pd.Series([1, 2, 3, 4, 5, 6, 7, 8, 9, 10])
# NumPy statistical functions
print(f"Mean: {np.mean(s):.2f}")
print(f"Standard deviation: {np.std(s):.2f}")
print(f"Maximum: {np.max(s)}")
print(f"Minimum: {np.min(s)}")
print()
# Median
print(f"Median: {np.median(s)}")
import numpy as np
s = pd.Series([1, 2, 3, 4, 5, 6, 7, 8, 9, 10])
# NumPy statistical functions
print(f"Mean: {np.mean(s):.2f}")
print(f"Standard deviation: {np.std(s):.2f}")
print(f"Maximum: {np.max(s)}")
print(f"Minimum: {np.min(s)}")
print()
# Median
print(f"Median: {np.median(s)}")
Other ExtensionsThe deep integration of Pandas and NumPy makes complex data processing simple and efficient.