Combining Pandas with NumPy

Pandas is built on NumPy, and the two are tightly integrated. Understanding their interaction can make data processing and scientific computing more efficient.


Interconversion

Converting between DataFrame/Series and NumPy

Example

import pandas as pd
import numpy as np

# Convert DataFrame to NumPy array
df = pd.DataFrame({
    "A": [1, 2, 3],
    "B": [4, 5, 6]
})

arr = df.to_numpy()
print("DataFrame to array:")
print(arr)
print(f"Type: {type(arr)}")
print()

# Convert Series to array
s = pd.Series([1, 2, 3])
arr = s.values  # Or s.to_numpy()
print("Series to array:")
print(arr)
print()

# Convert NumPy to DataFrame
arr = np.array([[1, 2], [3, 4], [5, 6]])
df = pd.DataFrame(arr, columns=["A", "B"])
print("Array to DataFrame:")
print(df)

Using NumPy Functions in Pandas

Example

import pandas as pd
import numpy as np

s = pd.Series([1, 2, 3, 4, 5])

# Use NumPy functions
print("Absolute value:", np.abs(s))
print("Square root:", np.sqrt(s))
print("Exponential:", np.exp(s))
print("Logarithm:", np.log(s))
print()

# Conditional filtering
print("Values greater than 3:")
print(s[s > 3])

Vectorized Operations

Example

import pandas as pd
import numpy as np

df = pd.DataFrame({"A": [1, 2, 3], "B": [4, 5, 6]})

# Vectorized operations
print("A + B:", (df["A"] + df["B"]).tolist())
print("A * B:", (df["A"] * df["B"]).tolist())
print("A > B:", (df["A"] > df["B"]).tolist())
print()

# Use apply for element-wise computation
result = df.apply(lambda x: np.multiply(x, 2))
print("Each element * 2:")
print(result)

Numerical Computation Techniques

Example

import pandas as pd
import numpy as np

df = pd.DataFrame({
    "A": [1, 2, 3, 4, 5],
    "B": [10, 20, 30, 40, 50]
})

# Calculate the percentage change of A relative to B
df["Rate of change"] = np.divide(df["A"], df["B"], where=df["B"] != 0) * 100
print("Percentage change:")
print(df)
print()

# Use np.where for conditional assignment
df["Label"] = np.where(df["A"] > 3, "Large", "Small")
print("Conditional label:")
print(df)

Statistical Computation

Example

import pandas as pd
import numpy as np

s = pd.Series([1, 2, 3, 4, 5, 6, 7, 8, 9, 10])

# NumPy statistical functions
print(f"Mean: {np.mean(s):.2f}")
print(f"Standard deviation: {np.std(s):.2f}")
print(f"Maximum: {np.max(s)}")
print(f"Minimum: {np.min(s)}")
print()

# Median
print(f"Median: {np.median(s)}")

The deep integration of Pandas and NumPy makes complex data processing simple and efficient.

Other Extensions