LangChain Model Call Interception -- @wrap_model_call

@wrap_model_call is the most powerful hook in middleware.

Unlike before/after, @wrap_model_call doesn't just observe; it canfully control the model execution process— retry, degrade, cache, or even skip the model and directly use a preset reply.


Understanding the handler callback

The core of @wrap_model_call is ahandler callback functionOnly calling handler(request) actually executes the model; if not called, the model is skipped.

Example

# Basic structure of wrap_model_call
# request: contains all information such as model, messages, tools, etc.
# handler: a callable object; only executing it actually calls the model

@wrap_model_call
def my_middleware(request, handler):
    # You can do anything before the model call
    print("Model is about to be called...")

    # Only calling handler(request) truly executes the model
    response = handler(request)

    # You can do anything after the model call
    print("Model call completed")

    return response

Scenario 1: Retry Mechanism

This is the most common scenario — model calls may fail due to network issues, and automatic retries can improve reliability:

Example

from dotenv import load_dotenv
load_dotenv()

from langchain.agents import create_agent
from langchain.agents.middleware import wrap_model_call
from langchain.chat_models import init_chat_model
from langchain.messages import HumanMessage


@wrap_model_call
def retry_on_error(request, handler):
    """Automatically retry when model call fails, up to 3 times"""
    max_retries = 3
    last_error = None

    for attempt in range(max_retries):
        try:
            result = handler(request)
            if attempt > 0:
                print(f" [Retry succeeded] Attempt {attempt + 1}")
            return result
        except Exception as e:
            last_error = e
            if attempt < max_retries - 1:
                wait_time = (attempt + 1) * 2  # Incremental waiting: 2s, 4s
                print(f" [Retry] Attempt {attempt + 1} failed, retrying after {wait_time} seconds...")
                import time
                time.sleep(wait_time)

    # All retries failed
    raise last_error


model = init_chat_model("deepseek:deepseek-v4-flash", temperature=0)
agent = create_agent(
    model=model,
    middleware=[retry_on_error],
    system_prompt="You are the assistant of EXAMPLE Tutorial.",
)

result = agent.invoke({
    "messages": [HumanMessage(content="Introduce Python Tutorial")]
})
print(f"\nReply: {result['messages'][-1].content[:100]}...")

Scenario 2: Model Degradation / Failover

When the primary model is unavailable, automatically switch to a backup model:

Example

from langchain.agents.middleware import wrap_model_call
from langchain.chat_models import init_chat_model
from langchain.messages import AIMessage


# Primary model and backup model
primary_model = init_chat_model("deepseek:deepseek-v4-flash", temperature=0)
fallback_model = init_chat_model("deepseek:deepseek-v4-flash", temperature=0,
                                  max_tokens=100)  # Limit tokens for the degraded model


@wrap_model_call
def fallback_on_error(request, handler):
    """Automatically switch to the backup model when the primary model fails"""
    try:
        # Try using the primary model
        return handler(request)
    except Exception as e:
        print(f"[Degraded] Primary model failed: {e}, switching to backup model...")

        # Override the model in request, switch to backup model
        request = request.override(model=fallback_model)
        try:
            return handler(request)
        except Exception as e2:
            print(f"[Degraded] Backup model also failed: {e2}")
            # Both models failed, return a friendly message
            return AIMessage(
                content="Sorry, the service is temporarily unavailable, please try again later."
            )

request.override() is an immutable method — it returns a new copy of request without modifying the original object. This ensures each call is independent and safe.


Scenario 3: Caching Model Responses

For repeated queries, you can cache model responses to reduce API call costs:

Example

from langchain.agents.middleware import wrap_model_call
from langchain.messages import AIMessage


# A simple in-memory cache
cache = {}


@wrap_model_call
def cache_responses(request, handler):
    """Cache model responses, do not call repeatedly for the same question"""
    # Use the content of the last user message as the cache key
    messages = request.messages
    if not messages:
        return handler(request)

    # Generate cache key
    last_content = str(messages[-1].content) if hasattr(messages[-1], 'content') else ""
    cache_key = last_content[:200]  # Truncate overly long content

    # Check cache
    if cache_key in cache:
        print(f"[Cache hit] Directly return the cached result")
        cached = cache[cache_key]
        return AIMessage(content=f"{cached}\n\n*(from cache)*")

    # Cache miss, call the model
    result = handler(request)
    # AIMessage's content may be mixed in a list, directly take the first one
    if hasattr(result, 'content'):
        cache[cache_key] = result.content
        print(f"[Cache miss] Stored in cache, current {len(cache)} entries")
    elif hasattr(result, 'model_response'):
        # The case of ExtendedModelResponse
        pass

    return result


model = init_chat_model("deepseek:deepseek-v4-flash", temperature=0)
agent = create_agent(
    model=model,
    middleware=[cache_responses],
    system_prompt="You are the assistant of EXAMPLE Tutorial, and you must answer concisely.",
)

# First query (cache miss)
result = agent.invoke({
    "messages": [{"role": "user", "content": "What is Python?"}]
})
print(f"First time: {result['messages'][-1].content[:80]}...\n")

# Second query with the same question (cache hit)
result = agent.invoke({
    "messages": [{"role": "user", "content": "What is Python?"}]
})
print(f"Second time: {result['messages'][-1].content[:80]}...")

Run result:

[缓存未命中] 已存入缓存,当前 1 条
第一次: Python 是一种高级编程语言,以简洁易读的语法著称...

[缓存命中] 直接返回缓存结果
第二次: Python 是一种高级编程语言,以简洁易读的语法著称...
*(来自缓存)*

Scenario 4: Modifying request — Dynamically Injecting System Messages

Example

from datetime import datetime
from langchain.agents.middleware import wrap_model_call
from langchain.messages import SystemMessage


@wrap_model_call
def inject_time_context(request, handler):
    """Inject current time information before each model call"""
    now = datetime.now()
    time_context = (
        f"Current time: {now.strftime('%Y year %m month %d day %H:%M')}."
        f"Today is {['Monday','Tuesday','Wednesday','Thursday','Friday','Saturday','Sunday'][now.weekday()]}."
    )

    # Append time information to the existing system_message
    if request.system_message:
        new_content = f"{request.system_message.content}\n\n{time_context}"
    else:
        new_content = time_context

    # Use override to create a new request
    new_request = request.override(
        system_message=SystemMessage(content=new_content)
    )

    return handler(new_request)

Scenario 5: Combining Multiple wrap_model_call

Multiple wrap_model_call middlewares are automatically combined in order — the first one is at the outermost layer:

Example

from langchain.agents.middleware import wrap_model_call


@wrap_model_call
def outer_middleware(request, handler):
    """Outermost middleware"""
    print("[Outer] Start")
    result = handler(request)       # Here it enters inner_middleware
    print("[Outer] End")
    return result


@wrap_model_call
def inner_middleware(request, handler):
    """Inner middleware"""
    print(" [Inner] Start")
    result = handler(request)       # Here the model is actually called
    print(" [Inner] End")
    return result


# Execution order:
# [Outer] Start
# [Inner] Start
# → Actually calling the model
# [Inner] End
# [Outer] End

Multiple wrap_model_call calls wrap layer by layer like an onion. The outermost layer executes first and returns last. This allows you to combine multiple independent features — for example, the outer layer does cache checking, the inner layer does retries, without interfering with each other.

Other Extensions