π Build and Evaluate a Web Search Agent - π¦ TruLens
π Build and Evaluate a Web Search Agent
Build a web-enabled data agent that can operate across perform web research, answer questions, and generate charts. Then evaluate it to identify failure modes.
For this example you will need access to LLMs (OpenAI) and web search (Tavily).
InΒ [Β ]:
!pip install trulens langgraph trulens-apps-langgraph trulens-providers-openai openai matplotlib langchain_openai langchain_tavily langchain_experimental -q
InΒ [Β ]:
import os
os.environ["OPENAI_API_KEY"] = "sk-proj..."
os.environ["TAVILY_API_KEY"] = "tvly-dev..."
1. Initialize the agent's state
State provides the agent a shared, evolving memory across nodes so that the agents have the context and instructions needed to act coherently and achieve the goal.
In addition to the additional state variables we're adding, our State will also inherit messages from MessageState to track the conversation.
InΒ [Β ]:
from typing import Any, Dict, List, Literal, Optional
from langgraph.graph import MessagesState
# Custom State class with specific keys
class State(MessagesState):
user_query: Optional[str] # The user's original query
enabled_agents: Optional[List[str]] # Makes our multi-agent system modular on which agents to include
plan: Optional[List[Dict[int, Dict[str, Any]]]] # Listing the steps in the plan needed to achieve the goal.
current_step: int # Marking the current step in the plan.
agent_query: Optional[str] # Inbox note: `agent_query` tells the next agent exactly what to do at the current step.
last_reason: Optional[str] # Explains the executorβs decision to help maintain continuity and provide traceability.
replan_flag: Optional[bool] # Set by the executor to indicate that the planner should revise the plan.
replan_attempts: Optional[Dict[int, Dict[int, int]]] # Replan attempts tracked per step number.
2. Create planner
InΒ [Β ]:
import json
from typing import Any, Dict, List
from langchain_core.messages import HumanMessage
MAX_REPLANS = 2
def get_agent_descriptions() -> Dict[str, Dict[str, Any]]:
"""
Return structured agent descriptions with capabilities and guidelines.
Edit this function to change how the planner/executor reason about agents.
"""
return {
"web_researcher": {
"name": "Web Researcher",
"capability": "Fetch public data via Tavily web search",
"use_when": "Public information, news, current events, or external facts are needed",
"limitations": "Cannot access private/internal company data",
"output_format": "Raw research data and findings from public sources",
},
"chart_generator": {
"name": "Chart Generator",
"capability": "Build visualizations from structured data",
"use_when": "User explicitly requests charts, graphs, plots, visualizations (keywords: chart, graph, plot, visualise, bar-chart, line-chart, histogram, etc.)",
"limitations": "Requires structured data input from previous steps",
"output_format": "Visual charts and graphs",
"position_requirement": "Must be used as final step after data gathering is complete",
},
"chart_summarizer": {
"name": "Chart Summarizer",
"capability": "Summarize and explain chart visualizations",
"use_when": "After chart_generator has created a visualization",
"limitations": "Requires a chart as input",
"output_format": "Written summary and analysis of chart content",
},
"synthesizer": {
"name": "Synthesizer",
"capability": "Write comprehensive prose summaries of findings",
"use_when": "Final step when no visualization is requested - combines all previous research",
"limitations": "Requires research data from previous steps",
"output_format": "Coherent written summary incorporating all findings",
"position_requirement": "Should be used as final step when no chart is needed",
},
}
def _get_enabled_agents(state: State | None = None) -> List[str]:
"""Return enabled agents; if absent, use baseline/default.
"""
baseline = ["web_researcher","chart_generator","chart_summarizer","synthesizer",]
if not state:
return baseline
val = (
state.get("enabled_agents")
if hasattr(state, "get")
else getattr(state, "enabled_agents", None)
)
if isinstance(val, list) and val:
allowed = {"web_researcher", "chart_generator", "chart_summarizer", "synthesizer",}
filtered = [a for a in val if a in allowed]
return filtered
return baseline
def format_agent_list_for_planning(state: State | None = None) -> str:
"""
Format agent descriptions for the planning prompt.
"""
descriptions = get_agent_descriptions()
enabled_list = _get_enabled_agents(state)
agent_list = []
for agent_key, details in descriptions.items():
if agent_key not in enabled_list:
continue
agent_list.append(f" β’ ` {agent_key}` β {details['capability']}")
return "\n".join(agent_list)
def format_agent_guidelines_for_planning(state: State | None = None) -> str:
"""
Format agent usage guidelines for the planning prompt.
"""
descriptions = get_agent_descriptions()
enabled = set(_get_enabled_agents(state))
guidelines = []
if "web_researcher" in enabled:
guidelines.append(
f"- Use `web_researcher` for {descriptions['web_researcher']['use_when'].lower()}.")
# Chart generator specific rules
if "chart_generator" in enabled:
chart_desc = descriptions["chart_generator"]
cs_hint = (
" A `chart_summarizer` should be used to summarize the chart."
if "chart_summarizer" in enabled
else ""
)
guidelines.append(
f"- **Include `chart_generator` _only_ if {chart_desc['use_when'].lower()}**. If included, `chart_generator` must be {chart_desc['position_requirement'].lower()}. Visualizations should include all of the data from the previous steps that are reasonable for the chart type.{cs_hint}")
# Synthesizer default
if "synthesizer" in enabled:
synth_desc = descriptions["synthesizer"]
guidelines.append(
f" β Otherwise use `synthesizer` as {synth_desc['position_requirement'].lower()}, and be sure to include all of the data from the previous steps.")
return "\n".join(guidelines)
def plan_prompt(state: State) -> HumanMessage:
"""
Build the prompt that instructs the LLM to return a highβlevel plan.
"""
replan_flag = state.get("replan_flag", False)
user_query = state.get("user_query", state["messages"][0].content)
prior_plan = state.get("plan") or {}
replan_reason = state.get("last_reason", "")
# Get agent descriptions dynamically
agent_list = format_agent_list_for_planning(state)
agent_guidelines = format_agent_guidelines_for_planning(state)
enabled_list = _get_enabled_agents(state)
# Build planner agent enum based on enabled agents
enabled_for_planner = [
a
for a in enabled_list
if a
in ("web_researcher","cortex_researcher","chart_generator","synthesizer",)
]
planner_agent_enum = (
" | ".join(enabled_for_planner)
or "web_researcher | chart_generator | synthesizer"
)
prompt = f"""
You are the **Planner** in a multiβagent system. Break the user's request
into a sequence of numbered steps (1,β―2,β―3, β¦). **There is no hard limit on
step count** as long as the plan is concise and each step has a clear goal.
You may decompose the user's query into sub-queries, each of which is a
separate step. Break the query into the smallest possible sub-queries
so that each sub-query is answerable with a single data source.
For example, if the user's query is "What were the key
action items in the last quarter, and what was a recent news story for
each of them?", you may break it into steps:
1. Fetch the key action items in the last quarter.
2. Fetch a recent news story for the first action item.
3. Fetch a recent news story for the second action item.
4. Fetch a recent news story for the last action item
Here is a list of available agents you can call upon to execute the tasks in your plan. You may call only one agent per step.
{agent_list}
Return **ONLY** valid JSON (no markdown, no explanations) in this form:
{{
"1": {{
"agent": "{planner_agent_enum}",
"action": "string",
"goal": "string",
"pre_conditions": ["string", ...],
"post_conditions": ["string", ...]
}},
"2": {{ ... }},
"3": {{ ... }}
}}
Guidelines:
{agent_guidelines}
"""
if replan_flag:
prompt += f"""
The current plan needs revision because: {replan_reason}
Current plan:
{json.dumps(prior_plan, indent=2)}
When replanning:
- Focus on UNBLOCKING the workflow rather than perfecting it.
- Only modify steps that are truly preventing progress.
- Prefer simpler, more achievable alternatives over complex rewrites.
"""
else:
prompt += "\nGenerate a new plan from scratch."
prompt += f'\nUser query: "{user_query}"'
return HumanMessage(content=prompt)
InΒ [Β ]:
from langchain_core.messages import HumanMessage
from langchain_openai import ChatOpenAI
from langgraph.types import Command
# ββ LLMs ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
reasoning_llm = ChatOpenAI(
model="o3",
model_kwargs={"response_format": {"type": "json_object"}},
)
def planner_node(state: State) -> Command[Literal["executor"]]:
"""
Runs the planning LLM and stores the resulting plan in state.
"""
# 1. Invoke LLM with the planner prompt
llm_reply = reasoning_llm.invoke([plan_prompt(state)])
# 2. Validate JSON
try:
content_str = (
llm_reply.content
if isinstance(llm_reply.content, str)
else str(llm_reply.content)
)
parsed_plan = json.loads(content_str)
except json.JSONDecodeError:
raise ValueError(f"Planner returned invalid JSON:\n{llm_reply.content}")
# 3. Store as current plan only
replan = state.get("replan_flag", False)
updated_plan: Dict[str, Any] = parsed_plan
return Command(
update={
"plan": updated_plan,
"messages": [
HumanMessage(
content=llm_reply.content,
name="replan" if replan else "initial_plan",
)
],
"user_query": state.get("user_query", state["messages"][0].content),
"current_step": 1 if not replan else state["current_step"],
# Preserve replan flag so executor runs planned agent once before reconsidering
"replan_flag": state.get("replan_flag", False),
"last_reason": "",
"enabled_agents": state.get("enabled_agents"),
},
goto="executor",
)
3. Create executor
InΒ [Β ]:
def format_agent_guidelines_for_executor(state: State | None = None) -> str:
"""
Format agent usage guidelines for the executor prompt.
"""
descriptions = get_agent_descriptions()
enabled = _get_enabled_agents(state)
guidelines = []
if "web_researcher" in enabled:
web_desc = descriptions["web_researcher"]
guidelines.append(
f"- Use `"web_researcher"` when {web_desc['use_when'].lower()}.")
if "cortex_researcher" in enabled:
cortex_desc = descriptions["cortex_researcher"]
guidelines.append(
f"- Use `"cortex_researcher"` for {cortex_desc['use_when'].lower()}.")
return "\n".join(guidelines)
def executor_prompt(state: State) -> HumanMessage:
"""
Build the singleβturn JSON prompt that drives the executor LLM.
"""
step = int(state.get("current_step", 0))
latest_plan: Dict[str, Any] = state.get("plan") or {}
plan_block: Dict[str, Any] = latest_plan.get(str(step), {})
max_replans = MAX_REPLANS
# Get agent guidelines dynamically
executor_guidelines = format_agent_guidelines_for_executor(state)
plan_agent = plan_block.get("agent", "web_researcher")
messages_tail = (state.get("messages") or [])[-4:]
executor_prompt = f"""
You are the **executor** in a multiβagent system with these agents:
`{"`, `".join(sorted(set([a for a in _get_enabled_agents(state) if a in ["web_researcher", "cortex_researcher", "chart_generator", "chart_summarizer", "synthesizer"]] + ["planner"]))}`.
**Tasks**
1. Decide if the current plan needs revision. β `"replan_flag": true|false`
2. Decide which agent to run next. β `"goto": "<agent_name>"`
3. Give oneβsentence justification. β `"reason": "<text>"`
4. Write the exact question that the chosen agent should answer
β "query": "<text>"
**Guidelines**
{executor_guidelines}
- After **{MAX_REPLANS}** failed replans for the same step, move on.
- If you *just replanned* (replan_flag is true) let the assigned agent try before
requesting another replan.
Respond **only** with valid JSON (no additional text):
{{
"replan": <true|false>,
"goto": "<{