Tool-Augmented Retrieval: Chaining Search Tools with Agent Reasoning
Build retrieval systems where agents chain multiple search tools—vector, keyword, database, and API—using reasoning to decide what to search and when to stop.
Published on • September 15, 2026
AI Assistant

Standard RAG retrieves from one index. Tool-augmented retrieval gives agents multiple search tools and lets them reason about which to use, chain them together, and synthesize across sources.
Beyond Single-Index RAG
Real knowledge lives in multiple places:
- Vector databases for semantic search
- Keyword indexes for exact matches
- SQL databases for structured queries
- APIs for real-time data
- File systems for documents
An agent that can chain these tools can answer questions no single-source RAG system can.
Architecture
User Query
│
▼
┌─────────────────┐
│ Agent Reasoner │
│ │
│ "What tools do │
│ I need?" │
└────┬────────┬───┘
│ │
▼ ▼
┌─────────┐ ┌─────────┐
│ Vector │ │Keyword │
│ Search │ │ Search │
└────┬────┘ └────┬────┘
│ │
▼ ▼
┌─────────────────┐
│ Synthesizer │
│ "Combine and │
│ rank results" │
└─────────────────┘
Implementation with LangChain
from langchain_core.tools import tool
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage
from langgraph.prebuilt import create_react_agent
# Define search tools
@tool
def vector_search(query: str) -> str:
"""Search the vector database for semantically similar content. Use for conceptual questions and finding related documents."""
results = vector_store.similarity_search(query, k=5)
return "\n\n".join([f"[{r.metadata['source']}] {r.page_content}" for r in results])
@tool
def keyword_search(query: str) -> str:
"""Search using exact keyword matching. Use when searching for specific terms, names, or codes."""
results = bm25_retriever.invoke(query)
return "\n\n".join([r.page_content for r in results[:5]])
@tool
def sql_query(query: str) -> str:
"""Execute a SQL query against the product database. Use for structured data like prices, inventory, specifications."""
import sqlite3
conn = sqlite3.connect("products.db")
try:
cursor = conn.execute(query)
columns = [desc[0] for desc in cursor.description]
rows = cursor.fetchall()
return str({"columns": columns, "rows": rows[:20]})
except Exception as e:
return f"SQL Error: {e}"
finally:
conn.close()
@tool
def web_search(query: str) -> str:
"""Search the web for current information. Use for recent events, news, or information not in internal databases."""
# Implementation using a search API
results = search_api.search(query, num_results=3)
return "\n\n".join([f"{r['title']}: {r['snippet']}" for r in results])
# Create the agent
llm = ChatOpenAI(model="gpt-4o", temperature=0)
tools = [vector_search, keyword_search, sql_query, web_search]
agent = create_react_agent(llm, tools)
Chaining Strategies
Sequential Chaining
Agent uses results from one search to inform the next:
def sequential_chain(agent, query: str) -> dict:
"""Run agent with context that encourages sequential tool use."""
enhanced_prompt = f"""
You have access to multiple search tools. Consider using them sequentially:
1. Start with vector_search for conceptual understanding
2. Use keyword_search for specific terms found in step 1
3. Use sql_query for structured data about specific items
4. Use web_search only if internal sources don't have the answer
Query: {query}
"""
result = agent.invoke({"messages": [HumanMessage(content=enhanced_prompt)]})
return {
"answer": result["messages"][-1].content,
"tools_used": extract_tool_calls(result["messages"]),
"num_steps": len(result["messages"]),
}
Parallel Chaining
Search multiple sources simultaneously:
import asyncio
from langchain_core.runnables import RunnableParallel
async def parallel_search(query: str) -> dict:
"""Search multiple sources in parallel and merge results."""
search_tasks = {
"vector": vector_search.ainvoke(query),
"keyword": keyword_search.ainvoke(query),
"sql": sql_query.ainvoke(f"SELECT * FROM products WHERE description LIKE '%{query}%'"),
}
results = await asyncio.gather(
*search_tasks.values(),
return_exceptions=True
)
merged = {}
for name, result in zip(search_tasks.keys(), results):
if isinstance(result, Exception):
merged[name] = f"Error: {result}"
else:
merged[name] = result
return merged
Adaptive Chaining
Agent decides dynamically based on intermediate results:
def adaptive_retrieve(query: str) -> dict:
"""Agent adapts its search strategy based on what it finds."""
# First pass: try vector search
vector_results = vector_search.invoke(query)
if is_high_confidence(vector_results):
return {"strategy": "vector_only", "results": vector_results}
# Second pass: add keyword search
keyword_results = keyword_search.invoke(query)
combined = combine_results(vector_results, keyword_results)
if is_high_confidence(combined):
return {"strategy": "vector+keyword", "results": combined}
# Third pass: try structured data
sql_results = sql_query.invoke(extract_entities(query))
return {
"strategy": "full_chain",
"results": {
"vector": vector_results,
"keyword": keyword_results,
"structured": sql_results,
}
}
Re-Ranking Across Sources
from sentence_transformers import CrossEncoder
class CrossSourceReranker:
def __init__(self):
self.reranker = CrossEncoder("cross-encoder/ms-marco-MiniLM-L-6-v2")
def rerank(self, query: str, results_by_source: dict[str, list], top_k: int = 5) -> list:
# Flatten all results
all_results = []
for source, results in results_by_source.items():
for r in results:
all_results.append({
"content": r["content"] if isinstance(r, dict) else r.page_content,
"source": source,
"original_score": r.get("score", 0) if isinstance(r, dict) else 0,
})
# Score with cross-encoder
pairs = [(query, r["content"]) for r in all_results]
scores = self.reranker.predict(pairs)
for i, score in enumerate(scores):
all_results[i]["rerank_score"] = float(score)
# Sort and return top_k
all_results.sort(key=lambda x: x["rerank_score"], reverse=True)
return all_results[:top_k]
Citation-Anchored Answers
def generate_cited_answer(query: str, results: list) -> dict:
"""Generate an answer with inline citations to source documents."""
context_parts = []
for i, r in enumerate(results, 1):
context_parts.append(f"[{i}] ({r['source']}) {r['content']}")
context = "\n\n".join(context_parts)
prompt = f"""Answer the question based on the provided context.
Include inline citations using the [number] format.
Context:
{context}
Question: {query}
Answer with citations:"""
response = llm.invoke([HumanMessage(content=prompt)])
return {
"answer": response.content,
"sources": [{"id": i, "source": r["source"], "content": r["content"][:200]}
for i, r in enumerate(results, 1)],
}
Evaluation Metrics for Tool-Augmented Retrieval
def evaluate_tool_augmented_retrieval(test_cases: list) -> dict:
metrics = {
"retrieval_quality": [],
"tool_selection_accuracy": [],
"chain_efficiency": [],
"answer_quality": [],
}
for case in test_cases:
result = adaptive_retrieve(case["query"])
# Did it use the right tools?
expected_tools = case.get("expected_tools", [])
actual_tools = result.get("strategy", "").split("+")
tool_accuracy = len(set(expected_tools) & set(actual_tools)) / max(len(expected_tools), 1)
metrics["tool_selection_accuracy"].append(tool_accuracy)
# How many steps did it take?
chain_length = len(actual_tools)
optimal_length = len(expected_tools)
efficiency = 1.0 if chain_length <= optimal_length else optimal_length / chain_length
metrics["chain_efficiency"].append(efficacy)
return {k: sum(v) / len(v) for k, v in metrics.items() if v}
Tool-augmented retrieval transforms RAG from a single-tool operation into a research workflow. The agent becomes a researcher who knows when to search broadly, when to search specifically, and when to combine sources for a complete answer.