Skip to content
Blog

Tool-Augmented Retrieval: Chaining Search Tools with Agent Reasoning

Build retrieval systems where agents chain multiple search tools—vector, keyword, database, and API—using reasoning to decide what to search and when to stop.

Published on • September 15, 2026

AI Assistant

Standard RAG retrieves from one index. Tool-augmented retrieval gives agents multiple search tools and lets them reason about which to use, chain them together, and synthesize across sources.

Beyond Single-Index RAG

Real knowledge lives in multiple places:

  • Vector databases for semantic search
  • Keyword indexes for exact matches
  • SQL databases for structured queries
  • APIs for real-time data
  • File systems for documents

An agent that can chain these tools can answer questions no single-source RAG system can.

Architecture

User Query
    │
    ▼
┌─────────────────┐
│  Agent Reasoner  │
│                  │
│  "What tools do  │
│   I need?"       │
└────┬────────┬───┘
     │        │
     ▼        ▼
┌─────────┐ ┌─────────┐
│ Vector  │ │Keyword  │
│ Search  │ │ Search  │
└────┬────┘ └────┬────┘
     │           │
     ▼           ▼
┌─────────────────┐
│  Synthesizer    │
│  "Combine and   │
│   rank results" │
└─────────────────┘

Implementation with LangChain

from langchain_core.tools import tool
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage
from langgraph.prebuilt import create_react_agent

# Define search tools
@tool
def vector_search(query: str) -> str:
    """Search the vector database for semantically similar content. Use for conceptual questions and finding related documents."""
    results = vector_store.similarity_search(query, k=5)
    return "\n\n".join([f"[{r.metadata['source']}] {r.page_content}" for r in results])

@tool
def keyword_search(query: str) -> str:
    """Search using exact keyword matching. Use when searching for specific terms, names, or codes."""
    results = bm25_retriever.invoke(query)
    return "\n\n".join([r.page_content for r in results[:5]])

@tool
def sql_query(query: str) -> str:
    """Execute a SQL query against the product database. Use for structured data like prices, inventory, specifications."""
    import sqlite3
    conn = sqlite3.connect("products.db")
    try:
        cursor = conn.execute(query)
        columns = [desc[0] for desc in cursor.description]
        rows = cursor.fetchall()
        return str({"columns": columns, "rows": rows[:20]})
    except Exception as e:
        return f"SQL Error: {e}"
    finally:
        conn.close()

@tool
def web_search(query: str) -> str:
    """Search the web for current information. Use for recent events, news, or information not in internal databases."""
    # Implementation using a search API
    results = search_api.search(query, num_results=3)
    return "\n\n".join([f"{r['title']}: {r['snippet']}" for r in results])

# Create the agent
llm = ChatOpenAI(model="gpt-4o", temperature=0)
tools = [vector_search, keyword_search, sql_query, web_search]

agent = create_react_agent(llm, tools)

Chaining Strategies

Sequential Chaining

Agent uses results from one search to inform the next:

def sequential_chain(agent, query: str) -> dict:
    """Run agent with context that encourages sequential tool use."""
    
    enhanced_prompt = f"""
    You have access to multiple search tools. Consider using them sequentially:
    
    1. Start with vector_search for conceptual understanding
    2. Use keyword_search for specific terms found in step 1
    3. Use sql_query for structured data about specific items
    4. Use web_search only if internal sources don't have the answer
    
    Query: {query}
    """
    
    result = agent.invoke({"messages": [HumanMessage(content=enhanced_prompt)]})
    
    return {
        "answer": result["messages"][-1].content,
        "tools_used": extract_tool_calls(result["messages"]),
        "num_steps": len(result["messages"]),
    }

Parallel Chaining

Search multiple sources simultaneously:

import asyncio
from langchain_core.runnables import RunnableParallel

async def parallel_search(query: str) -> dict:
    """Search multiple sources in parallel and merge results."""
    
    search_tasks = {
        "vector": vector_search.ainvoke(query),
        "keyword": keyword_search.ainvoke(query),
        "sql": sql_query.ainvoke(f"SELECT * FROM products WHERE description LIKE '%{query}%'"),
    }
    
    results = await asyncio.gather(
        *search_tasks.values(),
        return_exceptions=True
    )
    
    merged = {}
    for name, result in zip(search_tasks.keys(), results):
        if isinstance(result, Exception):
            merged[name] = f"Error: {result}"
        else:
            merged[name] = result
    
    return merged

Adaptive Chaining

Agent decides dynamically based on intermediate results:

def adaptive_retrieve(query: str) -> dict:
    """Agent adapts its search strategy based on what it finds."""
    
    # First pass: try vector search
    vector_results = vector_search.invoke(query)
    
    if is_high_confidence(vector_results):
        return {"strategy": "vector_only", "results": vector_results}
    
    # Second pass: add keyword search
    keyword_results = keyword_search.invoke(query)
    combined = combine_results(vector_results, keyword_results)
    
    if is_high_confidence(combined):
        return {"strategy": "vector+keyword", "results": combined}
    
    # Third pass: try structured data
    sql_results = sql_query.invoke(extract_entities(query))
    
    return {
        "strategy": "full_chain",
        "results": {
            "vector": vector_results,
            "keyword": keyword_results,
            "structured": sql_results,
        }
    }

Re-Ranking Across Sources

from sentence_transformers import CrossEncoder

class CrossSourceReranker:
    def __init__(self):
        self.reranker = CrossEncoder("cross-encoder/ms-marco-MiniLM-L-6-v2")
    
    def rerank(self, query: str, results_by_source: dict[str, list], top_k: int = 5) -> list:
        # Flatten all results
        all_results = []
        for source, results in results_by_source.items():
            for r in results:
                all_results.append({
                    "content": r["content"] if isinstance(r, dict) else r.page_content,
                    "source": source,
                    "original_score": r.get("score", 0) if isinstance(r, dict) else 0,
                })
        
        # Score with cross-encoder
        pairs = [(query, r["content"]) for r in all_results]
        scores = self.reranker.predict(pairs)
        
        for i, score in enumerate(scores):
            all_results[i]["rerank_score"] = float(score)
        
        # Sort and return top_k
        all_results.sort(key=lambda x: x["rerank_score"], reverse=True)
        return all_results[:top_k]

Citation-Anchored Answers

def generate_cited_answer(query: str, results: list) -> dict:
    """Generate an answer with inline citations to source documents."""
    
    context_parts = []
    for i, r in enumerate(results, 1):
        context_parts.append(f"[{i}] ({r['source']}) {r['content']}")
    
    context = "\n\n".join(context_parts)
    
    prompt = f"""Answer the question based on the provided context. 
    Include inline citations using the [number] format.
    
    Context:
    {context}
    
    Question: {query}
    
    Answer with citations:"""
    
    response = llm.invoke([HumanMessage(content=prompt)])
    
    return {
        "answer": response.content,
        "sources": [{"id": i, "source": r["source"], "content": r["content"][:200]} 
                    for i, r in enumerate(results, 1)],
    }

Evaluation Metrics for Tool-Augmented Retrieval

def evaluate_tool_augmented_retrieval(test_cases: list) -> dict:
    metrics = {
        "retrieval_quality": [],
        "tool_selection_accuracy": [],
        "chain_efficiency": [],
        "answer_quality": [],
    }
    
    for case in test_cases:
        result = adaptive_retrieve(case["query"])
        
        # Did it use the right tools?
        expected_tools = case.get("expected_tools", [])
        actual_tools = result.get("strategy", "").split("+")
        tool_accuracy = len(set(expected_tools) & set(actual_tools)) / max(len(expected_tools), 1)
        metrics["tool_selection_accuracy"].append(tool_accuracy)
        
        # How many steps did it take?
        chain_length = len(actual_tools)
        optimal_length = len(expected_tools)
        efficiency = 1.0 if chain_length <= optimal_length else optimal_length / chain_length
        metrics["chain_efficiency"].append(efficacy)
    
    return {k: sum(v) / len(v) for k, v in metrics.items() if v}

Tool-augmented retrieval transforms RAG from a single-tool operation into a research workflow. The agent becomes a researcher who knows when to search broadly, when to search specifically, and when to combine sources for a complete answer.