Tool Search with Embeddings: Scaling Haijun to Thousands of Tools
Building Haijun applications with dozens of specialized tools quickly hits a wall: providing all tool definitions upfront consumes your context window, increases latency and costs, and makes it harder for Haijun to find the right tool. Beyond ~100 tools, this approach becomes impractical.
Semantic tool search solves this by treating tools as discoverable resources. Instead of front-loading hundreds of definitions, you give Haijun a single tool_search tool that returns relevant capabilities on demand, cutting context usage by 90%+ while enabling applications that scale to thousands of tools.
Use semantic embeddings to dynamically discover relevant tools based on task context
Apply this pattern to domain-specific tool libraries (APIs, databases, internal systems)
This pattern is used in production by teams managing large tool ecosystems where context efficiency is critical. While we'll demonstrate with a small set of tools for clarity, the same approach scales seamlessly to libraries with hundreds or thousands of tools.
Setup
First, install the required dependencies:
all-MiniLM-L6-v2 is a lightweight model with 384 dimensional embeddings
It will be downloaded from HuggingFace on first use
print("Loading SentenceTransformer model...")
embedding_model = SentenceTransformer("sentence-transformers/all-MiniLM-L6-v2")
print("✓ Clients initialized successfully")
Loading SentenceTransformer model... ✓ Clients initialized successfully Define Tool Library Before we can implement semantic search, we need tools to search through. We'll create a library of 8 tools across two categories: Weather and Finance.
"properties": {
"ticker": {
"type": "string",
"description": "Stock ticker symbol (e.g., AAPL, GOOGL)",
},
"include_history": {
"type": "boolean",
"description": "Include historical data",
},
},
"required": ["ticker"],
},
},
{
"name": "convert_currency",
"description": "Convert an amount from one currency to another using current exchange rates",
"input_schema": {
"type": "object",
"properties": {
"amount": {
"type": "number",
"description": "Amount to convert",
},
"from_currency": {
"type": "string",
"description": "Source currency code (e.g., USD)",
},
"to_currency": {
"type": "string",
"description": "Target currency code (e.g., EUR)",
},
},
"required": ["amount", "from_currency", "to_currency"],
},
},
{
"name": "calculate_compound_interest",
"description": "Calculate compound interest for investments over time",
"input_schema": {
"type": "object",
"properties": {
"principal": {
"type": "number",
"description": "Initial investment amount",
},
"rate": {
"type": "number",
"description": "Annual interest rate (as percentage)",
},
"years": {"type": "number", "description": "Number of years"},
"frequency": {
"type": "string",
"enum": ["daily", "monthly", "quarterly", "annually"],
"description": "Compounding frequency",
},
},
"required": ["principal", "rate", "years"],
},
},
{
"name": "get_market_news",
"description": "Get recent financial news and market updates for a specific company or sector",
"input_schema": {
"type": "object",
"properties": {
"query": {
"type": "string",
"description": "Company name, ticker symbol, or sector",
},
"limit": {
"type": "number",
"description": "Maximum number of news articles to return",
},
},
"required": ["query"],
},
},
]
print(f"✓ Defined {len(TOOL_LIBRARY)} tools in the library")
✓ Defined 8 tools in the library Create Tool Embeddings Semantic search works by comparing the meaning of text, rather than just searching for keywords. To enable this, we need to convert each tool definition into an embedding vector that captures its semantic meaning.
"timezone": "UTC+9",
"current_time": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
"utc_offset": "+09:00",
}
)
elif tool_name == "get_air_quality":
location = tool_input.get("location", "Unknown")
aqi = random.randint(20, 150)
categories = {
(0, 50): "Good",
(51, 100): "Moderate",
(101, 150): "Unhealthy for Sensitive Groups",
}
category = next(cat for (low, high), cat in categories.items() if low <= aqi <= high)
return json.dumps(
{
"location": location,
"aqi": aqi,
"category": category,
"pollutants": {
"pm25": random.randint(5, 50),
"pm10": random.randint(10, 100),
"o3": random.randint(20, 80),
},
}
)
Finance tools
elif tool_name == "get_stock_price":
ticker = tool_input.get("ticker", "UNKNOWN")
return json.dumps(
{
"ticker": ticker,
"price": round(random.uniform(100, 500), 2),
"change": round(random.uniform(-5, 5), 2),
"change_percent": round(random.uniform(-2, 2), 2),
"volume": random.randint(1000000, 10000000),
"market_cap": f"${random.randint(100, 1000)}B",
}
)
elif tool_name == "convert_currency":
amount = tool_input.get("amount", 0)
from_currency = tool_input.get("from_currency", "USD")
to_currency = tool_input.get("to_currency", "EUR")
Mock exchange rate
rate = random.uniform(0.8, 1.2)
converted = round(amount * rate, 2)
return json.dumps(
{
"original_amount": amount,
"from_currency": from_currency,
"to_currency": to_currency,
"exchange_rate": round(rate, 4),
"converted_amount": converted,
}
)
elif tool_name == "calculate_compound_interest":
principal = tool_input.get("principal", 0)
rate = tool_input.get("rate", 0)
years = tool_input.get("years", 0)
frequency = tool_input.get("frequency", "monthly")
Calculate compound interest
n_map = {"daily": 365, "monthly": 12, "quarterly": 4, "annually": 1}
n = n_map.get(frequency, 12)
final_amount = principal * (1 + rate / 100 / n) ** (n * years)
interest_earned = final_amount - principal
return json.dumps(
{
"principal": principal,
"rate": rate,
"years": years,
"compounding_frequency": frequency,
"final_amount": round(final_amount, 2),
"interest_earned": round(interest_earned, 2),
}
)
elif tool_name == "get_market_news":
query = tool_input.get("query", "")
limit = tool_input.get("limit", 5)
news = []
for i in range(min(limit, 5)):
news.append(
{
"title": f"{query} - News Article {i + 1}",
"source": random.choice(
[
"Bloomberg",
"Reuters",
"Financial Times",
"Wall Street Journal",
]
),
"published": (datetime.now() - timedelta(hours=random.randint(1, 24))).strftime(
"%Y-%m-%d %H:%M"
),
"summary": f"Latest developments regarding {query}...",
}
)
return json.dumps({"query": query, "articles": news, "count": len(news)})
Default fallback
else:
return json.dumps(
{
"status": "executed",
"tool": tool_name,
"message": f"Tool {tool_name} executed successfully with input: {json.dumps(tool_input)}",
}
)
print("✓ Mock tool execution function created")
✓ Mock tool execution function created Implement Conversation Loop Now let's put it all together! We'll create a conversation loop that handles the complete tool search workflow.