google adk - agentic RAG basic code using custom model

There's alot of example that uses Gemini to do a RAG call. In this example I am going to use a custom llm engine to perform a RAG that I have setup in Agentic platform. I assume you have setup a RAG with your document. 

So instead of using the genai apporach, we are now using agentic approach to query our RAG. 

import asyncio
from urllib import response
from google.adk.agents import Agent
from google.adk.runners import Runner
from google.adk.sessions import InMemorySessionService
from google.genai import types
from google.adk.agents.callback_context import CallbackContext
from google.adk.models.llm_request import LlmRequest
from datetime import datetime
import os
import logging
from vertexai.preview import rag
from google.adk.agents.llm_agent import LlmAgent
from google.adk.models.lite_llm import LiteLlm
import google.cloud.logging
from google.adk.tools.retrieval.vertex_ai_rag_retrieval import VertexAiRagRetrieval
from google.adk.tools.agent_tool import AgentTool
from google.adk.agents import LlmAgent
import pandas as pd

import vertexai
# Ensure this matches your actual GCP Project ID, not a placeholder
vertexai.init(project="project-your-project-id", location="us-central1")

def get_stock_price(date: str) -> dict:
    """Gets the closing stock price for a given date.

    Args:
        date: The date to get the stock price for, in YYYY-MM-DD format.

    Returns:
        A dictionary containing the closing price, or an error message if the
        date is not found or the format is incorrect.
    """
    try:
        # Load the CSV file
        df = pd.read_csv('goog.csv')
        # Convert the 'Date' column to datetime objects
        df['Date'] = pd.to_datetime(df['Date'])
        # Convert the input string to a datetime object
        query_date = datetime.strptime(date, '%Y-%m-%d')
        # Find the row for the given date
        row = df[df['Date'] == query_date]

        if not row.empty:
            # Get the closing price
            close_price = row['Close'].iloc[0]
            return {"status": "success", "date": date, "closing_price": close_price}
        else:
            return {"status": "error", "message": f"No data found for date: {date}"}
    except FileNotFoundError:
        return {"status": "error", "message": "Stock data file (goog.csv) not found."}
    except Exception as e:
        return {"status": "error", "message": f"An error occurred: {str(e)}"}
   

# 1. Create the Vertex AI Search tool
full_rag_id="projects/your-project-number-not-id/
locations/us-central1/ragCorpora/3690422817699921920"

rag_vertex_retrieval = VertexAiRagRetrieval(
    name="retrieve_rag_documentation",
    description=(
        "Use this tool to retrieve documentation and reference
materials for the question from the RAG corpus,"
    ),
    rag_resources=[
        rag.RagResource(
            rag_corpus=full_rag_id
        )
    ],
    similarity_top_k=10,
    vector_distance_threshold=0.6,
)


model = LiteLlm(
    model="openai/empero-ai/Qwen3.8-2B-Distill-GGUF",
    api_base="http://localhost:8888/v1",  # Your local server
    api_key="sk-unsloth-4d0a1b198bd177a2a72ee1954585342a"  # Local servers don't require auth
)

# -------------------------------------------------------------
# 2. CREATE THE ADK AGENT
# -------------------------------------------------------------
banking_agent =  LlmAgent(
    name="search_and_qna_agent",
    model=model,
    tools=[rag_vertex_retrieval],
    instruction="""You are a helpful assistant that answers questions based on information found in the document store.
    Use the search tool to find relevant information before answering.
    If the answer isn't in the documents, say that you couldn't find the information.
    """,
    description="Answers questions using a specific Vertex AI Search datastore.",
)

# -------------------------------------------------------------
# 3. RUNNER & SESSION ORCHESTRATION
# -------------------------------------------------------------
async def main():
    # Initialize the session service and runner
    session_service = InMemorySessionService()
    runner = Runner(
        agent=banking_agent,
        app_name="banking_app",
        session_service=session_service
    )
   
    # Create a new session for a user
    app_name = "banking_app"
    user_id = "user_123"
    session = await session_service.create_session(app_name=app_name, user_id=user_id)

    # Wrap user input in a Content object
    user_message = types.Content(  
        role="user",
        parts=[types.Part.from_text(text="What is reference AA9483 for?")]
    )

    print("=== STARTING ADK AGENT RUNNER LOOP ===")
   
    # run_async executes the LLM call, handles function calling,
    # runs 'get_user_account_balance', and feeds the output back.
    async for event in runner.run_async(
        user_id=user_id,
        session_id=session.id,
        new_message=user_message
    ):
        #print(f"\n[EVENT TYPE]: {event.type}")

        # Check if the event contains content parts (responses or tool outputs)
        if hasattr(event, "content") and event.content:
            for part in event.content.parts:
                if part.text:
                    print(f"[TEXT OUTPUT]: {part.text}")
                elif part.function_call:
                    print(f"[TOOL CALL]  : {part.function_call.name}({dict(part.function_call.args)})")
                elif part.function_response:
                    print(f"[TOOL RESULT]: {part.function_response.response}")


        print("-------------------------------------------------------------")
        print(event)

asyncio.run(main())



And here is example of our outputs :-



I am quite surprise with the results. I didn't think that Qwen would actually make the call. Instead it would return some python script for me to run. This time, it actually query and return results from the RAG












Comments

Popular posts from this blog

NodeJS: Error: spawn EINVAL in window for node version 20.20 and 18.20

llama cpp running it in google colab