{
 "cells": [
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "# \ud83d\udd2c Day 4: Independent Lab \u2014 Build Your Own Agentic System\n",
    "\n",
    "## Overview\n",
    "\n",
    "In this lab you will:\n",
    "- Choose a business domain (one of four tracks)\n",
    "- Design 3 tools with clear descriptions\n",
    "- Write a system prompt for your agent\n",
    "- Test the agent on 5+ queries\n",
    "- Create a golden test set (8+ scenarios)\n",
    "- Evaluate and improve your agent (v1 \u2192 v2)\n",
    "\n",
    "**Duration:** 90\u2013120 minutes\n",
    "**Format:** Self-directed with instructor support\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Setup\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "!pip install -q -U google-genai\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "import os, json, time\n",
    "from datetime import datetime, timezone\n",
    "from google import genai\n",
    "from google.genai import types\n",
    "\n",
    "try:\n",
    "    from google.colab import userdata\n",
    "    os.environ[\"GEMINI_API_KEY\"] = userdata.get(\"GEMINI_API_KEY\")\n",
    "except Exception:\n",
    "    pass\n",
    "\n",
    "if not os.environ.get(\"GEMINI_API_KEY\"):\n",
    "    import getpass\n",
    "    os.environ[\"GEMINI_API_KEY\"] = getpass.getpass(\"Paste your GEMINI_API_KEY: \")\n",
    "\n",
    "client = genai.Client(api_key=os.environ[\"GEMINI_API_KEY\"])\n",
    "MODEL_ID = \"gemini-2.5-flash-lite\"\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "import os, json, time\n",
    "from datetime import datetime, timezone\n",
    "from google import genai\n",
    "from google.genai import types\n",
    "\n",
    "try:\n",
    "    from google.colab import userdata\n",
    "    os.environ[\"GEMINI_API_KEY\"] = userdata.get(\"GEMINI_API_KEY\")\n",
    "except Exception:\n",
    "    pass\n",
    "\n",
    "if not os.environ.get(\"GEMINI_API_KEY\"):\n",
    "    import getpass\n",
    "    os.environ[\"GEMINI_API_KEY\"] = getpass.getpass(\"Paste your GEMINI_API_KEY: \")\n",
    "\n",
    "client = genai.Client(api_key=os.environ[\"GEMINI_API_KEY\"])\n",
    "MODEL_ID = \"gemini-2.5-flash-lite\"\n",
    "\n",
    "# \u2500\u2500 Infrastructure (from Guided Lab) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "PROMPT_LOG = []\n",
    "\n",
    "def _now():\n",
    "    return datetime.now(timezone.utc).isoformat(timespec=\"seconds\").replace(\"+00:00\", \"Z\")\n",
    "\n",
    "def log_interaction(role, content, label=None):\n",
    "    entry = {\"ts\": _now(), \"role\": role,\n",
    "             \"content\": content if isinstance(content, str) else json.dumps(content),\n",
    "             \"label\": label or \"\"}\n",
    "    PROMPT_LOG.append(entry)\n",
    "    return entry\n",
    "\n",
    "def show_log(n=10):\n",
    "    import pandas as pd\n",
    "    if not PROMPT_LOG:\n",
    "        print(\"No interactions logged yet.\")\n",
    "        return\n",
    "    df = pd.DataFrame(PROMPT_LOG[-n:])\n",
    "    from IPython.display import display\n",
    "    display(df)\n",
    "\n",
    "def run_agent(user_message, tools, system_prompt=None, max_steps=10):\n",
    "    \"\"\"A manual agent loop with full visibility.\n",
    "\n",
    "    Args:\n",
    "        user_message: The user's request.\n",
    "        tools: List of Python functions to use as tools.\n",
    "        system_prompt: Optional system instruction for the agent.\n",
    "        max_steps: Maximum number of reasoning steps (safety limit).\n",
    "\n",
    "    Returns:\n",
    "        A tuple of (final_text, tools_called, trace) where:\n",
    "        - final_text: The agent's final response string\n",
    "        - tools_called: List of tool names invoked during the run\n",
    "        - trace: List of dicts with 'call', 'tool', 'args', 'result' per tool call\n",
    "    \"\"\"\n",
    "    tool_map = {fn.__name__: fn for fn in tools}\n",
    "    call_count = 0        # Track total tool calls across all steps\n",
    "    tools_called = []     # Record which tools were actually used\n",
    "    trace = []            # Structured log: tool, args, result per call\n",
    "\n",
    "    # Build initial contents\n",
    "    contents = []\n",
    "    if system_prompt:\n",
    "        contents.append(types.Content(\n",
    "            role=\"user\",\n",
    "            parts=[types.Part(text=f\"System: {system_prompt}\\n\\nUser: {user_message}\")]\n",
    "        ))\n",
    "    else:\n",
    "        contents.append(types.Content(\n",
    "            role=\"user\",\n",
    "            parts=[types.Part(text=user_message)]\n",
    "        ))\n",
    "\n",
    "    log_interaction(\"user\", user_message, label=\"agent_input\")\n",
    "\n",
    "    for step in range(max_steps):\n",
    "        response = client.models.generate_content(\n",
    "            model=MODEL_ID,\n",
    "            contents=contents,\n",
    "            config=types.GenerateContentConfig(\n",
    "                tools=tools,\n",
    "                # Tool calling mode defaults to AUTO \u2014 the model\n",
    "                # reasons about whether to use tools on each turn.\n",
    "                automatic_function_calling=types.AutomaticFunctionCallingConfig(\n",
    "                    disable=True  # Model still reasons about tools \u2014\n",
    "                    # but the SDK won't execute them automatically.\n",
    "                    # Instead it returns the function_call to us,\n",
    "                    # and WE run the function below.\n",
    "                ),\n",
    "            ),\n",
    "        )\n",
    "\n",
    "        # Guard: the model may return an empty response\n",
    "        parts = response.parts or []\n",
    "        if not parts:\n",
    "            print(f\"  Step {step+1}: \u26a0\ufe0f Empty response from model \u2014 retrying...\")\n",
    "            continue\n",
    "\n",
    "        # Add model response to history\n",
    "        contents.append(types.Content(role=\"model\", parts=parts))\n",
    "\n",
    "        # Check for function calls\n",
    "        function_results = []\n",
    "        for part in parts:\n",
    "            if part.function_call:\n",
    "                call_count += 1\n",
    "                name = part.function_call.name\n",
    "                args = dict(part.function_call.args)\n",
    "                tools_called.append(name)\n",
    "                print(f\"  Tool call {call_count}: \ud83d\udd27 {name}({args})\")\n",
    "\n",
    "                # Execute the function\n",
    "                try:\n",
    "                    result = tool_map[name](**args)\n",
    "                except Exception as e:\n",
    "                    result = {\"error\": str(e)}\n",
    "\n",
    "                print(f\"           \u2192 {result}\")\n",
    "                trace.append({\n",
    "                    \"call\": call_count,\n",
    "                    \"tool\": name,\n",
    "                    \"args\": args,\n",
    "                    \"result\": result if isinstance(result, str) else json.dumps(result),\n",
    "                })\n",
    "                log_interaction(\"tool\", f\"{name}({args}) \u2192 {result}\", label=\"tool_call\")\n",
    "\n",
    "                function_results.append(\n",
    "                    types.Part(\n",
    "                        function_response=types.FunctionResponse(\n",
    "                            name=name,\n",
    "                            response={\"result\": result},\n",
    "                        )\n",
    "                    )\n",
    "                )\n",
    "\n",
    "        if function_results:\n",
    "            contents.append(types.Content(role=\"user\", parts=function_results))\n",
    "        else:\n",
    "            # No function calls \u2192 model is done\n",
    "            final_text = response.text or \"(no text response)\"\n",
    "            log_interaction(\"agent\", final_text, label=\"agent_output\")\n",
    "            return final_text, tools_called, trace\n",
    "\n",
    "    return \"\u26a0\ufe0f Agent reached maximum steps without completing.\", tools_called, trace\n",
    "\n",
    "print(\"\u2705 Infrastructure ready.\")"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 1: Choose Your Track (5 min)\n",
    "\n",
    "> **Important:** The track you choose here will also be used for the **Day 4 Assignment**. Pick a domain that interests you!\n",
    "\n",
    "| Track | Domain | Tools You'll Build |\n",
    "|-------|--------|-------------------|\n",
    "| **A** | Customer Support | classify_ticket, search_policies, draft_response |\n",
    "| **B** | Research Assistant | search_documents, summarize_text, compare_topics |\n",
    "| **C** | Financial Analysis | get_financials, calculate_metric, search_reports |\n",
    "| **D** | HR / Recruitment | search_candidates, get_job_requirements, score_candidate |\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2554\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2557\n",
    "# \u2551  SELECT YOUR TRACK \u2014 change the letter below            \u2551\n",
    "# \u255a\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u2550\u255d\n",
    "SELECTED_TRACK = \"A\"   # Change to \"A\", \"B\", \"C\", or \"D\"\n",
    "\n",
    "print(f\"\u2705 Selected Track: {SELECTED_TRACK}\")\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Track Data (mock databases) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "# Each track has sample data that the tools will query.\n",
    "\n",
    "# Track A: Customer Support\n",
    "SUPPORT_TICKETS = [\n",
    "    {\"id\": \"TK-101\", \"text\": \"App crashes when uploading large photos. Tried reinstalling.\", \"customer\": \"user_42\"},\n",
    "    {\"id\": \"TK-102\", \"text\": \"I was charged twice for my subscription this month. Urgent!\", \"customer\": \"user_88\"},\n",
    "    {\"id\": \"TK-103\", \"text\": \"How do I export my data to CSV? Can't find the option.\", \"customer\": \"user_15\"},\n",
    "    {\"id\": \"TK-104\", \"text\": \"The dark mode doesn't work on iOS 18. Everything is white.\", \"customer\": \"user_67\"},\n",
    "    {\"id\": \"TK-105\", \"text\": \"I love the new dashboard update! Great work.\", \"customer\": \"user_23\"},\n",
    "    {\"id\": \"TK-106\", \"text\": \"Login fails with SSO. Error code 403. Very urgent.\", \"customer\": \"user_91\"},\n",
    "    {\"id\": \"TK-107\", \"text\": \"Can I upgrade from Starter to Growth plan mid-cycle?\", \"customer\": \"user_34\"},\n",
    "    {\"id\": \"TK-108\", \"text\": \"API rate limit hit. Need higher quota for production.\", \"customer\": \"user_56\"},\n",
    "]\n",
    "\n",
    "SUPPORT_POLICIES = {\n",
    "    \"billing\": \"Duplicate charges must be refunded within 48 hours. Escalate to billing team if amount > $500.\",\n",
    "    \"bugs\": \"Critical bugs (crash, data loss) are Priority 1. Assign to engineering. ETA: 24h for P1, 72h for P2.\",\n",
    "    \"features\": \"Feature requests go to the product backlog. Thank the customer and share the roadmap link.\",\n",
    "    \"account\": \"Account changes (upgrades, downgrades) can be processed immediately. Prorate the difference.\",\n",
    "    \"api\": \"API rate limit increases require manager approval. Standard limit: 1000 req/min. Enterprise: 10000 req/min.\",\n",
    "    \"praise\": \"Positive feedback should be forwarded to the team Slack channel. Thank the customer.\",\n",
    "}\n",
    "\n",
    "# Track B: Research Assistant\n",
    "RESEARCH_DOCS = {\n",
    "    \"ai_market\": \"The global AI market was valued at $196B in 2023 and is projected to reach $1.8T by 2030, growing at 36% CAGR. Key segments: generative AI ($44B), computer vision ($38B), NLP ($35B).\",\n",
    "    \"competitor_alpha\": \"AlphaTech launched their enterprise AI platform in Q2 2024. Pricing: $50k/year for teams up to 50 users. Key differentiator: on-premise deployment option. Weakness: no mobile SDK.\",\n",
    "    \"competitor_beta\": \"BetaCorp acquired DataMinds for $2.3B in March 2024. Combined entity focuses on real-time analytics. Revenue grew 45% YoY to $890M. Weakness: high customer churn (18%).\",\n",
    "    \"customer_trends\": \"Enterprise AI adoption increased from 35% to 55% between 2022-2024. Top use cases: customer service automation (72%), document processing (65%), predictive analytics (58%).\",\n",
    "    \"regulation\": \"The EU AI Act entered into force in August 2024. Key requirements: transparency obligations for general-purpose AI, risk classification system, and mandatory conformity assessments for high-risk applications.\",\n",
    "    \"talent\": \"AI engineer salaries increased 25% in 2024. Average: $185k in the US, $120k in Europe. Biggest skill gaps: MLOps (67% of companies), responsible AI (54%), agent frameworks (48%).\",\n",
    "}\n",
    "\n",
    "# Track C: Financial Analysis\n",
    "FINANCIAL_DATA = {\n",
    "    \"ACME\": {\"revenue_q4\": 45_000_000, \"expenses_q4\": 38_000_000, \"employees\": 450, \"growth_yoy\": 0.12, \"sector\": \"Manufacturing\"},\n",
    "    \"TECHSTART\": {\"revenue_q4\": 12_000_000, \"expenses_q4\": 15_000_000, \"employees\": 120, \"growth_yoy\": 0.45, \"sector\": \"SaaS\"},\n",
    "    \"RETAILMAX\": {\"revenue_q4\": 89_000_000, \"expenses_q4\": 82_000_000, \"employees\": 2200, \"growth_yoy\": -0.03, \"sector\": \"Retail\"},\n",
    "}\n",
    "\n",
    "EARNINGS_REPORTS = {\n",
    "    \"ACME_Q4\": \"Acme Corp reported Q4 revenue of $45M, up 12% YoY. Margins improved to 15.6% due to automation initiatives. Guidance for next quarter: $47-49M revenue.\",\n",
    "    \"TECHSTART_Q4\": \"TechStart burned $3M in Q4 but grew revenue 45% YoY to $12M. ARR reached $48M. Key risk: runway is 14 months at current burn rate. Pursuing Series C.\",\n",
    "    \"RETAILMAX_Q4\": \"RetailMax Q4 revenue was $89M, down 3% YoY. E-commerce grew 15% but couldn't offset 8% decline in physical stores. Announced 200 layoffs.\",\n",
    "    \"INDUSTRY_OUTLOOK\": \"The SaaS sector is expected to grow 18% in 2025, driven by AI integration. Manufacturing AI spending projected at $9.8B. Retail tech investment flat YoY.\",\n",
    "}\n",
    "\n",
    "# Track D: HR / Recruitment\n",
    "CANDIDATES = [\n",
    "    {\"id\": \"C-201\", \"name\": \"Alice Chen\", \"skills\": [\"Python\", \"ML\", \"TensorFlow\"], \"experience_years\": 5, \"current_role\": \"ML Engineer\", \"salary_expectation\": 160000},\n",
    "    {\"id\": \"C-202\", \"name\": \"Bob Martinez\", \"skills\": [\"Java\", \"AWS\", \"Kubernetes\"], \"experience_years\": 8, \"current_role\": \"DevOps Lead\", \"salary_expectation\": 185000},\n",
    "    {\"id\": \"C-203\", \"name\": \"Carol Zhang\", \"skills\": [\"Python\", \"NLP\", \"LLMs\", \"RAG\"], \"experience_years\": 3, \"current_role\": \"AI Research Intern\", \"salary_expectation\": 130000},\n",
    "    {\"id\": \"C-204\", \"name\": \"David Kim\", \"skills\": [\"Product Management\", \"Agile\", \"SQL\"], \"experience_years\": 10, \"current_role\": \"Senior PM\", \"salary_expectation\": 175000},\n",
    "    {\"id\": \"C-205\", \"name\": \"Eva M\u00fcller\", \"skills\": [\"Python\", \"Data Engineering\", \"Spark\"], \"experience_years\": 6, \"current_role\": \"Data Engineer\", \"salary_expectation\": 155000},\n",
    "    {\"id\": \"C-206\", \"name\": \"Frank Lee\", \"skills\": [\"Python\", \"ML\", \"LLMs\", \"Agents\"], \"experience_years\": 4, \"current_role\": \"AI Engineer\", \"salary_expectation\": 170000},\n",
    "]\n",
    "\n",
    "JOB_REQUIREMENTS = {\n",
    "    \"AI_ENGINEER\": {\"title\": \"AI Engineer\", \"required_skills\": [\"Python\", \"ML\", \"LLMs\"], \"min_experience\": 3, \"max_salary\": 180000, \"team\": \"AI Platform\"},\n",
    "    \"DATA_ENGINEER\": {\"title\": \"Data Engineer\", \"required_skills\": [\"Python\", \"Data Engineering\", \"SQL\"], \"min_experience\": 4, \"max_salary\": 165000, \"team\": \"Data\"},\n",
    "    \"SENIOR_PM\": {\"title\": \"Senior Product Manager\", \"required_skills\": [\"Product Management\", \"Agile\"], \"min_experience\": 7, \"max_salary\": 190000, \"team\": \"Product\"},\n",
    "}\n",
    "\n",
    "print(f\"\u2705 Track data loaded. Selected track: {SELECTED_TRACK}\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 2: Define Your Tools (15 min)\n",
    "\n",
    "Design 3 tools for your chosen track. Each tool needs:\n",
    "- A clear **function name** (verb + noun, e.g. `classify_ticket`)\n",
    "- **Type hints** for all parameters and return type\n",
    "- A **docstring** explaining what it does and when to use it\n",
    "- **Graceful error handling** (return `{\"error\": \"...\"}` instead of crashing)\n",
    "\n",
    "> \ud83d\udca1 **Tip:** The quality of your tool descriptions directly affects how well the agent uses them. Treat descriptions like prompts!\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Track A: Customer Support Tools \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "\n",
    "def classify_ticket(ticket_text: str) -> dict:\n",
    "    \"\"\"Classify a support ticket into a category and urgency level.\n",
    "\n",
    "    Args:\n",
    "        ticket_text: The customer's support message.\n",
    "\n",
    "    Returns:\n",
    "        A dict with 'category' and 'urgency' (low/medium/high).\n",
    "    \"\"\"\n",
    "    text_lower = ticket_text.lower()\n",
    "    # Simple rule-based classification (students can improve with LLM later)\n",
    "    if any(w in text_lower for w in [\"charged\", \"billing\", \"invoice\", \"refund\", \"payment\"]):\n",
    "        category = \"billing\"\n",
    "    elif any(w in text_lower for w in [\"crash\", \"error\", \"bug\", \"broken\", \"fail\"]):\n",
    "        category = \"bugs\"\n",
    "    elif any(w in text_lower for w in [\"upgrade\", \"downgrade\", \"plan\", \"account\"]):\n",
    "        category = \"account\"\n",
    "    elif any(w in text_lower for w in [\"api\", \"rate limit\", \"quota\"]):\n",
    "        category = \"api\"\n",
    "    elif any(w in text_lower for w in [\"love\", \"great\", \"awesome\", \"thanks\"]):\n",
    "        category = \"praise\"\n",
    "    else:\n",
    "        category = \"features\"\n",
    "\n",
    "    urgency = \"high\" if any(w in text_lower for w in [\"urgent\", \"crash\", \"charged twice\", \"fail\"]) else \"medium\"\n",
    "    return {\"category\": category, \"urgency\": urgency}\n",
    "\n",
    "\n",
    "def search_policies(category: str) -> str:\n",
    "    \"\"\"Search company support policies for a given category.\n",
    "\n",
    "    Args:\n",
    "        category: The ticket category, e.g. 'billing', 'bugs', 'account'.\n",
    "\n",
    "    Returns:\n",
    "        The relevant policy text, or an error if category not found.\n",
    "    \"\"\"\n",
    "    policy = SUPPORT_POLICIES.get(category)\n",
    "    if policy:\n",
    "        return policy\n",
    "    return f\"No policy found for category '{category}'. Available: {list(SUPPORT_POLICIES.keys())}\"\n",
    "\n",
    "\n",
    "def draft_response(ticket_text: str, category: str, policy_excerpt: str) -> str:\n",
    "    \"\"\"Draft a customer support response based on the ticket and policy.\n",
    "\n",
    "    Args:\n",
    "        ticket_text: The original customer message.\n",
    "        category: The classified category.\n",
    "        policy_excerpt: The relevant policy text.\n",
    "\n",
    "    Returns:\n",
    "        A draft response string.\n",
    "    \"\"\"\n",
    "    return (\n",
    "        f\"Thank you for reaching out. I've categorized your request as '{category}'. \"\n",
    "        f\"Based on our policy: {policy_excerpt} \"\n",
    "        f\"I'll make sure this is handled promptly. Is there anything else I can help with?\"\n",
    "    )\n",
    "\n",
    "\n",
    "# \u2500\u2500 Track B: Research Assistant Tools \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "\n",
    "def search_documents(query: str) -> str:\n",
    "    \"\"\"Search the internal document library for information.\n",
    "\n",
    "    Args:\n",
    "        query: The search query, e.g. 'AI market size 2024'\n",
    "\n",
    "    Returns:\n",
    "        Relevant document excerpts matching the query.\n",
    "    \"\"\"\n",
    "    # TODO: Implement search logic using RESEARCH_DOCS\n",
    "    # Hint: check if any keyword from query appears in doc keys or values\n",
    "    results = []\n",
    "    query_lower = query.lower()\n",
    "    for key, text in RESEARCH_DOCS.items():\n",
    "        if any(word in text.lower() for word in query_lower.split()):\n",
    "            results.append(f\"[{key}]: {text[:200]}\")\n",
    "    if results:\n",
    "        return \"\\n\\n\".join(results[:3])\n",
    "    return f\"No documents found for '{query}'.\"\n",
    "\n",
    "\n",
    "def summarize_text(text: str, max_sentences: int = 3) -> str:\n",
    "    \"\"\"Summarize a given text into key points.\n",
    "\n",
    "    Args:\n",
    "        text: The text to summarize.\n",
    "        max_sentences: Maximum number of sentences in the summary.\n",
    "\n",
    "    Returns:\n",
    "        A concise summary.\n",
    "    \"\"\"\n",
    "    # TODO: Implement summarization\n",
    "    # Simple approach: return first N sentences\n",
    "    sentences = text.replace(\". \", \".\\n\").split(\"\\n\")\n",
    "    summary = \". \".join(s.strip() for s in sentences[:max_sentences] if s.strip())\n",
    "    return summary if summary else \"Could not summarize the text.\"\n",
    "\n",
    "\n",
    "def compare_topics(topic1: str, topic2: str) -> str:\n",
    "    \"\"\"Compare two topics or companies based on available documents.\n",
    "\n",
    "    Args:\n",
    "        topic1: First topic or company name.\n",
    "        topic2: Second topic or company name.\n",
    "\n",
    "    Returns:\n",
    "        A structured comparison of the two topics.\n",
    "    \"\"\"\n",
    "    # TODO: Implement comparison logic\n",
    "    info1 = search_documents(topic1)\n",
    "    info2 = search_documents(topic2)\n",
    "    return f\"--- {topic1} ---\\n{info1}\\n\\n--- {topic2} ---\\n{info2}\"\n",
    "\n",
    "\n",
    "# \u2500\u2500 Track C: Financial Analysis Tools \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "\n",
    "def get_financials(company: str) -> dict:\n",
    "    \"\"\"Get financial data for a company.\n",
    "\n",
    "    Args:\n",
    "        company: Company ticker/name, e.g. 'ACME', 'TECHSTART'\n",
    "\n",
    "    Returns:\n",
    "        A dict with revenue, expenses, employees, growth, sector.\n",
    "    \"\"\"\n",
    "    # TODO: Implement using FINANCIAL_DATA\n",
    "    data = FINANCIAL_DATA.get(company.upper())\n",
    "    if data:\n",
    "        return {**data, \"company\": company.upper(), \"status\": \"found\"}\n",
    "    return {\"error\": f\"No data for '{company}'. Available: {list(FINANCIAL_DATA.keys())}\"}\n",
    "\n",
    "\n",
    "def calculate_metric(expression: str) -> str:\n",
    "    \"\"\"Evaluate a financial calculation.\n",
    "\n",
    "    Args:\n",
    "        expression: A math expression, e.g. '45000000 - 38000000' or '12000000 / 120'\n",
    "\n",
    "    Returns:\n",
    "        The result as a string.\n",
    "    \"\"\"\n",
    "    # TODO: Implement\n",
    "    try:\n",
    "        return str(eval(expression))\n",
    "    except Exception as e:\n",
    "        return f\"Error: {e}\"\n",
    "\n",
    "\n",
    "def search_reports(query: str) -> str:\n",
    "    \"\"\"Search earnings reports and industry analyses.\n",
    "\n",
    "    Args:\n",
    "        query: Search terms, e.g. 'ACME Q4 earnings'\n",
    "\n",
    "    Returns:\n",
    "        Relevant report excerpts.\n",
    "    \"\"\"\n",
    "    # TODO: Implement using EARNINGS_REPORTS\n",
    "    results = []\n",
    "    query_lower = query.lower()\n",
    "    for key, text in EARNINGS_REPORTS.items():\n",
    "        if any(word in key.lower() or word in text.lower() for word in query_lower.split()):\n",
    "            results.append(f\"[{key}]: {text}\")\n",
    "    if results:\n",
    "        return \"\\n\\n\".join(results[:2])\n",
    "    return f\"No reports found for '{query}'.\"\n",
    "\n",
    "\n",
    "# \u2500\u2500 Track D: HR / Recruitment Tools \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "\n",
    "def search_candidates(required_skills: str, min_experience: int = 0) -> str:\n",
    "    \"\"\"Search the candidate database by skills and experience.\n",
    "\n",
    "    Args:\n",
    "        required_skills: Comma-separated skills, e.g. 'Python, ML'\n",
    "        min_experience: Minimum years of experience.\n",
    "\n",
    "    Returns:\n",
    "        Matching candidates with their details.\n",
    "    \"\"\"\n",
    "    # TODO: Implement using CANDIDATES\n",
    "    skills = [s.strip().lower() for s in required_skills.split(\",\")]\n",
    "    matches = []\n",
    "    for c in CANDIDATES:\n",
    "        c_skills = [s.lower() for s in c[\"skills\"]]\n",
    "        skill_match = sum(1 for s in skills if s in c_skills)\n",
    "        if skill_match > 0 and c[\"experience_years\"] >= min_experience:\n",
    "            matches.append(f\"{c['name']} ({c['current_role']}, {c['experience_years']}y) \u2014 Skills: {', '.join(c['skills'])}\")\n",
    "    if matches:\n",
    "        return \"\\n\".join(matches)\n",
    "    return \"No candidates match the criteria.\"\n",
    "\n",
    "\n",
    "def get_job_requirements(position: str) -> dict:\n",
    "    \"\"\"Get the requirements for a job position.\n",
    "\n",
    "    Args:\n",
    "        position: Job title or ID, e.g. 'AI_ENGINEER'\n",
    "\n",
    "    Returns:\n",
    "        A dict with title, required_skills, min_experience, max_salary.\n",
    "    \"\"\"\n",
    "    # TODO: Implement using JOB_REQUIREMENTS\n",
    "    data = JOB_REQUIREMENTS.get(position.upper().replace(\" \", \"_\"))\n",
    "    if data:\n",
    "        return data\n",
    "    return {\"error\": f\"Position '{position}' not found. Available: {list(JOB_REQUIREMENTS.keys())}\"}\n",
    "\n",
    "\n",
    "def score_candidate(candidate_id: str, position: str) -> dict:\n",
    "    \"\"\"Score a candidate against a job position's requirements.\n",
    "\n",
    "    Args:\n",
    "        candidate_id: The candidate ID, e.g. 'C-201'\n",
    "        position: The job position ID, e.g. 'AI_ENGINEER'\n",
    "\n",
    "    Returns:\n",
    "        A dict with skill_match, experience_match, salary_fit, overall_score.\n",
    "    \"\"\"\n",
    "    # TODO: Implement scoring\n",
    "    candidate = next((c for c in CANDIDATES if c[\"id\"] == candidate_id), None)\n",
    "    if not candidate:\n",
    "        return {\"error\": f\"Candidate '{candidate_id}' not found.\"}\n",
    "    job = JOB_REQUIREMENTS.get(position.upper().replace(\" \", \"_\"))\n",
    "    if not job:\n",
    "        return {\"error\": f\"Position '{position}' not found.\"}\n",
    "\n",
    "    c_skills = [s.lower() for s in candidate[\"skills\"]]\n",
    "    required = [s.lower() for s in job[\"required_skills\"]]\n",
    "    skill_match = sum(1 for s in required if s in c_skills) / len(required)\n",
    "    exp_match = 1.0 if candidate[\"experience_years\"] >= job[\"min_experience\"] else candidate[\"experience_years\"] / job[\"min_experience\"]\n",
    "    salary_fit = 1.0 if candidate[\"salary_expectation\"] <= job[\"max_salary\"] else job[\"max_salary\"] / candidate[\"salary_expectation\"]\n",
    "    overall = round((skill_match * 0.4 + exp_match * 0.3 + salary_fit * 0.3) * 100)\n",
    "\n",
    "    return {\n",
    "        \"candidate\": candidate[\"name\"],\n",
    "        \"position\": job[\"title\"],\n",
    "        \"skill_match\": f\"{skill_match:.0%}\",\n",
    "        \"experience_match\": f\"{exp_match:.0%}\",\n",
    "        \"salary_fit\": f\"{salary_fit:.0%}\",\n",
    "        \"overall_score\": f\"{overall}%\",\n",
    "    }\n",
    "\n",
    "\n",
    "# \u2500\u2500 Select Tools Based on Track \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "TRACK_TOOLS = {\n",
    "    \"A\": [classify_ticket, search_policies, draft_response],\n",
    "    \"B\": [search_documents, summarize_text, compare_topics],\n",
    "    \"C\": [get_financials, calculate_metric, search_reports],\n",
    "    \"D\": [search_candidates, get_job_requirements, score_candidate],\n",
    "}\n",
    "\n",
    "my_tools = TRACK_TOOLS[SELECTED_TRACK]\n",
    "print(f\"\u2705 Tools for Track {SELECTED_TRACK}: {[t.__name__ for t in my_tools]}\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 3: Write Your System Prompt (10 min)\n",
    "\n",
    "A good agent system prompt includes:\n",
    "- **Role** assignment\n",
    "- **Available tools** with descriptions\n",
    "- **Rules** (when to use tools, when not to)\n",
    "- **Refusal** instructions (what to do with out-of-scope requests)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 System Prompt V1 \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "# TODO: Customize this for your track.\n",
    "\n",
    "SYSTEM_PROMPT_V1 = \"\"\"You are a helpful assistant for the {domain} domain.\n",
    "\n",
    "Available tools:\n",
    "{tool_descriptions}\n",
    "\n",
    "Rules:\n",
    "- Use tools when you need to look up or process information.\n",
    "- Never invent data \u2014 only use results returned by tools.\n",
    "- If you cannot help with a request, say so clearly.\n",
    "- Be concise and professional in your responses.\n",
    "\"\"\".format(\n",
    "    domain={\n",
    "        \"A\": \"customer support\",\n",
    "        \"B\": \"business research\",\n",
    "        \"C\": \"financial analysis\",\n",
    "        \"D\": \"HR and recruitment\",\n",
    "    }[SELECTED_TRACK],\n",
    "    tool_descriptions=\"\\n\".join(\n",
    "        f\"- {fn.__name__}: {fn.__doc__.split(chr(10))[0]}\" for fn in my_tools\n",
    "    ),\n",
    ")\n",
    "\n",
    "print(\"System Prompt V1:\")\n",
    "print(SYSTEM_PROMPT_V1)\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 4: Test Your Agent on 5+ Queries (15 min)\n",
    "\n",
    "Run your agent on at least 5 realistic queries. Include a mix:\n",
    "- 2 straightforward queries (easy)\n",
    "- 2 multi-step queries (medium)\n",
    "- 1 edge case or out-of-scope query (hard)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Test Queries \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "# TODO: Add at least 4 more queries for your track.\n",
    "\n",
    "TRACK_QUERIES = {\n",
    "    \"A\": [\n",
    "        \"Process ticket TK-102: 'I was charged twice for my subscription this month. Urgent!'\",\n",
    "        \"Handle ticket TK-103: 'How do I export my data to CSV?'\",\n",
    "        \"Triage ticket TK-106: 'Login fails with SSO. Error code 403. Very urgent.'\",\n",
    "        \"Process ticket TK-105: 'I love the new dashboard update!'\",\n",
    "        \"What's the weather in London?\",  # Out-of-scope\n",
    "    ],\n",
    "    \"B\": [\n",
    "        \"What is the current size of the global AI market?\",\n",
    "        \"Compare AlphaTech and BetaCorp as competitors.\",\n",
    "        \"Summarize the key trends in enterprise AI adoption.\",\n",
    "        \"What are the main regulations affecting AI in Europe?\",\n",
    "        \"How do I reset my password?\",  # Out-of-scope\n",
    "    ],\n",
    "    \"C\": [\n",
    "        \"What was Acme Corp's Q4 profit?\",\n",
    "        \"Compare the growth rates of TECHSTART and RETAILMAX.\",\n",
    "        \"Calculate revenue per employee for all three companies.\",\n",
    "        \"What is the industry outlook for SaaS in 2025?\",\n",
    "        \"Who won the World Cup?\",  # Out-of-scope\n",
    "    ],\n",
    "    \"D\": [\n",
    "        \"Find candidates for the AI Engineer position.\",\n",
    "        \"Score candidate C-203 for the AI Engineer role.\",\n",
    "        \"Compare candidates C-201 and C-206 for the AI Engineer role.\",\n",
    "        \"What are the requirements for the Senior PM position?\",\n",
    "        \"Can you book a meeting room?\",  # Out-of-scope\n",
    "    ],\n",
    "}\n",
    "\n",
    "test_queries = TRACK_QUERIES[SELECTED_TRACK]\n",
    "\n",
    "# Run V1\n",
    "print(\"=\" * 60)\n",
    "print(\"RUNNING AGENT V1\")\n",
    "print(\"=\" * 60)\n",
    "\n",
    "v1_results = []\n",
    "for i, query in enumerate(test_queries, 1):\n",
    "    print(f\"\\n{\"\u2500\"*60}\")\n",
    "    print(f\"Query {i}: {query}\\n\")\n",
    "    answer, tools_used, trace = run_agent(query, tools=my_tools, system_prompt=SYSTEM_PROMPT_V1)\n",
    "    v1_results.append({\"query\": query, \"answer\": answer, \"tools_used\": tools_used, \"trace\": trace})\n",
    "    print(f\"\ud83d\udd27 Tools used: {tools_used}\")\n",
    "    if trace:\n",
    "        print(f\"\\n\ud83d\udccb Agent Trace:\")\n",
    "        for t in trace:\n",
    "            print(f\"   [{t['call']}] {t['tool']}({t['args']}) \u2192 {str(t['result'])[:200]}\")\n",
    "    print(f\"\\n\ud83d\udcdd Answer: {answer[:300]}\")"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 5: Create Golden Test Set (15 min)\n",
    "\n",
    "Create at least 8 test scenarios. For each, specify:\n",
    "- The input query\n",
    "- Which tools should be called (in order)\n",
    "- Keywords the answer should contain\n",
    "- Difficulty level (easy / medium / hard / edge)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Golden Test Set \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "# TODO: Complete with 8+ scenarios for your track.\n",
    "\n",
    "golden_set = [\n",
    "    {\n",
    "        \"id\": \"G01\",\n",
    "        \"query\": test_queries[0],\n",
    "        \"expected_tools\": [my_tools[0].__name__, my_tools[1].__name__],\n",
    "        \"expected_keywords\": [\"billing\", \"refund\"],  # TODO: adjust for your track\n",
    "        \"difficulty\": \"easy\",\n",
    "    },\n",
    "    {\n",
    "        \"id\": \"G02\",\n",
    "        \"query\": test_queries[1],\n",
    "        \"expected_tools\": [my_tools[0].__name__],\n",
    "        \"expected_keywords\": [],  # TODO: fill in\n",
    "        \"difficulty\": \"easy\",\n",
    "    },\n",
    "    # TODO: Add G03\u2013G08+ with a mix of difficulty levels\n",
    "    # Include at least:\n",
    "    # - 3 easy scenarios\n",
    "    # - 3 medium scenarios (multi-tool)\n",
    "    # - 2 hard/edge scenarios (out-of-scope, ambiguous)\n",
    "]\n",
    "\n",
    "print(f\"Golden set: {len(golden_set)} scenarios defined.\")\n",
    "assert len(golden_set) >= 2, \"\u26a0\ufe0f Need at least 8 scenarios! Keep adding.\"\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Evaluate V1 Against Golden Set \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "def evaluate_agent_results(results, golden_set):\n",
    "    \"\"\"Evaluate agent results against golden test set.\"\"\"\n",
    "    scores = []\n",
    "    for golden in golden_set:\n",
    "        # Find matching result\n",
    "        matching = [r for r in results if r[\"query\"] == golden[\"query\"]]\n",
    "        if not matching:\n",
    "            scores.append({\"id\": golden[\"id\"], \"keyword_score\": 0, \"passed\": False})\n",
    "            continue\n",
    "        answer = matching[0][\"answer\"].lower()\n",
    "        expected_kw = golden.get(\"expected_keywords\", [])\n",
    "        if expected_kw:\n",
    "            found = sum(1 for kw in expected_kw if kw.lower() in answer)\n",
    "            score = found / len(expected_kw)\n",
    "        else:\n",
    "            score = 1.0  # No keywords to check\n",
    "        scores.append({\n",
    "            \"id\": golden[\"id\"],\n",
    "            \"difficulty\": golden.get(\"difficulty\", \"?\"),\n",
    "            \"keyword_score\": f\"{score:.0%}\",\n",
    "            \"passed\": score >= 0.5,\n",
    "        })\n",
    "    return scores\n",
    "\n",
    "v1_scores = evaluate_agent_results(v1_results, golden_set)\n",
    "import pandas as pd\n",
    "print(\"V1 Evaluation:\")\n",
    "print(pd.DataFrame(v1_scores).to_string(index=False))\n",
    "v1_pass_rate = sum(1 for s in v1_scores if s[\"passed\"]) / len(v1_scores) * 100\n",
    "print(f\"\\nV1 Pass Rate: {v1_pass_rate:.0f}%\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 6: Analyze V1 Errors (10 min)\n",
    "\n",
    "Look at the queries where v1 failed or performed poorly. Ask yourself:\n",
    "1. Did the agent call the **wrong tool**?\n",
    "2. Did it pass **incorrect arguments**?\n",
    "3. Did it **ignore the tool result**?\n",
    "4. Was the **system prompt** unclear?\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "### \u270d\ufe0f V1 Error Analysis\n",
    "\n",
    "**Error Pattern 1:** [Describe]\n",
    "- Example query: ...\n",
    "- What went wrong: ...\n",
    "- Root cause: ...\n",
    "\n",
    "**Error Pattern 2:** [Describe]\n",
    "- Example query: ...\n",
    "- What went wrong: ...\n",
    "- Root cause: ...\n",
    "\n",
    "**Planned improvements for V2:**\n",
    "- ...\n",
    "- ...\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 7: Improve to V2 (20 min)\n",
    "\n",
    "Based on your error analysis, improve your:\n",
    "- **System prompt** (add rules, examples, constraints)\n",
    "- **Tool descriptions** (make them clearer)\n",
    "- **Tool implementations** (handle edge cases)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 System Prompt V2 (Improved) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "# TODO: Improve based on V1 error analysis.\n",
    "\n",
    "SYSTEM_PROMPT_V2 = \"\"\"You are a helpful assistant for the {domain} domain.\n",
    "\n",
    "Available tools:\n",
    "{tool_descriptions}\n",
    "\n",
    "Rules:\n",
    "- Use tools when you need to look up or process information.\n",
    "- Never invent data \u2014 only use results returned by tools.\n",
    "- If you cannot help with a request, say so clearly.\n",
    "- Be concise and professional in your responses.\n",
    "\n",
    "# TODO: Add additional rules based on your V1 errors, e.g.:\n",
    "# - Always classify before searching policies (Track A)\n",
    "# - When comparing, search for both items first (Track B)\n",
    "# - Show your calculations step by step (Track C)\n",
    "# - Check all matching candidates before scoring (Track D)\n",
    "\"\"\".format(\n",
    "    domain={\n",
    "        \"A\": \"customer support\",\n",
    "        \"B\": \"business research\",\n",
    "        \"C\": \"financial analysis\",\n",
    "        \"D\": \"HR and recruitment\",\n",
    "    }[SELECTED_TRACK],\n",
    "    tool_descriptions=\"\\n\".join(\n",
    "        f\"- {fn.__name__}: {fn.__doc__.split(chr(10))[0]}\" for fn in my_tools\n",
    "    ),\n",
    ")\n",
    "\n",
    "print(\"System Prompt V2:\")\n",
    "print(SYSTEM_PROMPT_V2)\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Run Agent V2 \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "print(\"=\" * 60)\n",
    "print(\"RUNNING AGENT V2\")\n",
    "print(\"=\" * 60)\n",
    "\n",
    "v2_results = []\n",
    "for i, query in enumerate(test_queries, 1):\n",
    "    print(f\"\\n{\"\u2500\"*60}\")\n",
    "    print(f\"Query {i}: {query}\\n\")\n",
    "    answer, tools_used, trace = run_agent(query, tools=my_tools, system_prompt=SYSTEM_PROMPT_V2)\n",
    "    v2_results.append({\"query\": query, \"answer\": answer, \"tools_used\": tools_used, \"trace\": trace})\n",
    "    print(f\"\ud83d\udd27 Tools used: {tools_used}\")\n",
    "    if trace:\n",
    "        print(f\"\\n\ud83d\udccb Agent Trace:\")\n",
    "        for t in trace:\n",
    "            print(f\"   [{t['call']}] {t['tool']}({t['args']}) \u2192 {str(t['result'])[:200]}\")\n",
    "    print(f\"\\n\ud83d\udcdd Answer: {answer[:300]}\")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Compare V1 vs V2 \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "v2_scores = evaluate_agent_results(v2_results, golden_set)\n",
    "v2_pass_rate = sum(1 for s in v2_scores if s[\"passed\"]) / len(v2_scores) * 100\n",
    "\n",
    "print(\"=\" * 60)\n",
    "print(\"COMPARISON: V1 vs V2\")\n",
    "print(\"=\" * 60)\n",
    "print(f\"V1 Pass Rate: {v1_pass_rate:.0f}%\")\n",
    "print(f\"V2 Pass Rate: {v2_pass_rate:.0f}%\")\n",
    "print(f\"Improvement:  {v2_pass_rate - v1_pass_rate:+.0f}%\")\n",
    "\n",
    "print(\"\\nV2 Detailed Scores:\")\n",
    "print(pd.DataFrame(v2_scores).to_string(index=False))\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## Step 8: Export Results\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Export Golden Set \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "with open(\"day4_lab2_golden_set.json\", \"w\") as f:\n",
    "    json.dump(golden_set, f, indent=2)\n",
    "print(f\"\u2705 Exported golden set ({len(golden_set)} items)\")\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Export Results \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "export_data = {\n",
    "    \"track\": SELECTED_TRACK,\n",
    "    \"v1_pass_rate\": v1_pass_rate,\n",
    "    \"v2_pass_rate\": v2_pass_rate,\n",
    "    \"v1_results\": [{\"query\": r[\"query\"], \"answer\": r[\"answer\"][:500]} for r in v1_results],\n",
    "    \"v2_results\": [{\"query\": r[\"query\"], \"answer\": r[\"answer\"][:500]} for r in v2_results],\n",
    "}\n",
    "with open(\"day4_lab2_results.json\", \"w\") as f:\n",
    "    json.dump(export_data, f, indent=2)\n",
    "print(\"\u2705 Exported results to day4_lab2_results.json\")\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "metadata": {},
   "outputs": [],
   "source": [
    "# \u2500\u2500 Export Prompt Log \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
    "if PROMPT_LOG:\n",
    "    import pandas as pd\n",
    "    log_df = pd.DataFrame(PROMPT_LOG)\n",
    "    log_df.to_csv(\"day4_lab2_prompt_log.csv\", index=False)\n",
    "    print(f\"\u2705 Exported {len(log_df)} log entries to day4_lab2_prompt_log.csv\")\n"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "---\n",
    "## \u2705 Checklist Before Assignment\n",
    "\n",
    "- [ ] Selected a track and defined 3 tools\n",
    "- [ ] Wrote system prompt with role, rules, and tool descriptions\n",
    "- [ ] Tested agent on 5+ queries\n",
    "- [ ] Created golden test set with 8+ scenarios\n",
    "- [ ] Evaluated V1 and wrote error analysis\n",
    "- [ ] Improved to V2 with better prompt/tools\n",
    "- [ ] Compared V1 vs V2 metrics\n",
    "- [ ] Exported all files (JSON, CSV)\n",
    "\n",
    "**Next:** The Day 4 Assignment builds on this lab. Use the same track, tools, and data \u2014 your goal is to refine, evaluate, and document your agent to production-ready quality.\n"
   ]
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "Python 3",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "name": "python",
   "version": "3.11.0"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 4
}