diff --git a/README.md b/README.md index 0b2b9c0..97f4062 100644 --- a/README.md +++ b/README.md @@ -8,33 +8,78 @@ AI-Supported Lightweight Code Editor built with Streamlit (AISE501 Spring 2026) AISE_AIAgent/ ├── frontend/ # Streamlit UI Components │ ├── __init__.py -│ ├── app.py # Main Streamlit application -│ ├── sidebar.py # File navigation sidebar -│ ├── editor.py # Code editor pane -│ └── chat.py # Chat interface +│ ├── app.py # Main Streamlit application entry point +│ ├── sidebar.py # File navigation sidebar component +│ ├── editor.py # Code editor pane component +│ └── chat.py # Chat interface component │ ├── backend/ # Backend Logic Modules │ ├── __init__.py -│ └── managers/ # Business logic managers +│ ├── managers/ # Business logic for UI operations +│ │ ├── __init__.py +│ │ ├── file_manager.py # File I/O operations for UI (read, write, list files) +│ │ ├── chat_manager.py # AI chat management and history +│ │ ├── system_prompter.py # System prompts and context injection +│ │ ├── search_manager.py # Internet search functionality +│ │ ├── execution_engine.py # Code execution and sandboxing +│ │ └── debug_logger.py # Logging, error handling, debug messages +│ │ +│ ├── agents/ # AI Agent System +│ │ ├── __init__.py +│ │ ├── coding_agent.py # Main agent loop (plan-act-observe cycle) +│ │ └── tools.py # Tools available to agent (7 functions + dispatcher) +│ │ +│ └── utils/ # Helper Utilities │ ├── __init__.py -│ ├── file_manager.py # File I/O operations -│ ├── chat_manager.py # AI chat management -│ ├── system_prompter.py # System prompts & context -│ ├── search_manager.py # Internet search -│ ├── execution_engine.py # Code execution -│ └── debug_logger.py # Logging & debugging +│ └── server_utils.py # LLM client init, chat functions, formatters │ -├── tests/ # Unit tests +├── tests/ # Unit Tests │ ├── __init__.py -│ ├── test_file_manager.py -│ ├── test_chat_manager.py -│ └── test_execution_engine.py +│ ├── test_file_manager.py # Tests for file operations +│ ├── test_chat_manager.py # Tests for chat functionality +│ ├── test_execution_engine.py # Tests for code execution +│ └── test_main.py # Integration tests │ -├── .gitignore # Git exclusions -├── requirements.txt # Python dependencies -└── README.md # This file +├── workspace/ # Agent Sandbox Directory +│ └── .gitkeep # Placeholder for agent to work safely in isolation +│ +├── .gitignore # Git exclusions (venv, .env, __pycache__, etc.) +├── .env # Local environment variables (NOT committed) +├── .env.example # Template for environment variables (IS committed) +├── requirements.txt # Python dependencies +├── README.md # This file +└── project_exercise.pdf # Project specification ``` +## Component Responsibilities + +### Frontend (`frontend/`) +- **app.py**: Main Streamlit application, layout orchestration +- **sidebar.py**: File browser and project navigation +- **editor.py**: Code editing interface with syntax highlighting +- **chat.py**: AI assistant chat interface + +### Backend Managers (`backend/managers/`) +Used directly by Frontend for UI operations: +- **file_manager.py**: CRUD operations on project files +- **chat_manager.py**: Chat history, message management +- **system_prompter.py**: System prompt generation and file context +- **execution_engine.py**: Safe code execution with output capture +- **debug_logger.py**: Error tracking and log formatting +- **search_manager.py**: Web search integration + +### Backend Agents (`backend/agents/`) +Independent AI agent system for complex tasks: +- **coding_agent.py**: Agent loop (Plan → Act → Observe → Repeat) +- **tools.py**: 7 tools agent can use (read/write/run/search/validate/grep/done) + +### Backend Utils (`backend/utils/`) +- **server_utils.py**: LLM client initialization, chat helpers, message formatters + +### Workspace (`workspace/`) +- Sandbox directory where agent executes and stores files +- Prevents agent from accessing files outside this directory + ## Features - **File Display & Management**: Browse and edit code files diff --git a/backend/__init__.py b/backend/__init__.py index e69de29..e44705a 100644 --- a/backend/__init__.py +++ b/backend/__init__.py @@ -0,0 +1,4 @@ +"""Backend - AI agents, managers, and utilities""" +from backend.managers import ChatManager + +__all__ = ["ChatManager"] diff --git a/backend/agent/__init__.py b/backend/agent/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/agent/coding_agent.py b/backend/agent/coding_agent.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/agent/coding_agent_example.py b/backend/agent/coding_agent_example.py new file mode 100644 index 0000000..8954d12 --- /dev/null +++ b/backend/agent/coding_agent_example.py @@ -0,0 +1,605 @@ +""" +Exercise 5b -- Build a Basic AI Coding Agent (Guided Version) +============================================================== +AISE501 . Prompting in Coding . Spring Semester 2026 + +This is a GUIDED version of Exercise 5 with more scaffolding. +It teaches the same concepts but reduces boilerplate so you can +focus on the key insight: how an LLM uses tools. + +The key insight +--------------- +An LLM cannot run code or read files by itself. But we can give it +"superpowers" through a simple trick: + + 1. TELL the LLM (via the system prompt) what tools exist. + 2. ASK the LLM to respond with JSON saying which tool to call. + 3. PARSE the JSON, call the real Python function, and + 4. FEED the result back into the conversation as a new message. + +This is how ALL AI coding agents work (Claude Code, Cursor, Copilot). +The LLM never actually "runs" code — it just asks us to run it! + +What is already provided +------------------------ +To let you focus on the interesting parts, the following are PRE-BUILT: + - All 7 tool functions (Part A) — read_file, grep_search, etc. + - The tool dispatcher (Part B) — maps tool names to functions. + - Helper functions: truncate_result, trim_messages, ask_human. + +What you need to build (the interesting parts) +----------------------------------------------- + Part C The SYSTEM PROMPT that teaches the LLM about its tools (TODOs 1-2). + Part D The AGENT LOOP that connects the LLM to the tools (TODOs 3-6). + Part E The INTERACTIVE CHAT interface (TODOs 7-8). + +Think of it like wiring a robot: + - Part A+B are the robot's HANDS (already built). + - Part C is the robot's INSTRUCTION MANUAL (you write it). + - Part D is the robot's BRAIN LOOP (you wire it). + - Part E is the ON SWITCH (you connect it). + +The conversation flow +--------------------- +Here is exactly what happens in one iteration of the agent loop: + + ┌─────────────────────────────────────────────────────────────┐ + │ messages = [ │ + │ {"role": "system", "content": ""}, │ + │ {"role": "user", "content": "Fix the bug in app.py"},│ + │ ] │ + └──────────────────────────┬──────────────────────────────────┘ + │ + ┌─────────▼─────────┐ + │ LLM generates │ + │ JSON response │ + └─────────┬─────────┘ + │ + ┌──────────────▼──────────────┐ + │ {"thought": "I should...", │ + │ "tool": "read_file", │ + │ "arguments": { │ + │ "path": "app.py" │ + │ }} │ + └──────────────┬──────────────┘ + │ + ┌─────────▼─────────┐ + │ You parse JSON, │ + │ call read_file() │ + └─────────┬─────────┘ + │ + ┌──────────────▼──────────────────┐ + │ Append to messages: │ + │ {"role":"assistant", "content":..}│ + │ {"role":"user", "content": │ + │ "file contents │ + │ "} │ + └──────────────┬──────────────────┘ + │ + ┌─────────▼─────────┐ + │ Next iteration: │ + │ LLM sees result, │ + │ picks next tool │ + └───────────────────┘ +""" + +import ast +import json +import subprocess +import sys +from pathlib import Path + +from server_utils import ( + chat, + chat_json, + get_client, + print_messages, + print_separator, + strip_code_fences, +) + +client = get_client() + +# ── Agent Configuration ────────────────────────────────────────────────────── +WORKSPACE = Path(__file__).parent / "workspace" +WORKSPACE.mkdir(exist_ok=True) + +MAX_ITERATIONS = 50 +MAX_RESULT_LENGTH = 8000 +MAX_HISTORY_CHARS = 60000 + + +# ═══════════════════════════════════════════════════════════════════════════════ +# PART A -- TOOL FUNCTIONS (pre-built) +# ═══════════════════════════════════════════════════════════════════════════════ +# +# These are the tools the agent can use. Each is a normal Python function. +# The LLM will never call these directly — it will OUTPUT JSON saying +# "please call read_file with path='app.py'", and OUR CODE will call it. + + +def read_file(path: str) -> str: + """Read a .txt or .py file from the workspace and return its contents.""" + target = (WORKSPACE / path).resolve() + if not str(target).startswith(str(WORKSPACE.resolve())): + return "ERROR: path is outside the workspace." + if not target.exists(): + return f"ERROR: file '{path}' not found." + if target.suffix not in (".py", ".txt"): + return f"ERROR: can only read .py and .txt files, got '{target.suffix}'." + return target.read_text() + + +def grep_search(pattern: str, file_glob: str = "*.py") -> str: + """Search for a pattern in workspace files matching the glob.""" + matches = [] + for filepath in sorted(WORKSPACE.glob(file_glob)): + if filepath.suffix not in (".py", ".txt"): + continue + try: + lines = filepath.read_text().splitlines() + except Exception: + continue + for i, line in enumerate(lines, 1): + if pattern in line: + rel = filepath.relative_to(WORKSPACE) + matches.append(f"{rel}:{i}: {line}") + if not matches: + return f"No matches for '{pattern}' in {file_glob}." + return "\n".join(matches) + + +def list_files(file_glob: str = "*") -> str: + """List files in the workspace matching the glob pattern.""" + found = sorted(WORKSPACE.glob(file_glob)) + found = [f.relative_to(WORKSPACE) for f in found if f.is_file()] + if not found: + return f"No files matching '{file_glob}' in workspace." + return "\n".join(str(f) for f in found) + + +def write_file(path: str, content: str) -> str: + """Write content to a .py or .txt file in the workspace.""" + target = (WORKSPACE / path).resolve() + if not str(target).startswith(str(WORKSPACE.resolve())): + return "ERROR: path is outside the workspace." + if target.suffix not in (".py", ".txt"): + return f"ERROR: can only write .py and .txt files, got '{target.suffix}'." + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content) + return f"OK: wrote {len(content)} chars to {path}." + + +def run_python(path: str) -> str: + """Execute a Python file in the workspace and return stdout + stderr.""" + target = (WORKSPACE / path).resolve() + if not str(target).startswith(str(WORKSPACE.resolve())): + return "ERROR: path is outside the workspace." + if not target.exists(): + return f"ERROR: file '{path}' not found." + result = subprocess.run( + [sys.executable, str(target)], + capture_output=True, + text=True, + timeout=30, + cwd=str(WORKSPACE), + ) + output = "" + if result.stdout: + output += f"STDOUT:\n{result.stdout}" + if result.stderr: + output += f"STDERR:\n{result.stderr}" + output += f"\nExit code: {result.returncode}" + return output.strip() + + +def validate_python(path: str) -> str: + """Check whether a Python file has valid syntax using ast.parse.""" + target = (WORKSPACE / path).resolve() + if not str(target).startswith(str(WORKSPACE.resolve())): + return "ERROR: path is outside the workspace." + if not target.exists(): + return f"ERROR: file '{path}' not found." + source = target.read_text() + try: + ast.parse(source) + return "OK: syntax is valid." + except SyntaxError as e: + return f"SYNTAX ERROR: {e}" + + +def done(summary: str) -> str: + """Signal that the agent has finished its task.""" + return f"DONE: {summary}" + + +# ═══════════════════════════════════════════════════════════════════════════════ +# PART B -- TOOL DISPATCHER (pre-built) +# ═══════════════════════════════════════════════════════════════════════════════ +# +# This is the bridge between the LLM's JSON output and Python function calls. +# +# When the LLM says: {"tool": "read_file", "arguments": {"path": "app.py"}} +# The dispatcher does: TOOL_FUNCTIONS["read_file"](path="app.py") +# +# The **arguments syntax means "unpack the dict as keyword arguments": +# {"path": "app.py"} → read_file(path="app.py") + +TOOL_FUNCTIONS = { + "read_file": read_file, + "grep_search": grep_search, + "list_files": list_files, + "write_file": write_file, + "run_python": run_python, + "validate_python": validate_python, + "done": done, +} + + +def dispatch_tool(tool_name: str, arguments: dict) -> str: + """Look up a tool by name and call it with the given arguments. + + Example: + dispatch_tool("read_file", {"path": "app.py"}) + → calls read_file(path="app.py") + → returns the file contents as a string + """ + if tool_name not in TOOL_FUNCTIONS: + return f"ERROR: unknown tool '{tool_name}'. Available: {list(TOOL_FUNCTIONS.keys())}" + func = TOOL_FUNCTIONS[tool_name] + try: + return func(**arguments) + except TypeError as e: + return f"ERROR calling {tool_name}: {e}" + except Exception as e: + return f"ERROR in {tool_name}: {type(e).__name__}: {e}" + + +# ═══════════════════════════════════════════════════════════════════════════════ +# PART C -- SYSTEM PROMPT (TODOs 1-2) +# ═══════════════════════════════════════════════════════════════════════════════ +# +# The system prompt is the MOST IMPORTANT part of the agent. It is the only +# way the LLM knows what tools it has and how to use them. +# +# Think about it: the LLM is just a text model. It has no built-in ability +# to read files or run code. The system prompt is where we TELL it: +# "You have these tools. When you want to use one, output this JSON format. +# I (the code) will parse your JSON, run the tool, and give you the result." +# +# The LLM then "plays along" — it outputs JSON that LOOKS LIKE a tool call, +# and our agent loop code makes it ACTUALLY happen. + +# TODO 1: Complete the TOOL_DESCRIPTIONS string below. +# This text will be embedded in the system prompt inside a section. +# The LLM needs to know: +# - The name of each tool (must match the keys in TOOL_FUNCTIONS above!) +# - What arguments each tool takes +# - What each tool does +# +# Four tools are already described for you as examples. +# Add the missing three: write_file, run_python, validate_python. +# +# Follow the same format: +# - tool_name({"param": ""}): What the tool does. + +TOOL_DESCRIPTIONS = """\ + - read_file({"path": ""}): Read a .py or .txt file from the workspace. + - grep_search({"pattern": "", "file_glob": ""}): Search for a pattern in files. + - list_files({"file_glob": ""}): List files matching the pattern. + - write_file({"path": "", "content": ""}): Write a .py or .txt file to the workspace. + - run_python({"path": ""}): Execute a Python file and return stdout + stderr. + - validate_python({"path": ""}): Check Python file syntax and return result. + - done({"summary": ""}): Signal that you are finished. +""" + +# TODO 2: Complete the system prompt. +# The structure is provided — fill in the and sections. +# +# For , describe these steps: +# 1. PLAN: Think about what steps are needed. List them in "thought". +# 2. ACT: Choose ONE tool to call. +# 3. OBSERVE: Analyse the tool's output carefully. +# 4. REPLAN: If the result was unexpected, revise your plan. +# 5. REPEAT: Go back to ACT if more work is needed. +# 6. DONE: Call the "done" tool when the task is complete. +# +# For , include at least: +# - Always plan before acting. +# - Call exactly ONE tool per response. +# - After writing code, always validate and run it. +# - If an error occurs, try to fix it (up to 3 retries). +# - Stay within the workspace directory. +# - When finished, call the "done" tool. +# +# IMPORTANT: The JSON example uses {{ and }} because this is an f-string. +# In an f-string, {{ produces a literal { in the output. +# So {{"thought": "..."}} becomes {"thought": "..."} when printed. + +SYSTEM_PROMPT = f"""\ +You are a coding agent that helps users with Python programming tasks. +You work inside a workspace directory and have access to tools. + + +Available tools: +{TOOL_DESCRIPTIONS} + + + +To accomplish a task, follow this workflow: + 1. PLAN: Think about what steps are needed. List them in "thought". + 2. ACT: Choose ONE tool to call. + 3. OBSERVE: Analyse the tool's output carefully. + 4. REPLAN: If the result was unexpected, revise your plan. + 5. REPEAT: Go back to ACT if more work is needed. + 6. DONE: Call the "done" tool when the task is complete. + + + +You MUST respond with a JSON object every time. The format is: +{{{{ + "thought": "", + "tool": "", + "arguments": {{{{ }}}} +}}}} + +Example — to read a file: +{{{{ + "thought": "I need to read app.py to understand the code.", + "tool": "read_file", + "arguments": {{{{"path": "app.py"}}}} +}}}} + +Example — to signal completion: +{{{{ + "thought": "I have fixed all the bugs and verified the code runs.", + "tool": "done", + "arguments": {{{{"summary": "Fixed 3 bugs in app.py and verified all tests pass."}}}} +}}}} + + + +Rules for operating as a coding agent: +- Always plan before acting. Your "thought" should explain your reasoning and strategy. +- Call exactly ONE tool per response. Do not try to call multiple tools. +- Always validate Python code after writing it using validate_python. +- Always run Python code after validation to verify it works using run_python. +- If an error occurs, analyze it carefully and retry up to 3 times. +- Stay within the workspace directory. Never try to access files outside it. +- When the task is complete, immediately call the "done" tool with a summary of what was accomplished. +- If a human provides feedback in tags, acknowledge it and adjust your plan accordingly. + +""" + + +# ═══════════════════════════════════════════════════════════════════════════════ +# PART D -- AGENT LOOP (TODOs 3-6) +# ═══════════════════════════════════════════════════════════════════════════════ +# +# This is where everything comes together. The agent loop: +# +# 1. Sends messages to the LLM (including the system prompt with tools). +# 2. The LLM responds with JSON like: {"tool": "read_file", "arguments": {"path": "app.py"}} +# 3. We parse that JSON and call the real Python function. +# 4. We put the result back into the conversation as a new message. +# 5. We send the updated conversation to the LLM again. +# 6. The LLM sees the result and decides what to do next. +# 7. Repeat until the LLM calls "done" or we hit the iteration limit. + + +def truncate_result(result: str) -> str: + """Truncate a tool result if it exceeds MAX_RESULT_LENGTH.""" + if len(result) <= MAX_RESULT_LENGTH: + return result + half = MAX_RESULT_LENGTH // 2 + return ( + result[:half] + + f"\n\n... [TRUNCATED — {len(result)} chars total, showing first and last {half}] ...\n\n" + + result[-half:] + ) + + +def trim_messages(messages: list) -> list: + """Trim older messages if total character count exceeds MAX_HISTORY_CHARS.""" + total = sum(len(m["content"]) for m in messages) + if total <= MAX_HISTORY_CHARS: + return messages + head = messages[:2] + tail = messages[2:] + original_task = messages[1]["content"] if len(messages) > 1 else "" + while tail and sum(len(m["content"]) for m in head + tail) > MAX_HISTORY_CHARS: + tail.pop(0) + reminder = { + "role": "user", + "content": ( + "Earlier conversation history was trimmed. " + f"REMINDER — your original task was:\n{original_task}\n" + "Continue from where you left off." + ), + } + return head + [reminder] + tail + + +def ask_human() -> str: + """Ask the user to approve, redirect, or stop before each action.""" + try: + reply = input( + "\n [Enter]=continue, or type a comment (stop to abort): " + ).strip() + return reply + except (EOFError, KeyboardInterrupt): + return "stop" + + +def agent_loop(user_task: str) -> None: + """Run the agent loop: plan -> user review -> act -> observe -> repeat. + + Study this function carefully — it IS the agent. Everything else is + just support. The loop implements this cycle: + + LLM produces JSON → we parse it → we call the tool → + we feed the result back → LLM produces next JSON → ... + """ + + # TODO 3: Initialise the message list. + # Create a list with two messages: + # 1. {"role": "system", "content": SYSTEM_PROMPT} + # 2. {"role": "user", "content": user_task} + # + # The system message teaches the LLM about its tools. + # The user message is the task to accomplish. + messages = [ + {"role": "system", "content": SYSTEM_PROMPT}, + {"role": "user", "content": user_task}, + ] + + for iteration in range(1, MAX_ITERATIONS + 1): + print_separator(f"Agent Iteration {iteration}") + messages = trim_messages(messages) + + # TODO 4: Get the LLM's next action. + try: + raw = chat_json(client, messages, temperature=0.2, max_tokens=4096) + action = json.loads(raw) + thought = action.get("thought", "") + tool_name = action.get("tool", "") + arguments = action.get("arguments", {}) + except json.JSONDecodeError as e: + print(f"Failed to parse JSON response: {e}") + messages.append({"role": "assistant", "content": raw}) + messages.append( + { + "role": "user", + "content": "Please respond with valid JSON in the specified format.", + } + ) + continue + + # Print the agent's plan + print(f"\nThought: {thought}") + print(f"Tool: {tool_name}") + print(f"Arguments: {arguments}") + + # TODO 5: Human-in-the-loop — let the user review before execution. + if tool_name == "done": + print( + f"\n✓ Agent proposed completion: {arguments.get('summary', 'Task completed')}" + ) + user_input = ask_human() + if user_input.lower() in {"stop", "abort", "cancel"}: + print("Aborted by user.") + return + elif user_input: + messages.append({"role": "assistant", "content": raw}) + messages.append( + { + "role": "user", + "content": f"{user_input}\nPlease revise your plan based on this feedback.", + } + ) + continue + # If user approved (pressed Enter), fall through to execute + else: + user_input = ask_human() + if user_input.lower() in {"stop", "abort", "cancel"}: + print("Agent aborted by user.") + return + elif user_input: + # User provided feedback - don't execute, ask to revise + messages.append({"role": "assistant", "content": raw}) + messages.append( + { + "role": "user", + "content": f"{user_input}\nPlease revise your plan based on this feedback.", + } + ) + continue + + # TODO 6: Execute the tool and feed the result back. + if tool_name == "done": + print(f"\n✓ Agent completed: {arguments.get('summary', 'Task completed')}") + return + + # Call the tool + result = dispatch_tool(tool_name, arguments) + result = truncate_result(result) + + # Append the assistant's response and tool result to the conversation + messages.append({"role": "assistant", "content": raw}) + messages.append( + { + "role": "user", + "content": f'\n{result}\n', + } + ) + + # Print result for debugging + print( + f"\nResult: {result[:200]}..." + if len(result) > 200 + else f"\nResult: {result}" + ) + + print_separator("Agent stopped (max iterations reached)") + + +# ═══════════════════════════════════════════════════════════════════════════════ +# PART E -- INTERACTIVE CHAT (TODOs 7-8) +# ═══════════════════════════════════════════════════════════════════════════════ + +# TODO 7: Implement the input loop. +# - Read input with: user_input = input("You> ").strip() +# - Handle EOFError and KeyboardInterrupt (Ctrl+C) +# - Skip empty input +# - Exit on "quit" or "exit" +# - Otherwise call agent_loop(user_input) + + +def interactive_chat(): + """Run an interactive chat loop where the user gives tasks to the agent.""" + print_separator("AI Coding Agent -- Interactive Mode") + print("Type your task and press Enter. Type 'quit' or 'exit' to stop.") + print(f"Workspace: {WORKSPACE.resolve()}\n") + + # Show what files are in the workspace + files = [f for f in sorted(WORKSPACE.glob("*")) if f.is_file()] + if files: + print("Files in workspace:") + for f in files: + print(f" {f.name}") + else: + print("Workspace is empty.") + print() + + # TODO 7: Implement the input loop. + while True: + try: + user_input = input("You> ").strip() + except (EOFError, KeyboardInterrupt): + print("\nExiting...") + return + + # Skip empty input + if not user_input: + continue + + # Check for exit commands + if user_input.lower() in {"quit", "exit"}: + print("Exiting interactive chat.") + return + + # Run the agent with the user's task + agent_loop(user_input) + print() + + +# ═══════════════════════════════════════════════════════════════════════════════ +# MAIN +# ═══════════════════════════════════════════════════════════════════════════════ + +if __name__ == "__main__": + source = Path(__file__).parent / "analyze_me.py" + dest = WORKSPACE / "analyze_me.py" + if not dest.exists(): + dest.write_text(source.read_text()) + interactive_chat() diff --git a/backend/agent/tools.py b/backend/agent/tools.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/managers/__init__.py b/backend/managers/__init__.py index e69de29..1e212a0 100644 --- a/backend/managers/__init__.py +++ b/backend/managers/__init__.py @@ -0,0 +1,5 @@ +"""Backend Managers - Business logic for UI components""" +from backend.managers.chat_manager import ChatManager +from backend.managers.system_prompter import SystemPrompter + +__all__ = ["ChatManager", "SystemPrompter"] diff --git a/backend/managers/chat_manager.py b/backend/managers/chat_manager.py index e69de29..bcfc2e7 100644 --- a/backend/managers/chat_manager.py +++ b/backend/managers/chat_manager.py @@ -0,0 +1,97 @@ +"""Chat Manager - Handles chat history and AI communication""" + +import os +from dotenv import load_dotenv +import requests +import json + +load_dotenv() + + +class ChatManager: + def __init__(self): + self.api_host = os.getenv("HOST") + self.api_port = os.getenv("PORT") + self.api_key = os.getenv("API_KEY") + self.model = os.getenv("MODEL") + + # API endpoint URL (OpenAI-compatible format) + self.api_url = f"http://{self.api_host}:{self.api_port}/v1/chat/completions" + + # Chat history stored in memory + self.chat_history = [] + + def add_message(self, role: str, content: str) -> None: + self.chat_history.append({"role": role, "content": content}) + + def get_history(self) -> list: + return self.chat_history + + def clear_history(self) -> None: + self.chat_history = [] + + def send_message(self, user_message: str) -> str: + # Add user message to history + self.add_message("user", user_message) + + try: + # Prepare request to OpenAI-compatible API + headers = { + "Content-Type": "application/json", + } + + # Add API key if available + if self.api_key and self.api_key != "EMPTY": + headers["Authorization"] = f"Bearer {self.api_key}" + + payload = { + "model": self.model, + "messages": self.chat_history, + "temperature": 0.7, + "max_tokens": 2000, + "stream": False, + } + + # Make API request + response = requests.post( + self.api_url, headers=headers, json=payload, timeout=30 + ) + + # Check if request was successful + if response.status_code != 200: + error_msg = f"API Error {response.status_code}: {response.text}" + raise Exception(error_msg) + + # Parse response + response_data = response.json() + + # Extract AI message + if "choices" in response_data and len(response_data["choices"]) > 0: + ai_message = response_data["choices"][0]["message"]["content"] + + # Add AI response to history + self.add_message("assistant", ai_message) + + return ai_message + else: + raise Exception("Invalid API response format") + + except requests.exceptions.RequestException as e: + error_msg = f"Connection Error: {str(e)}" + # Add error message to history so user sees it + self.add_message("assistant", f"Error: {error_msg}") + raise Exception(error_msg) + except json.JSONDecodeError as e: + error_msg = f"JSON Decode Error: {str(e)}" + self.add_message("assistant", f"Error: {error_msg}") + raise Exception(error_msg) + except Exception as e: + error_msg = f"Error: {str(e)}" + self.add_message("assistant", f"Error: {error_msg}") + raise Exception(error_msg) + + def get_chat_display(self) -> list: + return [ + {"role": msg["role"], "content": msg["content"]} + for msg in self.chat_history + ] diff --git a/backend/managers/system_prompter.py b/backend/managers/system_prompter.py index e69de29..4fa880c 100644 --- a/backend/managers/system_prompter.py +++ b/backend/managers/system_prompter.py @@ -0,0 +1,38 @@ +"""System Prompter - Builds system prompts with optional file context""" + +MAX_FILE_CHARS = 4000 # Limit file context to avoid token overflow + + +class SystemPrompter: + @staticmethod + def generate_prompt(file_context: dict | None = None) -> str: + """Build a system prompt, optionally embedding a file's content. + + Args: + file_context: dict with keys 'name' and 'content', or None. + + Returns: + A system prompt string. + """ + base = ( + "You are an expert code assistant integrated into a lightweight code editor. " + "Help the user with code suggestions, debugging, explanations, and improvements. " + "Be concise and precise. Use markdown and fenced code blocks where appropriate." + ) + + if file_context: + name = file_context.get("name", "unknown") + content = file_context.get("content", "") + # Truncate large files to avoid exceeding token limits + if len(content) > MAX_FILE_CHARS: + content = content[:MAX_FILE_CHARS] + "\n... [truncated]" + file_section = ( + f"\n\nThe user currently has the following file open in the editor:\n" + f"\n" + f"\n{content}\n\n" + f"\n" + f"Refer to this file when answering questions about the code." + ) + return base + file_section + + return base diff --git a/backend/utils/__init__.py b/backend/utils/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/utils/server_utils.py b/backend/utils/server_utils.py new file mode 100644 index 0000000..e69de29 diff --git a/frontend/chat.py b/frontend/chat.py index 7070c73..888739e 100644 --- a/frontend/chat.py +++ b/frontend/chat.py @@ -1,8 +1,10 @@ import streamlit as st +from backend.managers.chat_manager import ChatManager +from backend.managers.system_prompter import SystemPrompter def render_chat(): st.subheader("Chat with AI Assistant") - + chat_section = st.container() setup_section = st.container() @@ -10,21 +12,37 @@ def render_chat(): if st.session_state.chat_history: for message in st.session_state.chat_history: st.markdown(f"**{message['role'].capitalize()}:** {message['content']}") - + + # Clear the input field before the widget is rendered (Streamlit requirement) + if st.session_state.get("_clear_chat_input"): + st.session_state.chat_input = "" + st.session_state._clear_chat_input = False + user_input = st.text_input("Type your message here:", key="chat_input") - + if st.button("Send", key="send_button") and user_input: + chat_manager = st.session_state.chat_manager + + # Inject system prompt on the first message + if not chat_manager.get_history(): + system_prompt = SystemPrompter.generate_prompt() + chat_manager.add_message("system", system_prompt) + st.session_state.chat_history.append({"role": "user", "content": user_input}) - # Here you would typically call your AI assistant to get a response - # For demonstration, we'll just echo the user's message - ai_response = f"Echo: {user_input}" + + try: + ai_response = chat_manager.send_message(user_input) + except Exception as e: + ai_response = f"Error: {e}" + st.session_state.chat_history.append({"role": "assistant", "content": ai_response}) - st.session_state.chat_input = "" # Clear input after sending - + st.session_state._clear_chat_input = True # Clear input on next rerun + st.rerun() + with setup_section: st.info("This is where you can set up your AI assistant. For now, this section is just a placeholder.") st.toggle("Use debug system prompt", key="use_system_prompt", value=True) - + # Here you could add options to configure the AI assistant, such as selecting a model, setting parameters, etc. diff --git a/frontend/state.py b/frontend/state.py index ffa7b41..9614ffa 100644 --- a/frontend/state.py +++ b/frontend/state.py @@ -1,5 +1,12 @@ import streamlit as st +from backend.managers.chat_manager import ChatManager +def init_state(): + # Sidebar state initialization + + # Chat manager (persists across reruns) + if "chat_manager" not in st.session_state: + st.session_state.chat_manager = ChatManager() def init_state(): # Sidebar if "last_selected" not in st.session_state: diff --git a/tests/test_chat_manager.py b/tests/test_chat_manager.py index e69de29..a771ec0 100644 --- a/tests/test_chat_manager.py +++ b/tests/test_chat_manager.py @@ -0,0 +1,242 @@ +"""Test script for ChatManager - Pytest compatible tests""" + +import sys +from pathlib import Path +import pytest +from unittest.mock import patch, MagicMock + +# Add project root to Python path +sys.path.insert(0, str(Path(__file__).parent.parent)) + +from backend.managers.chat_manager import ChatManager + + +class TestChatManager: + """Test suite for ChatManager functionality.""" + + @pytest.fixture + def chat_manager(self): + return ChatManager() + + def test_initialization(self, chat_manager): + """Test that ChatManager initializes correctly.""" + assert chat_manager.api_url is not None + assert chat_manager.model is not None + assert chat_manager.chat_history == [] + + def test_add_message(self, chat_manager): + """Test adding messages to chat history.""" + chat_manager.add_message("user", "Hello") + assert len(chat_manager.chat_history) == 1 + assert chat_manager.chat_history[0]["role"] == "user" + assert chat_manager.chat_history[0]["content"] == "Hello" + + def test_get_history(self, chat_manager): + """Test retrieving chat history.""" + chat_manager.add_message("user", "Hello") + chat_manager.add_message("assistant", "Hi there!") + + history = chat_manager.get_history() + assert len(history) == 2 + assert history[0]["role"] == "user" + assert history[1]["role"] == "assistant" + + def test_clear_history(self, chat_manager): + """Test clearing chat history.""" + chat_manager.add_message("user", "Hello") + assert len(chat_manager.chat_history) == 1 + + chat_manager.clear_history() + assert len(chat_manager.chat_history) == 0 + + def test_send_message_integration(self, chat_manager): + """ + Integration test for sending message to AI. + This test actually communicates with the API. + """ + try: + # Send a simple test message + response = chat_manager.send_message("Hello, what is 2+2?") + + # Verify response is not empty + assert isinstance(response, str) + assert len(response) > 0 + + # Verify message was added to history + assert len(chat_manager.chat_history) == 2 # user + assistant + assert chat_manager.chat_history[0]["role"] == "user" + assert chat_manager.chat_history[1]["role"] == "assistant" + + print(f"API Test Passed") + print(f"Response: {response}") + + except Exception as e: + # If API is not reachable, mark as skipped + pytest.skip(f"API not reachable: {str(e)}") + + def test_multiple_messages(self, chat_manager): + """Test sending multiple messages in a conversation.""" + try: + # Send first message + response1 = chat_manager.send_message("What is your name?") + assert len(response1) > 0 + + # Send follow-up message + response2 = chat_manager.send_message("Tell me more") + assert len(response2) > 0 + + # Verify full conversation is in history + assert len(chat_manager.chat_history) == 4 # 2 user + 2 assistant + + print(f"Conversation Test Passed") + print(f"Messages: {len(chat_manager.chat_history)}") + + except Exception as e: + pytest.skip(f"API not reachable: {str(e)}") + + +class TestChatManagerSendMessage: + """Unit tests for send_message using mocked HTTP requests.""" + + @pytest.fixture + def chat_manager(self): + return ChatManager() + + def _mock_response(self, content="AI reply", status_code=200): + mock = MagicMock() + mock.status_code = status_code + mock.json.return_value = { + "choices": [{"message": {"role": "assistant", "content": content}}] + } + mock.text = "error text" + return mock + + def test_send_message_adds_user_message_to_history(self, chat_manager): + with patch("requests.post", return_value=self._mock_response()): + chat_manager.send_message("Hello") + assert chat_manager.chat_history[0] == {"role": "user", "content": "Hello"} + + def test_send_message_adds_assistant_response_to_history(self, chat_manager): + with patch("requests.post", return_value=self._mock_response("Hi there")): + chat_manager.send_message("Hello") + assert chat_manager.chat_history[1] == {"role": "assistant", "content": "Hi there"} + + def test_send_message_returns_ai_content(self, chat_manager): + with patch("requests.post", return_value=self._mock_response("Answer")): + response = chat_manager.send_message("Question") + assert response == "Answer" + + def test_send_message_history_grows_with_each_call(self, chat_manager): + with patch("requests.post", return_value=self._mock_response()): + chat_manager.send_message("First") + chat_manager.send_message("Second") + assert len(chat_manager.chat_history) == 4 # 2 user + 2 assistant + + def test_send_message_connection_error_raises(self, chat_manager): + import requests + with patch("requests.post", side_effect=requests.exceptions.ConnectionError("refused")): + with pytest.raises(Exception, match="Connection Error"): + chat_manager.send_message("Hello") + + def test_send_message_api_error_status_raises(self, chat_manager): + mock = self._mock_response(status_code=500) + with patch("requests.post", return_value=mock): + with pytest.raises(Exception, match="API Error 500"): + chat_manager.send_message("Hello") + + def test_send_message_empty_choices_raises(self, chat_manager): + mock = MagicMock() + mock.status_code = 200 + mock.json.return_value = {"choices": []} + with patch("requests.post", return_value=mock): + with pytest.raises(Exception, match="Invalid API response format"): + chat_manager.send_message("Hello") + + def test_send_message_missing_choices_key_raises(self, chat_manager): + mock = MagicMock() + mock.status_code = 200 + mock.json.return_value = {} + with patch("requests.post", return_value=mock): + with pytest.raises(Exception): + chat_manager.send_message("Hello") + + def test_send_message_timeout_raises(self, chat_manager): + import requests + with patch("requests.post", side_effect=requests.exceptions.Timeout()): + with pytest.raises(Exception): + chat_manager.send_message("Hello") + + +class TestChatManagerGetChatDisplay: + """Tests for get_chat_display().""" + + @pytest.fixture + def chat_manager(self): + return ChatManager() + + def test_empty_history_returns_empty_list(self, chat_manager): + assert chat_manager.get_chat_display() == [] + + def test_display_contains_role_and_content_keys(self, chat_manager): + chat_manager.add_message("user", "Hello") + display = chat_manager.get_chat_display() + assert "role" in display[0] + assert "content" in display[0] + + def test_display_preserves_message_order(self, chat_manager): + chat_manager.add_message("user", "First") + chat_manager.add_message("assistant", "Second") + display = chat_manager.get_chat_display() + assert display[0]["role"] == "user" + assert display[1]["role"] == "assistant" + + def test_display_matches_history(self, chat_manager): + chat_manager.add_message("user", "Hi") + chat_manager.add_message("assistant", "Hello!") + assert chat_manager.get_chat_display() == chat_manager.get_history() + + def test_system_message_included_in_display(self, chat_manager): + chat_manager.add_message("system", "You are a helper.") + display = chat_manager.get_chat_display() + assert display[0]["role"] == "system" + + +def test_chat_manager_demo(): + """Demo test - Shows interactive chat (can be run manually).""" + print("\n" + "=" * 60) + print("ChatManager Demo - Interactive Test") + print("=" * 60 + "\n") + + chat_manager = ChatManager() + + print(f"Connected to API: {chat_manager.api_url}") + print(f"Model: {chat_manager.model}\n") + + # Demo conversation + test_messages = ["Hello! What can you do?", "Tell me a joke", "What is Python?"] + + print("Starting conversation...\n") + + for message in test_messages: + print(f"User: {message}") + + try: + response = chat_manager.send_message(message) + print(f"Assistant: {response}\n") + + except Exception as e: + print(f"Error: {str(e)}\n") + pytest.skip(f"API not reachable: {str(e)}") + + # Display full chat history + print("=" * 60) + print("Chat History:") + print("=" * 60) + + for msg in chat_manager.get_history(): + print(f"{msg['role'].upper()}: {msg['content']}\n") + + +if __name__ == "__main__": + # Run with: pytest tests/test_chat_manager.py -v -s + pytest.main([__file__, "-v", "-s"]) diff --git a/tests/test_system_prompter.py b/tests/test_system_prompter.py new file mode 100644 index 0000000..e64b30d --- /dev/null +++ b/tests/test_system_prompter.py @@ -0,0 +1,88 @@ +"""Tests for SystemPrompter.""" + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent.parent)) + +import pytest +from backend.managers.system_prompter import SystemPrompter, MAX_FILE_CHARS + + +class TestSystemPrompterBasePrompt: + """Tests for generate_prompt() without file context.""" + + def test_returns_non_empty_string(self): + prompt = SystemPrompter.generate_prompt() + assert isinstance(prompt, str) + assert len(prompt) > 0 + + def test_describes_code_assistant(self): + prompt = SystemPrompter.generate_prompt() + assert "code assistant" in prompt.lower() + + def test_contains_no_file_xml_tag(self): + prompt = SystemPrompter.generate_prompt() + assert "" not in prompt + + def test_none_equals_no_argument(self): + assert SystemPrompter.generate_prompt(file_context=None) == SystemPrompter.generate_prompt() + + +class TestSystemPrompterWithFileContext: + """Tests for generate_prompt() with file_context provided.""" + + def test_includes_filename(self): + prompt = SystemPrompter.generate_prompt(file_context={"name": "main.py", "content": ""}) + assert "main.py" in prompt + + def test_includes_file_content(self): + prompt = SystemPrompter.generate_prompt(file_context={"name": "app.py", "content": "x = 42"}) + assert "x = 42" in prompt + + def test_uses_xml_file_tag(self): + prompt = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": "pass"}) + assert "" in prompt + + def test_with_context_is_longer_than_base(self): + base = SystemPrompter.generate_prompt() + with_ctx = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": "x=1"}) + assert len(with_ctx) > len(base) + + def test_missing_name_key_uses_unknown(self): + prompt = SystemPrompter.generate_prompt(file_context={"content": "some code"}) + assert "unknown" in prompt + + def test_missing_content_key_does_not_raise(self): + prompt = SystemPrompter.generate_prompt(file_context={"name": "empty.py"}) + assert "empty.py" in prompt + + +class TestSystemPrompterTruncation: + """Tests for file content truncation.""" + + def test_large_file_is_truncated(self): + large = "a" * (MAX_FILE_CHARS + 500) + prompt = SystemPrompter.generate_prompt(file_context={"name": "big.py", "content": large}) + assert "[truncated]" in prompt + + def test_small_file_is_not_truncated(self): + content = "print('hello')" + prompt = SystemPrompter.generate_prompt(file_context={"name": "small.py", "content": content}) + assert "[truncated]" not in prompt + assert content in prompt + + def test_file_exactly_at_limit_is_not_truncated(self): + content = "x" * MAX_FILE_CHARS + prompt = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": content}) + assert "[truncated]" not in prompt + + def test_file_one_over_limit_is_truncated(self): + content = "x" * (MAX_FILE_CHARS + 1) + prompt = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": content}) + assert "[truncated]" in prompt diff --git a/workspace/.gitkeep b/workspace/.gitkeep new file mode 100644 index 0000000..e69de29