Merge branch 'main' of https://gitea.fhgr.ch/meulilivio/AISE1_Project into ui-setup
This commit is contained in:
commit
881edf7a47
81
README.md
81
README.md
@ -8,33 +8,78 @@ AI-Supported Lightweight Code Editor built with Streamlit (AISE501 Spring 2026)
|
||||
AISE_AIAgent/
|
||||
├── frontend/ # Streamlit UI Components
|
||||
│ ├── __init__.py
|
||||
│ ├── app.py # Main Streamlit application
|
||||
│ ├── sidebar.py # File navigation sidebar
|
||||
│ ├── editor.py # Code editor pane
|
||||
│ └── chat.py # Chat interface
|
||||
│ ├── app.py # Main Streamlit application entry point
|
||||
│ ├── sidebar.py # File navigation sidebar component
|
||||
│ ├── editor.py # Code editor pane component
|
||||
│ └── chat.py # Chat interface component
|
||||
│
|
||||
├── backend/ # Backend Logic Modules
|
||||
│ ├── __init__.py
|
||||
│ └── managers/ # Business logic managers
|
||||
│ ├── managers/ # Business logic for UI operations
|
||||
│ │ ├── __init__.py
|
||||
│ │ ├── file_manager.py # File I/O operations for UI (read, write, list files)
|
||||
│ │ ├── chat_manager.py # AI chat management and history
|
||||
│ │ ├── system_prompter.py # System prompts and context injection
|
||||
│ │ ├── search_manager.py # Internet search functionality
|
||||
│ │ ├── execution_engine.py # Code execution and sandboxing
|
||||
│ │ └── debug_logger.py # Logging, error handling, debug messages
|
||||
│ │
|
||||
│ ├── agents/ # AI Agent System
|
||||
│ │ ├── __init__.py
|
||||
│ │ ├── coding_agent.py # Main agent loop (plan-act-observe cycle)
|
||||
│ │ └── tools.py # Tools available to agent (7 functions + dispatcher)
|
||||
│ │
|
||||
│ └── utils/ # Helper Utilities
|
||||
│ ├── __init__.py
|
||||
│ ├── file_manager.py # File I/O operations
|
||||
│ ├── chat_manager.py # AI chat management
|
||||
│ ├── system_prompter.py # System prompts & context
|
||||
│ ├── search_manager.py # Internet search
|
||||
│ ├── execution_engine.py # Code execution
|
||||
│ └── debug_logger.py # Logging & debugging
|
||||
│ └── server_utils.py # LLM client init, chat functions, formatters
|
||||
│
|
||||
├── tests/ # Unit tests
|
||||
├── tests/ # Unit Tests
|
||||
│ ├── __init__.py
|
||||
│ ├── test_file_manager.py
|
||||
│ ├── test_chat_manager.py
|
||||
│ └── test_execution_engine.py
|
||||
│ ├── test_file_manager.py # Tests for file operations
|
||||
│ ├── test_chat_manager.py # Tests for chat functionality
|
||||
│ ├── test_execution_engine.py # Tests for code execution
|
||||
│ └── test_main.py # Integration tests
|
||||
│
|
||||
├── .gitignore # Git exclusions
|
||||
├── requirements.txt # Python dependencies
|
||||
└── README.md # This file
|
||||
├── workspace/ # Agent Sandbox Directory
|
||||
│ └── .gitkeep # Placeholder for agent to work safely in isolation
|
||||
│
|
||||
├── .gitignore # Git exclusions (venv, .env, __pycache__, etc.)
|
||||
├── .env # Local environment variables (NOT committed)
|
||||
├── .env.example # Template for environment variables (IS committed)
|
||||
├── requirements.txt # Python dependencies
|
||||
├── README.md # This file
|
||||
└── project_exercise.pdf # Project specification
|
||||
```
|
||||
|
||||
## Component Responsibilities
|
||||
|
||||
### Frontend (`frontend/`)
|
||||
- **app.py**: Main Streamlit application, layout orchestration
|
||||
- **sidebar.py**: File browser and project navigation
|
||||
- **editor.py**: Code editing interface with syntax highlighting
|
||||
- **chat.py**: AI assistant chat interface
|
||||
|
||||
### Backend Managers (`backend/managers/`)
|
||||
Used directly by Frontend for UI operations:
|
||||
- **file_manager.py**: CRUD operations on project files
|
||||
- **chat_manager.py**: Chat history, message management
|
||||
- **system_prompter.py**: System prompt generation and file context
|
||||
- **execution_engine.py**: Safe code execution with output capture
|
||||
- **debug_logger.py**: Error tracking and log formatting
|
||||
- **search_manager.py**: Web search integration
|
||||
|
||||
### Backend Agents (`backend/agents/`)
|
||||
Independent AI agent system for complex tasks:
|
||||
- **coding_agent.py**: Agent loop (Plan → Act → Observe → Repeat)
|
||||
- **tools.py**: 7 tools agent can use (read/write/run/search/validate/grep/done)
|
||||
|
||||
### Backend Utils (`backend/utils/`)
|
||||
- **server_utils.py**: LLM client initialization, chat helpers, message formatters
|
||||
|
||||
### Workspace (`workspace/`)
|
||||
- Sandbox directory where agent executes and stores files
|
||||
- Prevents agent from accessing files outside this directory
|
||||
|
||||
## Features
|
||||
|
||||
- **File Display & Management**: Browse and edit code files
|
||||
|
||||
@ -0,0 +1,4 @@
|
||||
"""Backend - AI agents, managers, and utilities"""
|
||||
from backend.managers import ChatManager
|
||||
|
||||
__all__ = ["ChatManager"]
|
||||
0
backend/agent/__init__.py
Normal file
0
backend/agent/__init__.py
Normal file
0
backend/agent/coding_agent.py
Normal file
0
backend/agent/coding_agent.py
Normal file
605
backend/agent/coding_agent_example.py
Normal file
605
backend/agent/coding_agent_example.py
Normal file
@ -0,0 +1,605 @@
|
||||
"""
|
||||
Exercise 5b -- Build a Basic AI Coding Agent (Guided Version)
|
||||
==============================================================
|
||||
AISE501 . Prompting in Coding . Spring Semester 2026
|
||||
|
||||
This is a GUIDED version of Exercise 5 with more scaffolding.
|
||||
It teaches the same concepts but reduces boilerplate so you can
|
||||
focus on the key insight: how an LLM uses tools.
|
||||
|
||||
The key insight
|
||||
---------------
|
||||
An LLM cannot run code or read files by itself. But we can give it
|
||||
"superpowers" through a simple trick:
|
||||
|
||||
1. TELL the LLM (via the system prompt) what tools exist.
|
||||
2. ASK the LLM to respond with JSON saying which tool to call.
|
||||
3. PARSE the JSON, call the real Python function, and
|
||||
4. FEED the result back into the conversation as a new message.
|
||||
|
||||
This is how ALL AI coding agents work (Claude Code, Cursor, Copilot).
|
||||
The LLM never actually "runs" code — it just asks us to run it!
|
||||
|
||||
What is already provided
|
||||
------------------------
|
||||
To let you focus on the interesting parts, the following are PRE-BUILT:
|
||||
- All 7 tool functions (Part A) — read_file, grep_search, etc.
|
||||
- The tool dispatcher (Part B) — maps tool names to functions.
|
||||
- Helper functions: truncate_result, trim_messages, ask_human.
|
||||
|
||||
What you need to build (the interesting parts)
|
||||
-----------------------------------------------
|
||||
Part C The SYSTEM PROMPT that teaches the LLM about its tools (TODOs 1-2).
|
||||
Part D The AGENT LOOP that connects the LLM to the tools (TODOs 3-6).
|
||||
Part E The INTERACTIVE CHAT interface (TODOs 7-8).
|
||||
|
||||
Think of it like wiring a robot:
|
||||
- Part A+B are the robot's HANDS (already built).
|
||||
- Part C is the robot's INSTRUCTION MANUAL (you write it).
|
||||
- Part D is the robot's BRAIN LOOP (you wire it).
|
||||
- Part E is the ON SWITCH (you connect it).
|
||||
|
||||
The conversation flow
|
||||
---------------------
|
||||
Here is exactly what happens in one iteration of the agent loop:
|
||||
|
||||
┌─────────────────────────────────────────────────────────────┐
|
||||
│ messages = [ │
|
||||
│ {"role": "system", "content": "<system prompt>"}, │
|
||||
│ {"role": "user", "content": "Fix the bug in app.py"},│
|
||||
│ ] │
|
||||
└──────────────────────────┬──────────────────────────────────┘
|
||||
│
|
||||
┌─────────▼─────────┐
|
||||
│ LLM generates │
|
||||
│ JSON response │
|
||||
└─────────┬─────────┘
|
||||
│
|
||||
┌──────────────▼──────────────┐
|
||||
│ {"thought": "I should...", │
|
||||
│ "tool": "read_file", │
|
||||
│ "arguments": { │
|
||||
│ "path": "app.py" │
|
||||
│ }} │
|
||||
└──────────────┬──────────────┘
|
||||
│
|
||||
┌─────────▼─────────┐
|
||||
│ You parse JSON, │
|
||||
│ call read_file() │
|
||||
└─────────┬─────────┘
|
||||
│
|
||||
┌──────────────▼──────────────────┐
|
||||
│ Append to messages: │
|
||||
│ {"role":"assistant", "content":..}│
|
||||
│ {"role":"user", "content": │
|
||||
│ "<tool_result>file contents │
|
||||
│ </tool_result>"} │
|
||||
└──────────────┬──────────────────┘
|
||||
│
|
||||
┌─────────▼─────────┐
|
||||
│ Next iteration: │
|
||||
│ LLM sees result, │
|
||||
│ picks next tool │
|
||||
└───────────────────┘
|
||||
"""
|
||||
|
||||
import ast
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from server_utils import (
|
||||
chat,
|
||||
chat_json,
|
||||
get_client,
|
||||
print_messages,
|
||||
print_separator,
|
||||
strip_code_fences,
|
||||
)
|
||||
|
||||
client = get_client()
|
||||
|
||||
# ── Agent Configuration ──────────────────────────────────────────────────────
|
||||
WORKSPACE = Path(__file__).parent / "workspace"
|
||||
WORKSPACE.mkdir(exist_ok=True)
|
||||
|
||||
MAX_ITERATIONS = 50
|
||||
MAX_RESULT_LENGTH = 8000
|
||||
MAX_HISTORY_CHARS = 60000
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# PART A -- TOOL FUNCTIONS (pre-built)
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
#
|
||||
# These are the tools the agent can use. Each is a normal Python function.
|
||||
# The LLM will never call these directly — it will OUTPUT JSON saying
|
||||
# "please call read_file with path='app.py'", and OUR CODE will call it.
|
||||
|
||||
|
||||
def read_file(path: str) -> str:
|
||||
"""Read a .txt or .py file from the workspace and return its contents."""
|
||||
target = (WORKSPACE / path).resolve()
|
||||
if not str(target).startswith(str(WORKSPACE.resolve())):
|
||||
return "ERROR: path is outside the workspace."
|
||||
if not target.exists():
|
||||
return f"ERROR: file '{path}' not found."
|
||||
if target.suffix not in (".py", ".txt"):
|
||||
return f"ERROR: can only read .py and .txt files, got '{target.suffix}'."
|
||||
return target.read_text()
|
||||
|
||||
|
||||
def grep_search(pattern: str, file_glob: str = "*.py") -> str:
|
||||
"""Search for a pattern in workspace files matching the glob."""
|
||||
matches = []
|
||||
for filepath in sorted(WORKSPACE.glob(file_glob)):
|
||||
if filepath.suffix not in (".py", ".txt"):
|
||||
continue
|
||||
try:
|
||||
lines = filepath.read_text().splitlines()
|
||||
except Exception:
|
||||
continue
|
||||
for i, line in enumerate(lines, 1):
|
||||
if pattern in line:
|
||||
rel = filepath.relative_to(WORKSPACE)
|
||||
matches.append(f"{rel}:{i}: {line}")
|
||||
if not matches:
|
||||
return f"No matches for '{pattern}' in {file_glob}."
|
||||
return "\n".join(matches)
|
||||
|
||||
|
||||
def list_files(file_glob: str = "*") -> str:
|
||||
"""List files in the workspace matching the glob pattern."""
|
||||
found = sorted(WORKSPACE.glob(file_glob))
|
||||
found = [f.relative_to(WORKSPACE) for f in found if f.is_file()]
|
||||
if not found:
|
||||
return f"No files matching '{file_glob}' in workspace."
|
||||
return "\n".join(str(f) for f in found)
|
||||
|
||||
|
||||
def write_file(path: str, content: str) -> str:
|
||||
"""Write content to a .py or .txt file in the workspace."""
|
||||
target = (WORKSPACE / path).resolve()
|
||||
if not str(target).startswith(str(WORKSPACE.resolve())):
|
||||
return "ERROR: path is outside the workspace."
|
||||
if target.suffix not in (".py", ".txt"):
|
||||
return f"ERROR: can only write .py and .txt files, got '{target.suffix}'."
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
target.write_text(content)
|
||||
return f"OK: wrote {len(content)} chars to {path}."
|
||||
|
||||
|
||||
def run_python(path: str) -> str:
|
||||
"""Execute a Python file in the workspace and return stdout + stderr."""
|
||||
target = (WORKSPACE / path).resolve()
|
||||
if not str(target).startswith(str(WORKSPACE.resolve())):
|
||||
return "ERROR: path is outside the workspace."
|
||||
if not target.exists():
|
||||
return f"ERROR: file '{path}' not found."
|
||||
result = subprocess.run(
|
||||
[sys.executable, str(target)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=30,
|
||||
cwd=str(WORKSPACE),
|
||||
)
|
||||
output = ""
|
||||
if result.stdout:
|
||||
output += f"STDOUT:\n{result.stdout}"
|
||||
if result.stderr:
|
||||
output += f"STDERR:\n{result.stderr}"
|
||||
output += f"\nExit code: {result.returncode}"
|
||||
return output.strip()
|
||||
|
||||
|
||||
def validate_python(path: str) -> str:
|
||||
"""Check whether a Python file has valid syntax using ast.parse."""
|
||||
target = (WORKSPACE / path).resolve()
|
||||
if not str(target).startswith(str(WORKSPACE.resolve())):
|
||||
return "ERROR: path is outside the workspace."
|
||||
if not target.exists():
|
||||
return f"ERROR: file '{path}' not found."
|
||||
source = target.read_text()
|
||||
try:
|
||||
ast.parse(source)
|
||||
return "OK: syntax is valid."
|
||||
except SyntaxError as e:
|
||||
return f"SYNTAX ERROR: {e}"
|
||||
|
||||
|
||||
def done(summary: str) -> str:
|
||||
"""Signal that the agent has finished its task."""
|
||||
return f"DONE: {summary}"
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# PART B -- TOOL DISPATCHER (pre-built)
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
#
|
||||
# This is the bridge between the LLM's JSON output and Python function calls.
|
||||
#
|
||||
# When the LLM says: {"tool": "read_file", "arguments": {"path": "app.py"}}
|
||||
# The dispatcher does: TOOL_FUNCTIONS["read_file"](path="app.py")
|
||||
#
|
||||
# The **arguments syntax means "unpack the dict as keyword arguments":
|
||||
# {"path": "app.py"} → read_file(path="app.py")
|
||||
|
||||
TOOL_FUNCTIONS = {
|
||||
"read_file": read_file,
|
||||
"grep_search": grep_search,
|
||||
"list_files": list_files,
|
||||
"write_file": write_file,
|
||||
"run_python": run_python,
|
||||
"validate_python": validate_python,
|
||||
"done": done,
|
||||
}
|
||||
|
||||
|
||||
def dispatch_tool(tool_name: str, arguments: dict) -> str:
|
||||
"""Look up a tool by name and call it with the given arguments.
|
||||
|
||||
Example:
|
||||
dispatch_tool("read_file", {"path": "app.py"})
|
||||
→ calls read_file(path="app.py")
|
||||
→ returns the file contents as a string
|
||||
"""
|
||||
if tool_name not in TOOL_FUNCTIONS:
|
||||
return f"ERROR: unknown tool '{tool_name}'. Available: {list(TOOL_FUNCTIONS.keys())}"
|
||||
func = TOOL_FUNCTIONS[tool_name]
|
||||
try:
|
||||
return func(**arguments)
|
||||
except TypeError as e:
|
||||
return f"ERROR calling {tool_name}: {e}"
|
||||
except Exception as e:
|
||||
return f"ERROR in {tool_name}: {type(e).__name__}: {e}"
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# PART C -- SYSTEM PROMPT (TODOs 1-2)
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
#
|
||||
# The system prompt is the MOST IMPORTANT part of the agent. It is the only
|
||||
# way the LLM knows what tools it has and how to use them.
|
||||
#
|
||||
# Think about it: the LLM is just a text model. It has no built-in ability
|
||||
# to read files or run code. The system prompt is where we TELL it:
|
||||
# "You have these tools. When you want to use one, output this JSON format.
|
||||
# I (the code) will parse your JSON, run the tool, and give you the result."
|
||||
#
|
||||
# The LLM then "plays along" — it outputs JSON that LOOKS LIKE a tool call,
|
||||
# and our agent loop code makes it ACTUALLY happen.
|
||||
|
||||
# TODO 1: Complete the TOOL_DESCRIPTIONS string below.
|
||||
# This text will be embedded in the system prompt inside a <tools> section.
|
||||
# The LLM needs to know:
|
||||
# - The name of each tool (must match the keys in TOOL_FUNCTIONS above!)
|
||||
# - What arguments each tool takes
|
||||
# - What each tool does
|
||||
#
|
||||
# Four tools are already described for you as examples.
|
||||
# Add the missing three: write_file, run_python, validate_python.
|
||||
#
|
||||
# Follow the same format:
|
||||
# - tool_name({"param": "<description>"}): What the tool does.
|
||||
|
||||
TOOL_DESCRIPTIONS = """\
|
||||
- read_file({"path": "<relative path>"}): Read a .py or .txt file from the workspace.
|
||||
- grep_search({"pattern": "<text>", "file_glob": "<glob, default='*.py'>"}): Search for a pattern in files.
|
||||
- list_files({"file_glob": "<glob, default='*'>"}): List files matching the pattern.
|
||||
- write_file({"path": "<relative path>", "content": "<file content>"}): Write a .py or .txt file to the workspace.
|
||||
- run_python({"path": "<relative path>"}): Execute a Python file and return stdout + stderr.
|
||||
- validate_python({"path": "<relative path>"}): Check Python file syntax and return result.
|
||||
- done({"summary": "<what you accomplished>"}): Signal that you are finished.
|
||||
"""
|
||||
|
||||
# TODO 2: Complete the system prompt.
|
||||
# The structure is provided — fill in the <workflow> and <rules> sections.
|
||||
#
|
||||
# For <workflow>, describe these steps:
|
||||
# 1. PLAN: Think about what steps are needed. List them in "thought".
|
||||
# 2. ACT: Choose ONE tool to call.
|
||||
# 3. OBSERVE: Analyse the tool's output carefully.
|
||||
# 4. REPLAN: If the result was unexpected, revise your plan.
|
||||
# 5. REPEAT: Go back to ACT if more work is needed.
|
||||
# 6. DONE: Call the "done" tool when the task is complete.
|
||||
#
|
||||
# For <rules>, include at least:
|
||||
# - Always plan before acting.
|
||||
# - Call exactly ONE tool per response.
|
||||
# - After writing code, always validate and run it.
|
||||
# - If an error occurs, try to fix it (up to 3 retries).
|
||||
# - Stay within the workspace directory.
|
||||
# - When finished, call the "done" tool.
|
||||
#
|
||||
# IMPORTANT: The JSON example uses {{ and }} because this is an f-string.
|
||||
# In an f-string, {{ produces a literal { in the output.
|
||||
# So {{"thought": "..."}} becomes {"thought": "..."} when printed.
|
||||
|
||||
SYSTEM_PROMPT = f"""\
|
||||
You are a coding agent that helps users with Python programming tasks.
|
||||
You work inside a workspace directory and have access to tools.
|
||||
|
||||
<tools>
|
||||
Available tools:
|
||||
{TOOL_DESCRIPTIONS}
|
||||
</tools>
|
||||
|
||||
<workflow>
|
||||
To accomplish a task, follow this workflow:
|
||||
1. PLAN: Think about what steps are needed. List them in "thought".
|
||||
2. ACT: Choose ONE tool to call.
|
||||
3. OBSERVE: Analyse the tool's output carefully.
|
||||
4. REPLAN: If the result was unexpected, revise your plan.
|
||||
5. REPEAT: Go back to ACT if more work is needed.
|
||||
6. DONE: Call the "done" tool when the task is complete.
|
||||
</workflow>
|
||||
|
||||
<response_format>
|
||||
You MUST respond with a JSON object every time. The format is:
|
||||
{{{{
|
||||
"thought": "<your reasoning about what to do next>",
|
||||
"tool": "<tool name from the list above>",
|
||||
"arguments": {{{{ <arguments for the tool> }}}}
|
||||
}}}}
|
||||
|
||||
Example — to read a file:
|
||||
{{{{
|
||||
"thought": "I need to read app.py to understand the code.",
|
||||
"tool": "read_file",
|
||||
"arguments": {{{{"path": "app.py"}}}}
|
||||
}}}}
|
||||
|
||||
Example — to signal completion:
|
||||
{{{{
|
||||
"thought": "I have fixed all the bugs and verified the code runs.",
|
||||
"tool": "done",
|
||||
"arguments": {{{{"summary": "Fixed 3 bugs in app.py and verified all tests pass."}}}}
|
||||
}}}}
|
||||
</response_format>
|
||||
|
||||
<rules>
|
||||
Rules for operating as a coding agent:
|
||||
- Always plan before acting. Your "thought" should explain your reasoning and strategy.
|
||||
- Call exactly ONE tool per response. Do not try to call multiple tools.
|
||||
- Always validate Python code after writing it using validate_python.
|
||||
- Always run Python code after validation to verify it works using run_python.
|
||||
- If an error occurs, analyze it carefully and retry up to 3 times.
|
||||
- Stay within the workspace directory. Never try to access files outside it.
|
||||
- When the task is complete, immediately call the "done" tool with a summary of what was accomplished.
|
||||
- If a human provides feedback in <human_message> tags, acknowledge it and adjust your plan accordingly.
|
||||
</rules>
|
||||
"""
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# PART D -- AGENT LOOP (TODOs 3-6)
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
#
|
||||
# This is where everything comes together. The agent loop:
|
||||
#
|
||||
# 1. Sends messages to the LLM (including the system prompt with tools).
|
||||
# 2. The LLM responds with JSON like: {"tool": "read_file", "arguments": {"path": "app.py"}}
|
||||
# 3. We parse that JSON and call the real Python function.
|
||||
# 4. We put the result back into the conversation as a new message.
|
||||
# 5. We send the updated conversation to the LLM again.
|
||||
# 6. The LLM sees the result and decides what to do next.
|
||||
# 7. Repeat until the LLM calls "done" or we hit the iteration limit.
|
||||
|
||||
|
||||
def truncate_result(result: str) -> str:
|
||||
"""Truncate a tool result if it exceeds MAX_RESULT_LENGTH."""
|
||||
if len(result) <= MAX_RESULT_LENGTH:
|
||||
return result
|
||||
half = MAX_RESULT_LENGTH // 2
|
||||
return (
|
||||
result[:half]
|
||||
+ f"\n\n... [TRUNCATED — {len(result)} chars total, showing first and last {half}] ...\n\n"
|
||||
+ result[-half:]
|
||||
)
|
||||
|
||||
|
||||
def trim_messages(messages: list) -> list:
|
||||
"""Trim older messages if total character count exceeds MAX_HISTORY_CHARS."""
|
||||
total = sum(len(m["content"]) for m in messages)
|
||||
if total <= MAX_HISTORY_CHARS:
|
||||
return messages
|
||||
head = messages[:2]
|
||||
tail = messages[2:]
|
||||
original_task = messages[1]["content"] if len(messages) > 1 else ""
|
||||
while tail and sum(len(m["content"]) for m in head + tail) > MAX_HISTORY_CHARS:
|
||||
tail.pop(0)
|
||||
reminder = {
|
||||
"role": "user",
|
||||
"content": (
|
||||
"<system_note>Earlier conversation history was trimmed. "
|
||||
f"REMINDER — your original task was:\n{original_task}\n"
|
||||
"Continue from where you left off.</system_note>"
|
||||
),
|
||||
}
|
||||
return head + [reminder] + tail
|
||||
|
||||
|
||||
def ask_human() -> str:
|
||||
"""Ask the user to approve, redirect, or stop before each action."""
|
||||
try:
|
||||
reply = input(
|
||||
"\n [Enter]=continue, or type a comment (stop to abort): "
|
||||
).strip()
|
||||
return reply
|
||||
except (EOFError, KeyboardInterrupt):
|
||||
return "stop"
|
||||
|
||||
|
||||
def agent_loop(user_task: str) -> None:
|
||||
"""Run the agent loop: plan -> user review -> act -> observe -> repeat.
|
||||
|
||||
Study this function carefully — it IS the agent. Everything else is
|
||||
just support. The loop implements this cycle:
|
||||
|
||||
LLM produces JSON → we parse it → we call the tool →
|
||||
we feed the result back → LLM produces next JSON → ...
|
||||
"""
|
||||
|
||||
# TODO 3: Initialise the message list.
|
||||
# Create a list with two messages:
|
||||
# 1. {"role": "system", "content": SYSTEM_PROMPT}
|
||||
# 2. {"role": "user", "content": user_task}
|
||||
#
|
||||
# The system message teaches the LLM about its tools.
|
||||
# The user message is the task to accomplish.
|
||||
messages = [
|
||||
{"role": "system", "content": SYSTEM_PROMPT},
|
||||
{"role": "user", "content": user_task},
|
||||
]
|
||||
|
||||
for iteration in range(1, MAX_ITERATIONS + 1):
|
||||
print_separator(f"Agent Iteration {iteration}")
|
||||
messages = trim_messages(messages)
|
||||
|
||||
# TODO 4: Get the LLM's next action.
|
||||
try:
|
||||
raw = chat_json(client, messages, temperature=0.2, max_tokens=4096)
|
||||
action = json.loads(raw)
|
||||
thought = action.get("thought", "")
|
||||
tool_name = action.get("tool", "")
|
||||
arguments = action.get("arguments", {})
|
||||
except json.JSONDecodeError as e:
|
||||
print(f"Failed to parse JSON response: {e}")
|
||||
messages.append({"role": "assistant", "content": raw})
|
||||
messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Please respond with valid JSON in the specified format.",
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
# Print the agent's plan
|
||||
print(f"\nThought: {thought}")
|
||||
print(f"Tool: {tool_name}")
|
||||
print(f"Arguments: {arguments}")
|
||||
|
||||
# TODO 5: Human-in-the-loop — let the user review before execution.
|
||||
if tool_name == "done":
|
||||
print(
|
||||
f"\n✓ Agent proposed completion: {arguments.get('summary', 'Task completed')}"
|
||||
)
|
||||
user_input = ask_human()
|
||||
if user_input.lower() in {"stop", "abort", "cancel"}:
|
||||
print("Aborted by user.")
|
||||
return
|
||||
elif user_input:
|
||||
messages.append({"role": "assistant", "content": raw})
|
||||
messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": f"<human_message>{user_input}</human_message>\nPlease revise your plan based on this feedback.",
|
||||
}
|
||||
)
|
||||
continue
|
||||
# If user approved (pressed Enter), fall through to execute
|
||||
else:
|
||||
user_input = ask_human()
|
||||
if user_input.lower() in {"stop", "abort", "cancel"}:
|
||||
print("Agent aborted by user.")
|
||||
return
|
||||
elif user_input:
|
||||
# User provided feedback - don't execute, ask to revise
|
||||
messages.append({"role": "assistant", "content": raw})
|
||||
messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": f"<human_message>{user_input}</human_message>\nPlease revise your plan based on this feedback.",
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
# TODO 6: Execute the tool and feed the result back.
|
||||
if tool_name == "done":
|
||||
print(f"\n✓ Agent completed: {arguments.get('summary', 'Task completed')}")
|
||||
return
|
||||
|
||||
# Call the tool
|
||||
result = dispatch_tool(tool_name, arguments)
|
||||
result = truncate_result(result)
|
||||
|
||||
# Append the assistant's response and tool result to the conversation
|
||||
messages.append({"role": "assistant", "content": raw})
|
||||
messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": f'<tool_result tool="{tool_name}">\n{result}\n</tool_result>',
|
||||
}
|
||||
)
|
||||
|
||||
# Print result for debugging
|
||||
print(
|
||||
f"\nResult: {result[:200]}..."
|
||||
if len(result) > 200
|
||||
else f"\nResult: {result}"
|
||||
)
|
||||
|
||||
print_separator("Agent stopped (max iterations reached)")
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# PART E -- INTERACTIVE CHAT (TODOs 7-8)
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
# TODO 7: Implement the input loop.
|
||||
# - Read input with: user_input = input("You> ").strip()
|
||||
# - Handle EOFError and KeyboardInterrupt (Ctrl+C)
|
||||
# - Skip empty input
|
||||
# - Exit on "quit" or "exit"
|
||||
# - Otherwise call agent_loop(user_input)
|
||||
|
||||
|
||||
def interactive_chat():
|
||||
"""Run an interactive chat loop where the user gives tasks to the agent."""
|
||||
print_separator("AI Coding Agent -- Interactive Mode")
|
||||
print("Type your task and press Enter. Type 'quit' or 'exit' to stop.")
|
||||
print(f"Workspace: {WORKSPACE.resolve()}\n")
|
||||
|
||||
# Show what files are in the workspace
|
||||
files = [f for f in sorted(WORKSPACE.glob("*")) if f.is_file()]
|
||||
if files:
|
||||
print("Files in workspace:")
|
||||
for f in files:
|
||||
print(f" {f.name}")
|
||||
else:
|
||||
print("Workspace is empty.")
|
||||
print()
|
||||
|
||||
# TODO 7: Implement the input loop.
|
||||
while True:
|
||||
try:
|
||||
user_input = input("You> ").strip()
|
||||
except (EOFError, KeyboardInterrupt):
|
||||
print("\nExiting...")
|
||||
return
|
||||
|
||||
# Skip empty input
|
||||
if not user_input:
|
||||
continue
|
||||
|
||||
# Check for exit commands
|
||||
if user_input.lower() in {"quit", "exit"}:
|
||||
print("Exiting interactive chat.")
|
||||
return
|
||||
|
||||
# Run the agent with the user's task
|
||||
agent_loop(user_input)
|
||||
print()
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# MAIN
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
if __name__ == "__main__":
|
||||
source = Path(__file__).parent / "analyze_me.py"
|
||||
dest = WORKSPACE / "analyze_me.py"
|
||||
if not dest.exists():
|
||||
dest.write_text(source.read_text())
|
||||
interactive_chat()
|
||||
0
backend/agent/tools.py
Normal file
0
backend/agent/tools.py
Normal file
@ -0,0 +1,5 @@
|
||||
"""Backend Managers - Business logic for UI components"""
|
||||
from backend.managers.chat_manager import ChatManager
|
||||
from backend.managers.system_prompter import SystemPrompter
|
||||
|
||||
__all__ = ["ChatManager", "SystemPrompter"]
|
||||
@ -0,0 +1,97 @@
|
||||
"""Chat Manager - Handles chat history and AI communication"""
|
||||
|
||||
import os
|
||||
from dotenv import load_dotenv
|
||||
import requests
|
||||
import json
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
class ChatManager:
|
||||
def __init__(self):
|
||||
self.api_host = os.getenv("HOST")
|
||||
self.api_port = os.getenv("PORT")
|
||||
self.api_key = os.getenv("API_KEY")
|
||||
self.model = os.getenv("MODEL")
|
||||
|
||||
# API endpoint URL (OpenAI-compatible format)
|
||||
self.api_url = f"http://{self.api_host}:{self.api_port}/v1/chat/completions"
|
||||
|
||||
# Chat history stored in memory
|
||||
self.chat_history = []
|
||||
|
||||
def add_message(self, role: str, content: str) -> None:
|
||||
self.chat_history.append({"role": role, "content": content})
|
||||
|
||||
def get_history(self) -> list:
|
||||
return self.chat_history
|
||||
|
||||
def clear_history(self) -> None:
|
||||
self.chat_history = []
|
||||
|
||||
def send_message(self, user_message: str) -> str:
|
||||
# Add user message to history
|
||||
self.add_message("user", user_message)
|
||||
|
||||
try:
|
||||
# Prepare request to OpenAI-compatible API
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
|
||||
# Add API key if available
|
||||
if self.api_key and self.api_key != "EMPTY":
|
||||
headers["Authorization"] = f"Bearer {self.api_key}"
|
||||
|
||||
payload = {
|
||||
"model": self.model,
|
||||
"messages": self.chat_history,
|
||||
"temperature": 0.7,
|
||||
"max_tokens": 2000,
|
||||
"stream": False,
|
||||
}
|
||||
|
||||
# Make API request
|
||||
response = requests.post(
|
||||
self.api_url, headers=headers, json=payload, timeout=30
|
||||
)
|
||||
|
||||
# Check if request was successful
|
||||
if response.status_code != 200:
|
||||
error_msg = f"API Error {response.status_code}: {response.text}"
|
||||
raise Exception(error_msg)
|
||||
|
||||
# Parse response
|
||||
response_data = response.json()
|
||||
|
||||
# Extract AI message
|
||||
if "choices" in response_data and len(response_data["choices"]) > 0:
|
||||
ai_message = response_data["choices"][0]["message"]["content"]
|
||||
|
||||
# Add AI response to history
|
||||
self.add_message("assistant", ai_message)
|
||||
|
||||
return ai_message
|
||||
else:
|
||||
raise Exception("Invalid API response format")
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
error_msg = f"Connection Error: {str(e)}"
|
||||
# Add error message to history so user sees it
|
||||
self.add_message("assistant", f"Error: {error_msg}")
|
||||
raise Exception(error_msg)
|
||||
except json.JSONDecodeError as e:
|
||||
error_msg = f"JSON Decode Error: {str(e)}"
|
||||
self.add_message("assistant", f"Error: {error_msg}")
|
||||
raise Exception(error_msg)
|
||||
except Exception as e:
|
||||
error_msg = f"Error: {str(e)}"
|
||||
self.add_message("assistant", f"Error: {error_msg}")
|
||||
raise Exception(error_msg)
|
||||
|
||||
def get_chat_display(self) -> list:
|
||||
return [
|
||||
{"role": msg["role"], "content": msg["content"]}
|
||||
for msg in self.chat_history
|
||||
]
|
||||
@ -0,0 +1,38 @@
|
||||
"""System Prompter - Builds system prompts with optional file context"""
|
||||
|
||||
MAX_FILE_CHARS = 4000 # Limit file context to avoid token overflow
|
||||
|
||||
|
||||
class SystemPrompter:
|
||||
@staticmethod
|
||||
def generate_prompt(file_context: dict | None = None) -> str:
|
||||
"""Build a system prompt, optionally embedding a file's content.
|
||||
|
||||
Args:
|
||||
file_context: dict with keys 'name' and 'content', or None.
|
||||
|
||||
Returns:
|
||||
A system prompt string.
|
||||
"""
|
||||
base = (
|
||||
"You are an expert code assistant integrated into a lightweight code editor. "
|
||||
"Help the user with code suggestions, debugging, explanations, and improvements. "
|
||||
"Be concise and precise. Use markdown and fenced code blocks where appropriate."
|
||||
)
|
||||
|
||||
if file_context:
|
||||
name = file_context.get("name", "unknown")
|
||||
content = file_context.get("content", "")
|
||||
# Truncate large files to avoid exceeding token limits
|
||||
if len(content) > MAX_FILE_CHARS:
|
||||
content = content[:MAX_FILE_CHARS] + "\n... [truncated]"
|
||||
file_section = (
|
||||
f"\n\nThe user currently has the following file open in the editor:\n"
|
||||
f"<file name=\"{name}\">\n"
|
||||
f"<code>\n{content}\n</code>\n"
|
||||
f"</file>\n"
|
||||
f"Refer to this file when answering questions about the code."
|
||||
)
|
||||
return base + file_section
|
||||
|
||||
return base
|
||||
0
backend/utils/__init__.py
Normal file
0
backend/utils/__init__.py
Normal file
0
backend/utils/server_utils.py
Normal file
0
backend/utils/server_utils.py
Normal file
@ -1,4 +1,6 @@
|
||||
import streamlit as st
|
||||
from backend.managers.chat_manager import ChatManager
|
||||
from backend.managers.system_prompter import SystemPrompter
|
||||
|
||||
def render_chat():
|
||||
st.subheader("Chat with AI Assistant")
|
||||
@ -11,15 +13,31 @@ def render_chat():
|
||||
for message in st.session_state.chat_history:
|
||||
st.markdown(f"**{message['role'].capitalize()}:** {message['content']}")
|
||||
|
||||
# Clear the input field before the widget is rendered (Streamlit requirement)
|
||||
if st.session_state.get("_clear_chat_input"):
|
||||
st.session_state.chat_input = ""
|
||||
st.session_state._clear_chat_input = False
|
||||
|
||||
user_input = st.text_input("Type your message here:", key="chat_input")
|
||||
|
||||
if st.button("Send", key="send_button") and user_input:
|
||||
chat_manager = st.session_state.chat_manager
|
||||
|
||||
# Inject system prompt on the first message
|
||||
if not chat_manager.get_history():
|
||||
system_prompt = SystemPrompter.generate_prompt()
|
||||
chat_manager.add_message("system", system_prompt)
|
||||
|
||||
st.session_state.chat_history.append({"role": "user", "content": user_input})
|
||||
# Here you would typically call your AI assistant to get a response
|
||||
# For demonstration, we'll just echo the user's message
|
||||
ai_response = f"Echo: {user_input}"
|
||||
|
||||
try:
|
||||
ai_response = chat_manager.send_message(user_input)
|
||||
except Exception as e:
|
||||
ai_response = f"Error: {e}"
|
||||
|
||||
st.session_state.chat_history.append({"role": "assistant", "content": ai_response})
|
||||
st.session_state.chat_input = "" # Clear input after sending
|
||||
st.session_state._clear_chat_input = True # Clear input on next rerun
|
||||
st.rerun()
|
||||
|
||||
with setup_section:
|
||||
st.info("This is where you can set up your AI assistant. For now, this section is just a placeholder.")
|
||||
|
||||
@ -1,5 +1,12 @@
|
||||
import streamlit as st
|
||||
from backend.managers.chat_manager import ChatManager
|
||||
|
||||
def init_state():
|
||||
# Sidebar state initialization
|
||||
|
||||
# Chat manager (persists across reruns)
|
||||
if "chat_manager" not in st.session_state:
|
||||
st.session_state.chat_manager = ChatManager()
|
||||
def init_state():
|
||||
# Sidebar
|
||||
if "last_selected" not in st.session_state:
|
||||
|
||||
@ -0,0 +1,242 @@
|
||||
"""Test script for ChatManager - Pytest compatible tests"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
from unittest.mock import patch, MagicMock
|
||||
|
||||
# Add project root to Python path
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent))
|
||||
|
||||
from backend.managers.chat_manager import ChatManager
|
||||
|
||||
|
||||
class TestChatManager:
|
||||
"""Test suite for ChatManager functionality."""
|
||||
|
||||
@pytest.fixture
|
||||
def chat_manager(self):
|
||||
return ChatManager()
|
||||
|
||||
def test_initialization(self, chat_manager):
|
||||
"""Test that ChatManager initializes correctly."""
|
||||
assert chat_manager.api_url is not None
|
||||
assert chat_manager.model is not None
|
||||
assert chat_manager.chat_history == []
|
||||
|
||||
def test_add_message(self, chat_manager):
|
||||
"""Test adding messages to chat history."""
|
||||
chat_manager.add_message("user", "Hello")
|
||||
assert len(chat_manager.chat_history) == 1
|
||||
assert chat_manager.chat_history[0]["role"] == "user"
|
||||
assert chat_manager.chat_history[0]["content"] == "Hello"
|
||||
|
||||
def test_get_history(self, chat_manager):
|
||||
"""Test retrieving chat history."""
|
||||
chat_manager.add_message("user", "Hello")
|
||||
chat_manager.add_message("assistant", "Hi there!")
|
||||
|
||||
history = chat_manager.get_history()
|
||||
assert len(history) == 2
|
||||
assert history[0]["role"] == "user"
|
||||
assert history[1]["role"] == "assistant"
|
||||
|
||||
def test_clear_history(self, chat_manager):
|
||||
"""Test clearing chat history."""
|
||||
chat_manager.add_message("user", "Hello")
|
||||
assert len(chat_manager.chat_history) == 1
|
||||
|
||||
chat_manager.clear_history()
|
||||
assert len(chat_manager.chat_history) == 0
|
||||
|
||||
def test_send_message_integration(self, chat_manager):
|
||||
"""
|
||||
Integration test for sending message to AI.
|
||||
This test actually communicates with the API.
|
||||
"""
|
||||
try:
|
||||
# Send a simple test message
|
||||
response = chat_manager.send_message("Hello, what is 2+2?")
|
||||
|
||||
# Verify response is not empty
|
||||
assert isinstance(response, str)
|
||||
assert len(response) > 0
|
||||
|
||||
# Verify message was added to history
|
||||
assert len(chat_manager.chat_history) == 2 # user + assistant
|
||||
assert chat_manager.chat_history[0]["role"] == "user"
|
||||
assert chat_manager.chat_history[1]["role"] == "assistant"
|
||||
|
||||
print(f"API Test Passed")
|
||||
print(f"Response: {response}")
|
||||
|
||||
except Exception as e:
|
||||
# If API is not reachable, mark as skipped
|
||||
pytest.skip(f"API not reachable: {str(e)}")
|
||||
|
||||
def test_multiple_messages(self, chat_manager):
|
||||
"""Test sending multiple messages in a conversation."""
|
||||
try:
|
||||
# Send first message
|
||||
response1 = chat_manager.send_message("What is your name?")
|
||||
assert len(response1) > 0
|
||||
|
||||
# Send follow-up message
|
||||
response2 = chat_manager.send_message("Tell me more")
|
||||
assert len(response2) > 0
|
||||
|
||||
# Verify full conversation is in history
|
||||
assert len(chat_manager.chat_history) == 4 # 2 user + 2 assistant
|
||||
|
||||
print(f"Conversation Test Passed")
|
||||
print(f"Messages: {len(chat_manager.chat_history)}")
|
||||
|
||||
except Exception as e:
|
||||
pytest.skip(f"API not reachable: {str(e)}")
|
||||
|
||||
|
||||
class TestChatManagerSendMessage:
|
||||
"""Unit tests for send_message using mocked HTTP requests."""
|
||||
|
||||
@pytest.fixture
|
||||
def chat_manager(self):
|
||||
return ChatManager()
|
||||
|
||||
def _mock_response(self, content="AI reply", status_code=200):
|
||||
mock = MagicMock()
|
||||
mock.status_code = status_code
|
||||
mock.json.return_value = {
|
||||
"choices": [{"message": {"role": "assistant", "content": content}}]
|
||||
}
|
||||
mock.text = "error text"
|
||||
return mock
|
||||
|
||||
def test_send_message_adds_user_message_to_history(self, chat_manager):
|
||||
with patch("requests.post", return_value=self._mock_response()):
|
||||
chat_manager.send_message("Hello")
|
||||
assert chat_manager.chat_history[0] == {"role": "user", "content": "Hello"}
|
||||
|
||||
def test_send_message_adds_assistant_response_to_history(self, chat_manager):
|
||||
with patch("requests.post", return_value=self._mock_response("Hi there")):
|
||||
chat_manager.send_message("Hello")
|
||||
assert chat_manager.chat_history[1] == {"role": "assistant", "content": "Hi there"}
|
||||
|
||||
def test_send_message_returns_ai_content(self, chat_manager):
|
||||
with patch("requests.post", return_value=self._mock_response("Answer")):
|
||||
response = chat_manager.send_message("Question")
|
||||
assert response == "Answer"
|
||||
|
||||
def test_send_message_history_grows_with_each_call(self, chat_manager):
|
||||
with patch("requests.post", return_value=self._mock_response()):
|
||||
chat_manager.send_message("First")
|
||||
chat_manager.send_message("Second")
|
||||
assert len(chat_manager.chat_history) == 4 # 2 user + 2 assistant
|
||||
|
||||
def test_send_message_connection_error_raises(self, chat_manager):
|
||||
import requests
|
||||
with patch("requests.post", side_effect=requests.exceptions.ConnectionError("refused")):
|
||||
with pytest.raises(Exception, match="Connection Error"):
|
||||
chat_manager.send_message("Hello")
|
||||
|
||||
def test_send_message_api_error_status_raises(self, chat_manager):
|
||||
mock = self._mock_response(status_code=500)
|
||||
with patch("requests.post", return_value=mock):
|
||||
with pytest.raises(Exception, match="API Error 500"):
|
||||
chat_manager.send_message("Hello")
|
||||
|
||||
def test_send_message_empty_choices_raises(self, chat_manager):
|
||||
mock = MagicMock()
|
||||
mock.status_code = 200
|
||||
mock.json.return_value = {"choices": []}
|
||||
with patch("requests.post", return_value=mock):
|
||||
with pytest.raises(Exception, match="Invalid API response format"):
|
||||
chat_manager.send_message("Hello")
|
||||
|
||||
def test_send_message_missing_choices_key_raises(self, chat_manager):
|
||||
mock = MagicMock()
|
||||
mock.status_code = 200
|
||||
mock.json.return_value = {}
|
||||
with patch("requests.post", return_value=mock):
|
||||
with pytest.raises(Exception):
|
||||
chat_manager.send_message("Hello")
|
||||
|
||||
def test_send_message_timeout_raises(self, chat_manager):
|
||||
import requests
|
||||
with patch("requests.post", side_effect=requests.exceptions.Timeout()):
|
||||
with pytest.raises(Exception):
|
||||
chat_manager.send_message("Hello")
|
||||
|
||||
|
||||
class TestChatManagerGetChatDisplay:
|
||||
"""Tests for get_chat_display()."""
|
||||
|
||||
@pytest.fixture
|
||||
def chat_manager(self):
|
||||
return ChatManager()
|
||||
|
||||
def test_empty_history_returns_empty_list(self, chat_manager):
|
||||
assert chat_manager.get_chat_display() == []
|
||||
|
||||
def test_display_contains_role_and_content_keys(self, chat_manager):
|
||||
chat_manager.add_message("user", "Hello")
|
||||
display = chat_manager.get_chat_display()
|
||||
assert "role" in display[0]
|
||||
assert "content" in display[0]
|
||||
|
||||
def test_display_preserves_message_order(self, chat_manager):
|
||||
chat_manager.add_message("user", "First")
|
||||
chat_manager.add_message("assistant", "Second")
|
||||
display = chat_manager.get_chat_display()
|
||||
assert display[0]["role"] == "user"
|
||||
assert display[1]["role"] == "assistant"
|
||||
|
||||
def test_display_matches_history(self, chat_manager):
|
||||
chat_manager.add_message("user", "Hi")
|
||||
chat_manager.add_message("assistant", "Hello!")
|
||||
assert chat_manager.get_chat_display() == chat_manager.get_history()
|
||||
|
||||
def test_system_message_included_in_display(self, chat_manager):
|
||||
chat_manager.add_message("system", "You are a helper.")
|
||||
display = chat_manager.get_chat_display()
|
||||
assert display[0]["role"] == "system"
|
||||
|
||||
|
||||
def test_chat_manager_demo():
|
||||
"""Demo test - Shows interactive chat (can be run manually)."""
|
||||
print("\n" + "=" * 60)
|
||||
print("ChatManager Demo - Interactive Test")
|
||||
print("=" * 60 + "\n")
|
||||
|
||||
chat_manager = ChatManager()
|
||||
|
||||
print(f"Connected to API: {chat_manager.api_url}")
|
||||
print(f"Model: {chat_manager.model}\n")
|
||||
|
||||
# Demo conversation
|
||||
test_messages = ["Hello! What can you do?", "Tell me a joke", "What is Python?"]
|
||||
|
||||
print("Starting conversation...\n")
|
||||
|
||||
for message in test_messages:
|
||||
print(f"User: {message}")
|
||||
|
||||
try:
|
||||
response = chat_manager.send_message(message)
|
||||
print(f"Assistant: {response}\n")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error: {str(e)}\n")
|
||||
pytest.skip(f"API not reachable: {str(e)}")
|
||||
|
||||
# Display full chat history
|
||||
print("=" * 60)
|
||||
print("Chat History:")
|
||||
print("=" * 60)
|
||||
|
||||
for msg in chat_manager.get_history():
|
||||
print(f"{msg['role'].upper()}: {msg['content']}\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Run with: pytest tests/test_chat_manager.py -v -s
|
||||
pytest.main([__file__, "-v", "-s"])
|
||||
88
tests/test_system_prompter.py
Normal file
88
tests/test_system_prompter.py
Normal file
@ -0,0 +1,88 @@
|
||||
"""Tests for SystemPrompter."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent))
|
||||
|
||||
import pytest
|
||||
from backend.managers.system_prompter import SystemPrompter, MAX_FILE_CHARS
|
||||
|
||||
|
||||
class TestSystemPrompterBasePrompt:
|
||||
"""Tests for generate_prompt() without file context."""
|
||||
|
||||
def test_returns_non_empty_string(self):
|
||||
prompt = SystemPrompter.generate_prompt()
|
||||
assert isinstance(prompt, str)
|
||||
assert len(prompt) > 0
|
||||
|
||||
def test_describes_code_assistant(self):
|
||||
prompt = SystemPrompter.generate_prompt()
|
||||
assert "code assistant" in prompt.lower()
|
||||
|
||||
def test_contains_no_file_xml_tag(self):
|
||||
prompt = SystemPrompter.generate_prompt()
|
||||
assert "<file" not in prompt
|
||||
assert "<code>" not in prompt
|
||||
|
||||
def test_none_equals_no_argument(self):
|
||||
assert SystemPrompter.generate_prompt(file_context=None) == SystemPrompter.generate_prompt()
|
||||
|
||||
|
||||
class TestSystemPrompterWithFileContext:
|
||||
"""Tests for generate_prompt() with file_context provided."""
|
||||
|
||||
def test_includes_filename(self):
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "main.py", "content": ""})
|
||||
assert "main.py" in prompt
|
||||
|
||||
def test_includes_file_content(self):
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "app.py", "content": "x = 42"})
|
||||
assert "x = 42" in prompt
|
||||
|
||||
def test_uses_xml_file_tag(self):
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": "pass"})
|
||||
assert "<file" in prompt
|
||||
|
||||
def test_uses_xml_code_tag(self):
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": "pass"})
|
||||
assert "<code>" in prompt
|
||||
|
||||
def test_with_context_is_longer_than_base(self):
|
||||
base = SystemPrompter.generate_prompt()
|
||||
with_ctx = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": "x=1"})
|
||||
assert len(with_ctx) > len(base)
|
||||
|
||||
def test_missing_name_key_uses_unknown(self):
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"content": "some code"})
|
||||
assert "unknown" in prompt
|
||||
|
||||
def test_missing_content_key_does_not_raise(self):
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "empty.py"})
|
||||
assert "empty.py" in prompt
|
||||
|
||||
|
||||
class TestSystemPrompterTruncation:
|
||||
"""Tests for file content truncation."""
|
||||
|
||||
def test_large_file_is_truncated(self):
|
||||
large = "a" * (MAX_FILE_CHARS + 500)
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "big.py", "content": large})
|
||||
assert "[truncated]" in prompt
|
||||
|
||||
def test_small_file_is_not_truncated(self):
|
||||
content = "print('hello')"
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "small.py", "content": content})
|
||||
assert "[truncated]" not in prompt
|
||||
assert content in prompt
|
||||
|
||||
def test_file_exactly_at_limit_is_not_truncated(self):
|
||||
content = "x" * MAX_FILE_CHARS
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": content})
|
||||
assert "[truncated]" not in prompt
|
||||
|
||||
def test_file_one_over_limit_is_truncated(self):
|
||||
content = "x" * (MAX_FILE_CHARS + 1)
|
||||
prompt = SystemPrompter.generate_prompt(file_context={"name": "f.py", "content": content})
|
||||
assert "[truncated]" in prompt
|
||||
0
workspace/.gitkeep
Normal file
0
workspace/.gitkeep
Normal file
Loading…
x
Reference in New Issue
Block a user