Commit ·
cbf1ed7
0
Parent(s):
clean initial commit
Browse files- .DS_Store +0 -0
- .gitattributes +35 -0
- .gitignore +6 -0
- Dockerfile +11 -0
- README.md +10 -0
- backend/.DS_Store +0 -0
- backend/app/__init__.py +9 -0
- backend/app/agent.py +334 -0
- backend/app/dataset.json +62 -0
- backend/app/environment.py +281 -0
- backend/app/evaluator.py +310 -0
- backend/main.py +332 -0
- backend/main_old.py +291 -0
- backend/requirements.txt +6 -0
- backend/test_env.py +96 -0
- inference.py +93 -0
- openenv.yaml +111 -0
.DS_Store
ADDED
|
Binary file (8.2 kB). View file
|
|
|
.gitattributes
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
.gitignore
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
venv/
|
| 2 |
+
__pycache__/
|
| 3 |
+
*.pyc
|
| 4 |
+
*.pyo
|
| 5 |
+
*.pyd
|
| 6 |
+
.env
|
Dockerfile
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM python:3.10-slim
|
| 2 |
+
|
| 3 |
+
WORKDIR /app
|
| 4 |
+
|
| 5 |
+
COPY . .
|
| 6 |
+
|
| 7 |
+
RUN pip install --no-cache-dir -r backend/requirements.txt
|
| 8 |
+
|
| 9 |
+
EXPOSE 7860
|
| 10 |
+
|
| 11 |
+
CMD ["uvicorn", "backend.main:app", "--host", "0.0.0.0", "--port", "7860"]
|
README.md
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
title: Scalar Hackathon Space 1
|
| 3 |
+
emoji: 🌍
|
| 4 |
+
colorFrom: indigo
|
| 5 |
+
colorTo: pink
|
| 6 |
+
sdk: docker
|
| 7 |
+
pinned: false
|
| 8 |
+
---
|
| 9 |
+
|
| 10 |
+
Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
|
backend/.DS_Store
ADDED
|
Binary file (10.2 kB). View file
|
|
|
backend/app/__init__.py
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
AI Interview Preparation - Core Application Package
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
from .environment import InterviewEnv
|
| 6 |
+
from .evaluator import Evaluator
|
| 7 |
+
from .agent import InterviewAgent
|
| 8 |
+
|
| 9 |
+
__all__ = ['InterviewEnv', 'Evaluator', 'InterviewAgent']
|
backend/app/agent.py
ADDED
|
@@ -0,0 +1,334 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
AI Agent with Retry and Memory Capabilities
|
| 3 |
+
Uses HuggingFace LLM via OpenAI-compatible API
|
| 4 |
+
|
| 5 |
+
Features:
|
| 6 |
+
- Real LLM integration (HuggingFace Qwen model)
|
| 7 |
+
- Retry mechanism with feedback integration
|
| 8 |
+
- Memory of previous attempts
|
| 9 |
+
- Mock fallback when API unavailable
|
| 10 |
+
"""
|
| 11 |
+
|
| 12 |
+
import os
|
| 13 |
+
from typing import List, Dict, Any, Optional
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
class InterviewAgent:
|
| 17 |
+
"""
|
| 18 |
+
AI Agent for Interview Preparation
|
| 19 |
+
|
| 20 |
+
Simulates a candidate that:
|
| 21 |
+
1. Generates answers to questions
|
| 22 |
+
2. Receives feedback on answers
|
| 23 |
+
3. Improves answers using feedback
|
| 24 |
+
4. Tracks performance over time
|
| 25 |
+
|
| 26 |
+
The agent can use:
|
| 27 |
+
- Mock/simulated answers (for testing)
|
| 28 |
+
- Real LLM API (GPT-4, Claude, etc.)
|
| 29 |
+
"""
|
| 30 |
+
|
| 31 |
+
def __init__(self, mode: str = "mock", api_key: str = None, model: str = None):
|
| 32 |
+
"""
|
| 33 |
+
Initialize the AI agent
|
| 34 |
+
|
| 35 |
+
Args:
|
| 36 |
+
mode: "mock" for simulated answers or "api" for real LLM
|
| 37 |
+
api_key: API key for HuggingFace (or from HF_TOKEN env variable)
|
| 38 |
+
model: Model name (or from MODEL_NAME env variable)
|
| 39 |
+
"""
|
| 40 |
+
self.mode = mode
|
| 41 |
+
self.api_key = api_key or os.getenv("HF_TOKEN")
|
| 42 |
+
self.model = model or os.getenv("MODEL_NAME", "Qwen/Qwen3-Coder-Next:novita")
|
| 43 |
+
self.base_url = os.getenv("API_BASE_URL", "https://router.huggingface.co/v1")
|
| 44 |
+
self.memory = []
|
| 45 |
+
self.current_question_memory = []
|
| 46 |
+
self.client = None
|
| 47 |
+
|
| 48 |
+
# Initialize API client if using real LLM
|
| 49 |
+
if mode == "api" and self.api_key:
|
| 50 |
+
self._init_api_client()
|
| 51 |
+
elif mode == "api":
|
| 52 |
+
print("⚠️ API mode selected but no API key provided. Falling back to mock mode.")
|
| 53 |
+
self.mode = "mock"
|
| 54 |
+
|
| 55 |
+
def _init_api_client(self):
|
| 56 |
+
"""Initialize LLM client via OpenAI-compatible API"""
|
| 57 |
+
try:
|
| 58 |
+
from openai import OpenAI
|
| 59 |
+
self.client = OpenAI(
|
| 60 |
+
base_url=self.base_url,
|
| 61 |
+
api_key=self.api_key
|
| 62 |
+
)
|
| 63 |
+
print(f"🔌 API client initialized")
|
| 64 |
+
print(f" Base URL: {self.base_url}")
|
| 65 |
+
print(f" Model: {self.model}")
|
| 66 |
+
except ImportError:
|
| 67 |
+
print("⚠️ openai package not installed. Install with: pip install openai")
|
| 68 |
+
print("⚠️ Falling back to mock mode")
|
| 69 |
+
self.mode = "mock"
|
| 70 |
+
except Exception as e:
|
| 71 |
+
print(f"⚠️ Error initializing API client: {e}")
|
| 72 |
+
print("⚠️ Falling back to mock mode")
|
| 73 |
+
self.mode = "mock"
|
| 74 |
+
|
| 75 |
+
def generate_answer(
|
| 76 |
+
self,
|
| 77 |
+
question: str,
|
| 78 |
+
feedback: Optional[str] = None,
|
| 79 |
+
missing_keywords: Optional[List[str]] = None,
|
| 80 |
+
previous_attempts: Optional[List[Dict]] = None
|
| 81 |
+
) -> str:
|
| 82 |
+
"""
|
| 83 |
+
Generate an answer to the interview question
|
| 84 |
+
|
| 85 |
+
Uses different strategies based on whether this is:
|
| 86 |
+
- Initial attempt: Generate fresh answer
|
| 87 |
+
- Retry attempt: Improve using feedback and missing keywords
|
| 88 |
+
|
| 89 |
+
Args:
|
| 90 |
+
question: The interview question
|
| 91 |
+
feedback: Feedback from previous attempt (if retry)
|
| 92 |
+
missing_keywords: Keywords that were missing (if retry)
|
| 93 |
+
previous_attempts: List of previous attempts (for context)
|
| 94 |
+
|
| 95 |
+
Returns:
|
| 96 |
+
Generated answer string
|
| 97 |
+
"""
|
| 98 |
+
if self.mode == "mock":
|
| 99 |
+
return self._generate_mock_answer(
|
| 100 |
+
question,
|
| 101 |
+
feedback,
|
| 102 |
+
missing_keywords,
|
| 103 |
+
previous_attempts
|
| 104 |
+
)
|
| 105 |
+
else:
|
| 106 |
+
return self._generate_api_answer(
|
| 107 |
+
question,
|
| 108 |
+
feedback,
|
| 109 |
+
missing_keywords,
|
| 110 |
+
previous_attempts
|
| 111 |
+
)
|
| 112 |
+
|
| 113 |
+
def _generate_mock_answer(
|
| 114 |
+
self,
|
| 115 |
+
question: str,
|
| 116 |
+
feedback: Optional[str] = None,
|
| 117 |
+
missing_keywords: Optional[List[str]] = None,
|
| 118 |
+
previous_attempts: Optional[List[Dict]] = None
|
| 119 |
+
) -> str:
|
| 120 |
+
"""
|
| 121 |
+
Generate simulated answer (for testing without API costs)
|
| 122 |
+
|
| 123 |
+
Mock Logic:
|
| 124 |
+
- Initial attempt: Returns partial answer (simulates incomplete knowledge)
|
| 125 |
+
- Retry attempt: Adds missing keywords to answer (simulates learning)
|
| 126 |
+
|
| 127 |
+
This allows testing the retry/feedback loop without LLM API calls.
|
| 128 |
+
|
| 129 |
+
Args:
|
| 130 |
+
question: Interview question
|
| 131 |
+
feedback: Previous feedback
|
| 132 |
+
missing_keywords: Keywords to add
|
| 133 |
+
previous_attempts: History
|
| 134 |
+
|
| 135 |
+
Returns:
|
| 136 |
+
Simulated answer
|
| 137 |
+
"""
|
| 138 |
+
# Database of mock answers for common topics
|
| 139 |
+
# These are intentionally partial to trigger retry mechanism
|
| 140 |
+
mock_knowledge_base = {
|
| 141 |
+
"binary search": "Binary search is an algorithm that finds items in a sorted array by dividing the search space.",
|
| 142 |
+
"process": "A process is a program in execution. A thread is a unit of execution within a process.",
|
| 143 |
+
"normalization": "Database normalization organizes data to reduce redundancy using normal forms.",
|
| 144 |
+
"oop": "The pillars of OOP include encapsulation and inheritance.",
|
| 145 |
+
"deadlock": "A deadlock is when processes wait for each other's resources.",
|
| 146 |
+
"hashmap": "A hashmap stores key-value pairs using a hash function.",
|
| 147 |
+
"sql": "SQL databases use structured schemas. NoSQL databases are more flexible.",
|
| 148 |
+
"solid": "SOLID includes Single Responsibility and Open-Closed principles.",
|
| 149 |
+
"stack": "Stack uses LIFO. Heap is for dynamic allocation.",
|
| 150 |
+
"rest": "REST is an architectural style for web services using HTTP.",
|
| 151 |
+
"big o": "Big O notation describes algorithm complexity.",
|
| 152 |
+
"cap": "CAP theorem states you can only achieve two of three properties in distributed systems."
|
| 153 |
+
}
|
| 154 |
+
|
| 155 |
+
# Find relevant base answer
|
| 156 |
+
question_lower = question.lower()
|
| 157 |
+
base_answer = None
|
| 158 |
+
|
| 159 |
+
for key, answer in mock_knowledge_base.items():
|
| 160 |
+
if key in question_lower:
|
| 161 |
+
base_answer = answer
|
| 162 |
+
break
|
| 163 |
+
|
| 164 |
+
if base_answer is None:
|
| 165 |
+
base_answer = "This is a complex topic in computer science."
|
| 166 |
+
|
| 167 |
+
# If this is a retry, improve the answer by adding missing keywords
|
| 168 |
+
if feedback and missing_keywords:
|
| 169 |
+
improved_answer = base_answer
|
| 170 |
+
|
| 171 |
+
# Add missing keywords to answer (simulating learning)
|
| 172 |
+
if missing_keywords:
|
| 173 |
+
improved_answer += " Important concepts include: "
|
| 174 |
+
improved_answer += ", ".join(missing_keywords[:4]) + "."
|
| 175 |
+
|
| 176 |
+
# Add some variety
|
| 177 |
+
attempt_count = len(previous_attempts) if previous_attempts else 1
|
| 178 |
+
if attempt_count > 1:
|
| 179 |
+
improved_answer += f" Additionally, this relates to fundamental principles of the domain."
|
| 180 |
+
|
| 181 |
+
return improved_answer
|
| 182 |
+
|
| 183 |
+
# Return initial (incomplete) answer
|
| 184 |
+
return base_answer
|
| 185 |
+
|
| 186 |
+
def _generate_api_answer(
|
| 187 |
+
self,
|
| 188 |
+
question: str,
|
| 189 |
+
feedback: Optional[str] = None,
|
| 190 |
+
missing_keywords: Optional[List[str]] = None,
|
| 191 |
+
previous_attempts: Optional[List[Dict]] = None
|
| 192 |
+
) -> str:
|
| 193 |
+
"""
|
| 194 |
+
Generate answer using HuggingFace LLM API
|
| 195 |
+
|
| 196 |
+
Constructs different prompts based on:
|
| 197 |
+
- Initial attempt: Clear, structured question
|
| 198 |
+
- Retry attempt: Includes feedback and missing concepts
|
| 199 |
+
"""
|
| 200 |
+
if not self.client:
|
| 201 |
+
return self._generate_mock_answer(question, feedback, missing_keywords, previous_attempts)
|
| 202 |
+
|
| 203 |
+
# Build prompt
|
| 204 |
+
if feedback and missing_keywords:
|
| 205 |
+
# Retry prompt with feedback
|
| 206 |
+
prompt = f"""You are a technical interview candidate who received feedback on your previous answer.
|
| 207 |
+
|
| 208 |
+
Question: {question}
|
| 209 |
+
|
| 210 |
+
Previous Feedback: {feedback}
|
| 211 |
+
|
| 212 |
+
Missing Concepts: {', '.join(missing_keywords)}
|
| 213 |
+
|
| 214 |
+
Please provide an IMPROVED answer that addresses the feedback and includes the missing concepts. Be clear, concise, and comprehensive."""
|
| 215 |
+
else:
|
| 216 |
+
# Initial prompt
|
| 217 |
+
prompt = f"""You are a technical interview candidate. Answer the following question clearly and comprehensively, covering all key concepts:
|
| 218 |
+
|
| 219 |
+
Question: {question}
|
| 220 |
+
|
| 221 |
+
Provide a well-structured answer:"""
|
| 222 |
+
|
| 223 |
+
try:
|
| 224 |
+
response = self.client.chat.completions.create(
|
| 225 |
+
model=self.model,
|
| 226 |
+
messages=[
|
| 227 |
+
{"role": "system", "content": "You are a knowledgeable technical interview candidate."},
|
| 228 |
+
{"role": "user", "content": prompt}
|
| 229 |
+
],
|
| 230 |
+
temperature=0.7,
|
| 231 |
+
max_tokens=300
|
| 232 |
+
)
|
| 233 |
+
return response.choices[0].message.content.strip()
|
| 234 |
+
except Exception as e:
|
| 235 |
+
print(f"Error calling HuggingFace API: {e}")
|
| 236 |
+
return self._generate_mock_answer(question, feedback, missing_keywords, previous_attempts)
|
| 237 |
+
|
| 238 |
+
def remember_attempt(
|
| 239 |
+
self,
|
| 240 |
+
question: str,
|
| 241 |
+
answer: str,
|
| 242 |
+
score: float,
|
| 243 |
+
feedback: str,
|
| 244 |
+
attempt: int
|
| 245 |
+
):
|
| 246 |
+
"""
|
| 247 |
+
Store attempt in memory for learning and analysis
|
| 248 |
+
|
| 249 |
+
Memory allows agent to:
|
| 250 |
+
- Track improvement over time
|
| 251 |
+
- Analyze common mistakes
|
| 252 |
+
- Build knowledge base
|
| 253 |
+
|
| 254 |
+
Args:
|
| 255 |
+
question: The question asked
|
| 256 |
+
answer: Answer provided
|
| 257 |
+
score: Score received
|
| 258 |
+
feedback: Feedback received
|
| 259 |
+
attempt: Attempt number
|
| 260 |
+
"""
|
| 261 |
+
memory_entry = {
|
| 262 |
+
"question": question,
|
| 263 |
+
"answer": answer,
|
| 264 |
+
"score": score,
|
| 265 |
+
"feedback": feedback,
|
| 266 |
+
"attempt": attempt
|
| 267 |
+
}
|
| 268 |
+
|
| 269 |
+
# Add to both global and question-specific memory
|
| 270 |
+
self.memory.append(memory_entry)
|
| 271 |
+
self.current_question_memory.append(memory_entry)
|
| 272 |
+
|
| 273 |
+
def reset_question_memory(self):
|
| 274 |
+
"""
|
| 275 |
+
Clear memory for current question
|
| 276 |
+
|
| 277 |
+
Called when starting a new question/episode.
|
| 278 |
+
Keeps global memory but clears question-specific memory.
|
| 279 |
+
"""
|
| 280 |
+
self.current_question_memory = []
|
| 281 |
+
|
| 282 |
+
def get_improvement_stats(self) -> Dict[str, Any]:
|
| 283 |
+
"""
|
| 284 |
+
Calculate improvement statistics from memory
|
| 285 |
+
|
| 286 |
+
Returns:
|
| 287 |
+
Dictionary with:
|
| 288 |
+
- total_attempts: Total questions attempted
|
| 289 |
+
- average_score: Average score across all attempts
|
| 290 |
+
- improvement_rate: How much scores improved on retries
|
| 291 |
+
- total_retries: Number of retry attempts made
|
| 292 |
+
"""
|
| 293 |
+
if not self.memory:
|
| 294 |
+
return {
|
| 295 |
+
"total_attempts": 0,
|
| 296 |
+
"average_score": 0.0,
|
| 297 |
+
"improvement_rate": 0.0,
|
| 298 |
+
"total_retries": 0
|
| 299 |
+
}
|
| 300 |
+
|
| 301 |
+
scores = [m["score"] for m in self.memory]
|
| 302 |
+
retries = [m for m in self.memory if m["attempt"] > 1]
|
| 303 |
+
|
| 304 |
+
# Calculate improvement rate
|
| 305 |
+
improvement_rate = 0.0
|
| 306 |
+
if len(self.current_question_memory) > 1:
|
| 307 |
+
first_score = self.current_question_memory[0]["score"]
|
| 308 |
+
last_score = self.current_question_memory[-1]["score"]
|
| 309 |
+
improvement_rate = last_score - first_score
|
| 310 |
+
|
| 311 |
+
return {
|
| 312 |
+
"total_attempts": len(self.memory),
|
| 313 |
+
"average_score": sum(scores) / len(scores),
|
| 314 |
+
"improvement_rate": improvement_rate,
|
| 315 |
+
"total_retries": len(retries)
|
| 316 |
+
}
|
| 317 |
+
|
| 318 |
+
def get_memory(self) -> List[Dict[str, Any]]:
|
| 319 |
+
"""
|
| 320 |
+
Get complete memory
|
| 321 |
+
|
| 322 |
+
Returns:
|
| 323 |
+
List of all stored attempts
|
| 324 |
+
"""
|
| 325 |
+
return self.memory
|
| 326 |
+
|
| 327 |
+
def get_current_question_memory(self) -> List[Dict[str, Any]]:
|
| 328 |
+
"""
|
| 329 |
+
Get memory for current question only
|
| 330 |
+
|
| 331 |
+
Returns:
|
| 332 |
+
List of attempts for current question
|
| 333 |
+
"""
|
| 334 |
+
return self.current_question_memory
|
backend/app/dataset.json
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"question": "What is a binary search and what is its time complexity?",
|
| 4 |
+
"keywords": ["sorted array", "divide", "logarithmic", "O(log n)", "middle element", "halving"],
|
| 5 |
+
"difficulty": "easy"
|
| 6 |
+
},
|
| 7 |
+
{
|
| 8 |
+
"question": "Explain the difference between a process and a thread.",
|
| 9 |
+
"keywords": ["process", "thread", "memory space", "lightweight", "shared memory", "independent"],
|
| 10 |
+
"difficulty": "medium"
|
| 11 |
+
},
|
| 12 |
+
{
|
| 13 |
+
"question": "What is database normalization and why is it important?",
|
| 14 |
+
"keywords": ["redundancy", "normal forms", "dependency", "data integrity", "anomalies"],
|
| 15 |
+
"difficulty": "medium"
|
| 16 |
+
},
|
| 17 |
+
{
|
| 18 |
+
"question": "Explain the four pillars of Object-Oriented Programming.",
|
| 19 |
+
"keywords": ["encapsulation", "abstraction", "inheritance", "polymorphism"],
|
| 20 |
+
"difficulty": "easy"
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"question": "What is a deadlock in operating systems and how can it be prevented?",
|
| 24 |
+
"keywords": ["mutual exclusion", "hold and wait", "circular wait", "prevention", "avoidance", "resources"],
|
| 25 |
+
"difficulty": "hard"
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"question": "Explain how a hashmap works internally.",
|
| 29 |
+
"keywords": ["hash function", "buckets", "collision", "linked list", "O(1)", "array"],
|
| 30 |
+
"difficulty": "medium"
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"question": "What is the difference between SQL and NoSQL databases?",
|
| 34 |
+
"keywords": ["structured", "schema", "scalability", "ACID", "flexible", "horizontal scaling"],
|
| 35 |
+
"difficulty": "easy"
|
| 36 |
+
},
|
| 37 |
+
{
|
| 38 |
+
"question": "Describe the SOLID principles in software design.",
|
| 39 |
+
"keywords": ["single responsibility", "open-closed", "liskov substitution", "interface segregation", "dependency inversion"],
|
| 40 |
+
"difficulty": "hard"
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
"question": "What is the difference between stack and heap memory?",
|
| 44 |
+
"keywords": ["static", "dynamic", "LIFO", "allocation", "scope", "lifetime"],
|
| 45 |
+
"difficulty": "medium"
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"question": "Explain what a RESTful API is and its key constraints.",
|
| 49 |
+
"keywords": ["stateless", "client-server", "cacheable", "uniform interface", "HTTP methods", "resources"],
|
| 50 |
+
"difficulty": "medium"
|
| 51 |
+
},
|
| 52 |
+
{
|
| 53 |
+
"question": "What is Big O notation and why is it important?",
|
| 54 |
+
"keywords": ["time complexity", "space complexity", "worst case", "algorithm efficiency", "growth rate"],
|
| 55 |
+
"difficulty": "easy"
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"question": "Explain the CAP theorem in distributed systems.",
|
| 59 |
+
"keywords": ["consistency", "availability", "partition tolerance", "distributed", "trade-off"],
|
| 60 |
+
"difficulty": "hard"
|
| 61 |
+
}
|
| 62 |
+
]
|
backend/app/environment.py
ADDED
|
@@ -0,0 +1,281 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
AI Interview Preparation Environment
|
| 3 |
+
OpenAI Gym-style RL Environment for Interview Practice
|
| 4 |
+
|
| 5 |
+
This module implements the core environment following the standard RL interface:
|
| 6 |
+
- reset() -> returns initial state (question)
|
| 7 |
+
- step(action) -> executes action and returns (reward, state, done, info)
|
| 8 |
+
|
| 9 |
+
Features:
|
| 10 |
+
- Random question selection from dataset
|
| 11 |
+
- Episode history tracking
|
| 12 |
+
- Configurable retry thresholds
|
| 13 |
+
- Support for multi-turn episodes (extensible)
|
| 14 |
+
"""
|
| 15 |
+
|
| 16 |
+
import json
|
| 17 |
+
import random
|
| 18 |
+
from typing import Dict, Any, List
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class InterviewEnv:
|
| 23 |
+
"""
|
| 24 |
+
Interview Preparation Environment
|
| 25 |
+
|
| 26 |
+
This environment simulates a technical interview scenario where:
|
| 27 |
+
- State: Interview question with metadata (keywords, difficulty)
|
| 28 |
+
- Action: Candidate's answer (text string)
|
| 29 |
+
- Reward: Score-based reward computed by evaluator
|
| 30 |
+
|
| 31 |
+
Design Decisions:
|
| 32 |
+
- Single question per episode (can be extended to multi-question)
|
| 33 |
+
- Stateful: maintains current question and episode history
|
| 34 |
+
- Delegated evaluation: uses injected evaluator for scoring
|
| 35 |
+
"""
|
| 36 |
+
|
| 37 |
+
def __init__(self, questions_path: str = None, evaluator=None):
|
| 38 |
+
"""
|
| 39 |
+
Initialize the interview environment
|
| 40 |
+
|
| 41 |
+
Args:
|
| 42 |
+
questions_path: Path to questions.json (defaults to ../data/questions.json)
|
| 43 |
+
evaluator: Evaluator instance for scoring answers (must implement evaluate())
|
| 44 |
+
"""
|
| 45 |
+
# Set default path if not provided
|
| 46 |
+
if questions_path is None:
|
| 47 |
+
questions_path = Path(__file__).parent.parent / "data" / "questions.json"
|
| 48 |
+
|
| 49 |
+
self.questions_path = questions_path
|
| 50 |
+
self.evaluator = evaluator
|
| 51 |
+
self.questions = self._load_questions()
|
| 52 |
+
|
| 53 |
+
# State tracking
|
| 54 |
+
self.current_question = None
|
| 55 |
+
self.episode_history = [] # Stores all attempts in current episode
|
| 56 |
+
self.retry_count = 0
|
| 57 |
+
self.max_retries = 3 # Configurable retry limit
|
| 58 |
+
|
| 59 |
+
def _load_questions(self) -> List[Dict[str, Any]]:
|
| 60 |
+
"""
|
| 61 |
+
Load questions from JSON dataset
|
| 62 |
+
|
| 63 |
+
Returns:
|
| 64 |
+
List of question dictionaries with structure:
|
| 65 |
+
{
|
| 66 |
+
"question": str,
|
| 67 |
+
"keywords": List[str],
|
| 68 |
+
"difficulty": str
|
| 69 |
+
}
|
| 70 |
+
"""
|
| 71 |
+
try:
|
| 72 |
+
with open(self.questions_path, 'r') as f:
|
| 73 |
+
questions = json.load(f)
|
| 74 |
+
print(f"✅ Loaded {len(questions)} questions from {self.questions_path}")
|
| 75 |
+
return questions
|
| 76 |
+
except FileNotFoundError:
|
| 77 |
+
print(f"❌ Questions file not found: {self.questions_path}")
|
| 78 |
+
return []
|
| 79 |
+
except json.JSONDecodeError:
|
| 80 |
+
print(f"❌ Invalid JSON in questions file")
|
| 81 |
+
return []
|
| 82 |
+
|
| 83 |
+
def reset(self) -> Dict[str, Any]:
|
| 84 |
+
"""
|
| 85 |
+
Reset environment to initial state
|
| 86 |
+
|
| 87 |
+
Selects a new random question and clears episode history.
|
| 88 |
+
This follows the Gym API convention.
|
| 89 |
+
|
| 90 |
+
Returns:
|
| 91 |
+
state: Dictionary containing:
|
| 92 |
+
- question: The interview question text
|
| 93 |
+
- keywords: Expected keywords for evaluation
|
| 94 |
+
- difficulty: Question difficulty level
|
| 95 |
+
- attempt: Current attempt number (always 0 after reset)
|
| 96 |
+
"""
|
| 97 |
+
# Select random question from dataset
|
| 98 |
+
if not self.questions:
|
| 99 |
+
raise ValueError("No questions loaded. Check questions.json")
|
| 100 |
+
|
| 101 |
+
self.current_question = random.choice(self.questions)
|
| 102 |
+
self.episode_history = []
|
| 103 |
+
self.retry_count = 0
|
| 104 |
+
|
| 105 |
+
# Return state representation
|
| 106 |
+
state = {
|
| 107 |
+
"question": self.current_question["question"],
|
| 108 |
+
"keywords": self.current_question["keywords"],
|
| 109 |
+
"difficulty": self.current_question["difficulty"],
|
| 110 |
+
"attempt": 0
|
| 111 |
+
}
|
| 112 |
+
|
| 113 |
+
print(f"\n{'='*60}")
|
| 114 |
+
print(f"📝 NEW QUESTION [{state['difficulty'].upper()}]")
|
| 115 |
+
print(f"{'='*60}")
|
| 116 |
+
print(f"{state['question']}")
|
| 117 |
+
print(f"{'='*60}\n")
|
| 118 |
+
|
| 119 |
+
return state
|
| 120 |
+
|
| 121 |
+
def step(self, action: str) -> Dict[str, Any]:
|
| 122 |
+
"""
|
| 123 |
+
Execute one environment step
|
| 124 |
+
|
| 125 |
+
Takes the candidate's answer, evaluates it, and returns feedback.
|
| 126 |
+
This is the core of the RL loop.
|
| 127 |
+
|
| 128 |
+
Args:
|
| 129 |
+
action: The candidate's answer (string)
|
| 130 |
+
|
| 131 |
+
Returns:
|
| 132 |
+
Dictionary containing:
|
| 133 |
+
- reward: Integer reward signal (+10, +5, 0, -5)
|
| 134 |
+
- score: Float score (0.0 to 1.0)
|
| 135 |
+
- feedback: Structured feedback message
|
| 136 |
+
- matched_keywords: List of correctly mentioned keywords
|
| 137 |
+
- missing_keywords: List of missed keywords
|
| 138 |
+
- done: Boolean indicating if episode is complete
|
| 139 |
+
- attempt: Current attempt number
|
| 140 |
+
|
| 141 |
+
Raises:
|
| 142 |
+
ValueError: If environment not initialized (call reset() first)
|
| 143 |
+
"""
|
| 144 |
+
# Validation
|
| 145 |
+
if self.current_question is None:
|
| 146 |
+
raise ValueError("Environment not initialized. Call reset() first.")
|
| 147 |
+
|
| 148 |
+
if self.evaluator is None:
|
| 149 |
+
raise ValueError("Evaluator not set. Pass evaluator to constructor.")
|
| 150 |
+
|
| 151 |
+
# Evaluate the answer using injected evaluator
|
| 152 |
+
evaluation = self.evaluator.evaluate(
|
| 153 |
+
answer=action,
|
| 154 |
+
keywords=self.current_question["keywords"]
|
| 155 |
+
)
|
| 156 |
+
|
| 157 |
+
# Extract evaluation results
|
| 158 |
+
score = evaluation["score"]
|
| 159 |
+
feedback = evaluation["feedback"]
|
| 160 |
+
matched_keywords = evaluation["matched_keywords"]
|
| 161 |
+
missing_keywords = evaluation["missing_keywords"]
|
| 162 |
+
|
| 163 |
+
# Compute reward signal
|
| 164 |
+
reward = self.evaluator.compute_reward(score)
|
| 165 |
+
|
| 166 |
+
# Increment retry counter
|
| 167 |
+
self.retry_count += 1
|
| 168 |
+
|
| 169 |
+
# Store in episode history for memory/analysis
|
| 170 |
+
self.episode_history.append({
|
| 171 |
+
"attempt": self.retry_count,
|
| 172 |
+
"answer": action,
|
| 173 |
+
"score": score,
|
| 174 |
+
"reward": reward,
|
| 175 |
+
"matched_keywords": matched_keywords,
|
| 176 |
+
"missing_keywords": missing_keywords,
|
| 177 |
+
"feedback": feedback
|
| 178 |
+
})
|
| 179 |
+
|
| 180 |
+
# Determine if episode is done
|
| 181 |
+
# Done if: high score OR max retries reached
|
| 182 |
+
done = (reward >= 0.9) or (self.retry_count >= self.max_retries)
|
| 183 |
+
|
| 184 |
+
# Return full step information
|
| 185 |
+
return {
|
| 186 |
+
"reward": reward,
|
| 187 |
+
"score": score,
|
| 188 |
+
"feedback": feedback,
|
| 189 |
+
"matched_keywords": matched_keywords,
|
| 190 |
+
"missing_keywords": missing_keywords,
|
| 191 |
+
"done": done,
|
| 192 |
+
"attempt": self.retry_count,
|
| 193 |
+
"question": self.current_question["question"]
|
| 194 |
+
}
|
| 195 |
+
|
| 196 |
+
def should_retry(self, reward: float) -> bool:
|
| 197 |
+
"""
|
| 198 |
+
Determine if agent should attempt to improve answer
|
| 199 |
+
|
| 200 |
+
Args:
|
| 201 |
+
reward: Current reward value (0.0 to 1.0)
|
| 202 |
+
|
| 203 |
+
Returns:
|
| 204 |
+
True if agent should retry (low reward and retries available)
|
| 205 |
+
"""
|
| 206 |
+
can_retry = self.retry_count < self.max_retries
|
| 207 |
+
needs_retry = reward < 0.7 # Threshold for improvement
|
| 208 |
+
|
| 209 |
+
return can_retry and needs_retry
|
| 210 |
+
|
| 211 |
+
def get_history(self) -> List[Dict[str, Any]]:
|
| 212 |
+
"""
|
| 213 |
+
Get complete episode history
|
| 214 |
+
|
| 215 |
+
Returns:
|
| 216 |
+
List of all attempts in current episode with full details
|
| 217 |
+
"""
|
| 218 |
+
return self.episode_history
|
| 219 |
+
|
| 220 |
+
def get_stats(self) -> Dict[str, Any]:
|
| 221 |
+
"""
|
| 222 |
+
Get episode statistics
|
| 223 |
+
|
| 224 |
+
Returns:
|
| 225 |
+
Dictionary with:
|
| 226 |
+
- total_attempts: Number of attempts made
|
| 227 |
+
- best_score: Highest score achieved
|
| 228 |
+
- improvement: Score improvement from first to last
|
| 229 |
+
- final_reward: Final reward value
|
| 230 |
+
"""
|
| 231 |
+
if not self.episode_history:
|
| 232 |
+
return {
|
| 233 |
+
"total_attempts": 0,
|
| 234 |
+
"best_score": 0.0,
|
| 235 |
+
"improvement": 0.0,
|
| 236 |
+
"final_reward": 0
|
| 237 |
+
}
|
| 238 |
+
|
| 239 |
+
scores = [h["score"] for h in self.episode_history]
|
| 240 |
+
rewards = [h["reward"] for h in self.episode_history]
|
| 241 |
+
|
| 242 |
+
return {
|
| 243 |
+
"total_attempts": len(self.episode_history),
|
| 244 |
+
"best_score": max(scores),
|
| 245 |
+
"improvement": scores[-1] - scores[0] if len(scores) > 1 else 0.0,
|
| 246 |
+
"final_reward": rewards[-1]
|
| 247 |
+
}
|
| 248 |
+
|
| 249 |
+
def set_max_retries(self, max_retries: int):
|
| 250 |
+
"""
|
| 251 |
+
Configure maximum retry attempts
|
| 252 |
+
|
| 253 |
+
Args:
|
| 254 |
+
max_retries: Maximum number of attempts allowed
|
| 255 |
+
"""
|
| 256 |
+
self.max_retries = max_retries
|
| 257 |
+
|
| 258 |
+
def state(self) -> Dict[str, Any]:
|
| 259 |
+
"""
|
| 260 |
+
Get current environment state
|
| 261 |
+
|
| 262 |
+
OpenEnv-compliant state getter.
|
| 263 |
+
Returns current question and attempt information.
|
| 264 |
+
|
| 265 |
+
Returns:
|
| 266 |
+
Dictionary containing:
|
| 267 |
+
- question: Current question text
|
| 268 |
+
- difficulty: Question difficulty level
|
| 269 |
+
- attempt: Current attempt number
|
| 270 |
+
|
| 271 |
+
Raises:
|
| 272 |
+
ValueError: If environment not initialized (call reset() first)
|
| 273 |
+
"""
|
| 274 |
+
if self.current_question is None:
|
| 275 |
+
raise ValueError("Environment not initialized. Call reset() first.")
|
| 276 |
+
|
| 277 |
+
return {
|
| 278 |
+
"question": self.current_question["question"],
|
| 279 |
+
"difficulty": self.current_question["difficulty"],
|
| 280 |
+
"attempt": self.retry_count
|
| 281 |
+
}
|
backend/app/evaluator.py
ADDED
|
@@ -0,0 +1,310 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Advanced Evaluator with Hybrid Scoring System
|
| 3 |
+
Evaluates interview answers using multiple metrics
|
| 4 |
+
|
| 5 |
+
Features:
|
| 6 |
+
- Keyword-based scoring (primary)
|
| 7 |
+
- AI-based scoring (placeholder/mock - extensible)
|
| 8 |
+
- Structured feedback generation
|
| 9 |
+
- Reward computation for RL
|
| 10 |
+
- Text normalization for robust matching
|
| 11 |
+
|
| 12 |
+
Design Philosophy:
|
| 13 |
+
- Modular: Easy to swap keyword logic or add LLM judge
|
| 14 |
+
- Extensible: Hybrid scoring allows future AI integration
|
| 15 |
+
- Explainable: Returns detailed feedback and matched/missing keywords
|
| 16 |
+
"""
|
| 17 |
+
|
| 18 |
+
import re
|
| 19 |
+
from typing import Dict, List, Any, Tuple
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class Evaluator:
|
| 23 |
+
"""
|
| 24 |
+
Hybrid Answer Evaluator
|
| 25 |
+
|
| 26 |
+
Evaluates answers using:
|
| 27 |
+
1. Keyword matching (primary metric)
|
| 28 |
+
2. AI scoring (placeholder for LLM-as-judge)
|
| 29 |
+
|
| 30 |
+
Final score = 0.7 * keyword_score + 0.3 * ai_score
|
| 31 |
+
|
| 32 |
+
This allows:
|
| 33 |
+
- Fast, deterministic evaluation (keywords)
|
| 34 |
+
- Future semantic understanding (AI judge)
|
| 35 |
+
- Explainable results (specific missing concepts)
|
| 36 |
+
"""
|
| 37 |
+
|
| 38 |
+
def __init__(self, keyword_weight: float = 0.8, ai_weight: float = 0.2):
|
| 39 |
+
"""
|
| 40 |
+
Initialize evaluator with scoring weights
|
| 41 |
+
|
| 42 |
+
Args:
|
| 43 |
+
keyword_weight: Weight for keyword-based score (default 0.8)
|
| 44 |
+
ai_weight: Weight for length-based score (default 0.2)
|
| 45 |
+
|
| 46 |
+
Weights should sum to 1.0 for normalized scoring
|
| 47 |
+
"""
|
| 48 |
+
self.keyword_weight = keyword_weight
|
| 49 |
+
self.ai_weight = ai_weight
|
| 50 |
+
|
| 51 |
+
# Validate weights
|
| 52 |
+
if not abs((keyword_weight + ai_weight) - 1.0) < 0.01:
|
| 53 |
+
print(f"⚠️ Warning: Weights don't sum to 1.0 ({keyword_weight + ai_weight})")
|
| 54 |
+
|
| 55 |
+
def normalize_text(self, text: str) -> str:
|
| 56 |
+
"""
|
| 57 |
+
Normalize text for robust keyword matching
|
| 58 |
+
|
| 59 |
+
Steps:
|
| 60 |
+
1. Convert to lowercase
|
| 61 |
+
2. Remove special characters (keep alphanumeric, spaces, hyphens, parens)
|
| 62 |
+
3. Collapse multiple spaces
|
| 63 |
+
|
| 64 |
+
Args:
|
| 65 |
+
text: Raw text string
|
| 66 |
+
|
| 67 |
+
Returns:
|
| 68 |
+
Normalized text string
|
| 69 |
+
|
| 70 |
+
Examples:
|
| 71 |
+
"O(log n)" -> "o(log n)"
|
| 72 |
+
"Process!!!" -> "process"
|
| 73 |
+
"REST API" -> "rest api"
|
| 74 |
+
"""
|
| 75 |
+
text = text.lower()
|
| 76 |
+
|
| 77 |
+
# Keep alphanumeric, spaces, hyphens, and parentheses
|
| 78 |
+
# This preserves "O(log n)", "client-server", etc.
|
| 79 |
+
text = re.sub(r'[^a-z0-9\s\-\(\)]', ' ', text)
|
| 80 |
+
|
| 81 |
+
# Collapse multiple spaces
|
| 82 |
+
text = ' '.join(text.split())
|
| 83 |
+
|
| 84 |
+
return text
|
| 85 |
+
|
| 86 |
+
def evaluate_keywords(self, answer: str, keywords: List[str]) -> Tuple[float, List[str], List[str]]:
|
| 87 |
+
"""
|
| 88 |
+
Evaluate answer based on keyword coverage
|
| 89 |
+
|
| 90 |
+
This is the primary evaluation metric.
|
| 91 |
+
|
| 92 |
+
Args:
|
| 93 |
+
answer: Candidate's answer text
|
| 94 |
+
keywords: List of expected keywords/phrases
|
| 95 |
+
|
| 96 |
+
Returns:
|
| 97 |
+
Tuple of (score, matched_keywords, missing_keywords)
|
| 98 |
+
- score: Proportion of keywords found (0.0 to 1.0)
|
| 99 |
+
- matched_keywords: Keywords found in answer
|
| 100 |
+
- missing_keywords: Keywords not found in answer
|
| 101 |
+
|
| 102 |
+
Algorithm:
|
| 103 |
+
1. Normalize both answer and keywords
|
| 104 |
+
2. Check if each keyword appears in answer (substring match)
|
| 105 |
+
3. Calculate score as ratio: matched / total
|
| 106 |
+
"""
|
| 107 |
+
if not answer or not answer.strip():
|
| 108 |
+
return 0.0, [], keywords
|
| 109 |
+
|
| 110 |
+
normalized_answer = self.normalize_text(answer)
|
| 111 |
+
matched_keywords = []
|
| 112 |
+
missing_keywords = []
|
| 113 |
+
|
| 114 |
+
# Check each keyword
|
| 115 |
+
for keyword in keywords:
|
| 116 |
+
normalized_keyword = self.normalize_text(keyword)
|
| 117 |
+
|
| 118 |
+
# Use substring matching for flexibility
|
| 119 |
+
# "O(log n)" matches "time complexity is O(log n)"
|
| 120 |
+
if normalized_keyword in normalized_answer:
|
| 121 |
+
matched_keywords.append(keyword)
|
| 122 |
+
else:
|
| 123 |
+
missing_keywords.append(keyword)
|
| 124 |
+
|
| 125 |
+
# Calculate proportional score
|
| 126 |
+
if len(keywords) == 0:
|
| 127 |
+
keyword_score = 0.5 # Neutral score if no keywords defined
|
| 128 |
+
else:
|
| 129 |
+
keyword_score = len(matched_keywords) / len(keywords)
|
| 130 |
+
|
| 131 |
+
return keyword_score, matched_keywords, missing_keywords
|
| 132 |
+
|
| 133 |
+
def evaluate_ai(self, answer: str, question: str = None) -> float:
|
| 134 |
+
"""
|
| 135 |
+
Length-based evaluation (deterministic)
|
| 136 |
+
|
| 137 |
+
DETERMINISTIC implementation for OpenEnv compliance.
|
| 138 |
+
Scores based on answer length relative to ideal length.
|
| 139 |
+
|
| 140 |
+
Current implementation:
|
| 141 |
+
- Deterministic scoring based on answer length
|
| 142 |
+
- No randomness - same input always gives same output
|
| 143 |
+
- Considers completeness via length heuristic
|
| 144 |
+
|
| 145 |
+
Args:
|
| 146 |
+
answer: Candidate's answer
|
| 147 |
+
question: Original question (for context)
|
| 148 |
+
|
| 149 |
+
Returns:
|
| 150 |
+
Length score between 0.0 and 1.0
|
| 151 |
+
"""
|
| 152 |
+
# DETERMINISTIC IMPLEMENTATION for OpenEnv compliance
|
| 153 |
+
|
| 154 |
+
# Ideal answer length: 100-300 characters
|
| 155 |
+
# Score based on how close to ideal range
|
| 156 |
+
if not answer or len(answer.strip()) < 10:
|
| 157 |
+
return 0.2
|
| 158 |
+
|
| 159 |
+
answer_length = len(answer.strip())
|
| 160 |
+
|
| 161 |
+
# Optimal range: 100-300 characters
|
| 162 |
+
if 100 <= answer_length <= 300:
|
| 163 |
+
return 1.0
|
| 164 |
+
elif answer_length < 100:
|
| 165 |
+
# Scale from 0.5 to 1.0 as length approaches 100
|
| 166 |
+
return 0.5 + 0.5 * (answer_length / 100.0)
|
| 167 |
+
else:
|
| 168 |
+
# Penalize excessively long answers (diminishing returns)
|
| 169 |
+
excess = answer_length - 300
|
| 170 |
+
penalty = min(excess / 500.0, 0.3) # Max 0.3 penalty
|
| 171 |
+
return max(0.7, 1.0 - penalty)
|
| 172 |
+
|
| 173 |
+
|
| 174 |
+
|
| 175 |
+
def evaluate(self, answer: str, keywords: List[str], question: str = None) -> Dict[str, Any]:
|
| 176 |
+
"""
|
| 177 |
+
Complete hybrid evaluation
|
| 178 |
+
|
| 179 |
+
Combines keyword-based and AI-based scoring with configurable weights.
|
| 180 |
+
|
| 181 |
+
Args:
|
| 182 |
+
answer: Candidate's answer
|
| 183 |
+
keywords: Expected keywords for this question
|
| 184 |
+
question: Original question (optional, for AI judge)
|
| 185 |
+
|
| 186 |
+
Returns:
|
| 187 |
+
Dictionary containing:
|
| 188 |
+
- score: Final hybrid score (0.0 to 1.0)
|
| 189 |
+
- keyword_score: Keyword-based score
|
| 190 |
+
- ai_score: AI-based score
|
| 191 |
+
- matched_keywords: Keywords found
|
| 192 |
+
- missing_keywords: Keywords not found
|
| 193 |
+
- feedback: Human-readable feedback string
|
| 194 |
+
"""
|
| 195 |
+
# Evaluate using both methods
|
| 196 |
+
keyword_score, matched, missing = self.evaluate_keywords(answer, keywords)
|
| 197 |
+
ai_score = self.evaluate_ai(answer, question)
|
| 198 |
+
|
| 199 |
+
# Compute weighted hybrid score
|
| 200 |
+
final_score = (self.keyword_weight * keyword_score) + (self.ai_weight * ai_score)
|
| 201 |
+
|
| 202 |
+
# Generate structured feedback
|
| 203 |
+
feedback = self._generate_feedback(
|
| 204 |
+
score=final_score,
|
| 205 |
+
keyword_score=keyword_score,
|
| 206 |
+
matched=matched,
|
| 207 |
+
missing=missing
|
| 208 |
+
)
|
| 209 |
+
|
| 210 |
+
return {
|
| 211 |
+
"score": final_score,
|
| 212 |
+
"keyword_score": keyword_score,
|
| 213 |
+
"ai_score": ai_score,
|
| 214 |
+
"matched_keywords": matched,
|
| 215 |
+
"missing_keywords": missing,
|
| 216 |
+
"feedback": feedback
|
| 217 |
+
}
|
| 218 |
+
|
| 219 |
+
def _generate_feedback(
|
| 220 |
+
self,
|
| 221 |
+
score: float,
|
| 222 |
+
keyword_score: float,
|
| 223 |
+
matched: List[str],
|
| 224 |
+
missing: List[str]
|
| 225 |
+
) -> str:
|
| 226 |
+
"""
|
| 227 |
+
Generate structured, actionable feedback
|
| 228 |
+
|
| 229 |
+
Feedback is designed to be:
|
| 230 |
+
- Clear and specific
|
| 231 |
+
- Actionable (tells what to add)
|
| 232 |
+
- Encouraging (positive framing)
|
| 233 |
+
- Machine-parseable (for agent improvement)
|
| 234 |
+
|
| 235 |
+
Args:
|
| 236 |
+
score: Final score
|
| 237 |
+
keyword_score: Keyword coverage score
|
| 238 |
+
matched: Matched keywords
|
| 239 |
+
missing: Missing keywords
|
| 240 |
+
|
| 241 |
+
Returns:
|
| 242 |
+
Formatted feedback string
|
| 243 |
+
"""
|
| 244 |
+
total_keywords = len(matched) + len(missing)
|
| 245 |
+
|
| 246 |
+
# Score-based feedback tier
|
| 247 |
+
if score >= 0.8:
|
| 248 |
+
base = f"✅ Excellent answer! "
|
| 249 |
+
base += f"Covered {len(matched)}/{total_keywords} key concepts."
|
| 250 |
+
if matched:
|
| 251 |
+
base += f"\n ✓ Mentioned: {', '.join(matched)}"
|
| 252 |
+
return base
|
| 253 |
+
|
| 254 |
+
elif score >= 0.5:
|
| 255 |
+
base = f"👍 Good answer. "
|
| 256 |
+
base += f"Covered {len(matched)}/{total_keywords} concepts."
|
| 257 |
+
if matched:
|
| 258 |
+
base += f"\n ✓ Covered: {', '.join(matched)}"
|
| 259 |
+
if missing:
|
| 260 |
+
# Show up to 3 missing keywords for actionable feedback
|
| 261 |
+
base += f"\n ⚠ Consider adding: {', '.join(missing[:3])}"
|
| 262 |
+
return base
|
| 263 |
+
|
| 264 |
+
elif score >= 0.3:
|
| 265 |
+
base = f"⚠️ Partial answer. "
|
| 266 |
+
base += f"Only {len(matched)}/{total_keywords} concepts covered."
|
| 267 |
+
if matched:
|
| 268 |
+
base += f"\n ✓ Included: {', '.join(matched)}"
|
| 269 |
+
if missing:
|
| 270 |
+
base += f"\n ✗ Missing: {', '.join(missing)}"
|
| 271 |
+
return base
|
| 272 |
+
|
| 273 |
+
else:
|
| 274 |
+
base = f"❌ Weak answer. "
|
| 275 |
+
base += f"Missing {len(missing)}/{total_keywords} critical concepts."
|
| 276 |
+
if missing:
|
| 277 |
+
base += f"\n ✗ Must include: {', '.join(missing)}"
|
| 278 |
+
base += "\n 💡 Tip: Cover fundamental concepts first"
|
| 279 |
+
return base
|
| 280 |
+
|
| 281 |
+
def compute_reward(self, score: float) -> float:
|
| 282 |
+
"""
|
| 283 |
+
Convert score to RL reward signal
|
| 284 |
+
|
| 285 |
+
OpenEnv-compliant reward function:
|
| 286 |
+
- Reward is IDENTICAL to score (0.0 to 1.0 range)
|
| 287 |
+
- Deterministic: same score always gives same reward
|
| 288 |
+
- Continuous: allows fine-grained learning signals
|
| 289 |
+
|
| 290 |
+
Args:
|
| 291 |
+
score: Evaluation score (0.0 to 1.0)
|
| 292 |
+
|
| 293 |
+
Returns:
|
| 294 |
+
Float reward value (0.0 to 1.0)
|
| 295 |
+
"""
|
| 296 |
+
# OpenEnv compliance: reward = score (0.0 to 1.0)
|
| 297 |
+
return float(score)
|
| 298 |
+
|
| 299 |
+
def set_weights(self, keyword_weight: float, ai_weight: float):
|
| 300 |
+
"""
|
| 301 |
+
Update scoring weights
|
| 302 |
+
|
| 303 |
+
Allows runtime adjustment of hybrid scoring balance.
|
| 304 |
+
|
| 305 |
+
Args:
|
| 306 |
+
keyword_weight: New keyword weight
|
| 307 |
+
ai_weight: New AI weight
|
| 308 |
+
"""
|
| 309 |
+
self.keyword_weight = keyword_weight
|
| 310 |
+
self.ai_weight = ai_weight
|
backend/main.py
ADDED
|
@@ -0,0 +1,332 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
FastAPI Backend for AI Interview Preparation RL Environment
|
| 3 |
+
OpenEnv-compliant API implementation
|
| 4 |
+
"""
|
| 5 |
+
|
| 6 |
+
import os
|
| 7 |
+
from fastapi import FastAPI, HTTPException
|
| 8 |
+
from fastapi.middleware.cors import CORSMiddleware
|
| 9 |
+
from pydantic import BaseModel
|
| 10 |
+
from typing import Optional
|
| 11 |
+
from pathlib import Path
|
| 12 |
+
from dotenv import load_dotenv
|
| 13 |
+
|
| 14 |
+
# Load environment variables
|
| 15 |
+
load_dotenv()
|
| 16 |
+
|
| 17 |
+
print("🚀 Server starting...")
|
| 18 |
+
|
| 19 |
+
# Import from app package
|
| 20 |
+
from app.environment import InterviewEnv
|
| 21 |
+
from app.evaluator import Evaluator
|
| 22 |
+
from app.agent import InterviewAgent
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
# Initialize FastAPI app
|
| 26 |
+
app = FastAPI(title="AI Interview Prep RL Environment")
|
| 27 |
+
|
| 28 |
+
# CORS middleware for frontend
|
| 29 |
+
app.add_middleware(
|
| 30 |
+
CORSMiddleware,
|
| 31 |
+
allow_origins=["*"],
|
| 32 |
+
allow_credentials=True,
|
| 33 |
+
allow_methods=["*"],
|
| 34 |
+
allow_headers=["*"],
|
| 35 |
+
)
|
| 36 |
+
|
| 37 |
+
# Initialize components
|
| 38 |
+
evaluator = Evaluator()
|
| 39 |
+
|
| 40 |
+
# Resolve dataset path — works in both local (backend/) and Docker (/app/) contexts
|
| 41 |
+
questions_path = Path(__file__).resolve().parent / "app" / "dataset.json"
|
| 42 |
+
print(f"📂 Dataset path: {questions_path}")
|
| 43 |
+
print(f"📂 Dataset exists: {questions_path.exists()}")
|
| 44 |
+
|
| 45 |
+
env = InterviewEnv(questions_path=str(questions_path), evaluator=evaluator)
|
| 46 |
+
|
| 47 |
+
# Graceful agent init — don't crash server if API key is missing
|
| 48 |
+
try:
|
| 49 |
+
agent = InterviewAgent(mode="api") # Will use HF_TOKEN from .env
|
| 50 |
+
print(f"🤖 Agent initialized in mode: {agent.mode}")
|
| 51 |
+
except Exception as e:
|
| 52 |
+
print(f"⚠️ Agent init failed ({e}), using mock mode")
|
| 53 |
+
agent = InterviewAgent(mode="mock")
|
| 54 |
+
|
| 55 |
+
print("✅ All components initialized successfully")
|
| 56 |
+
|
| 57 |
+
# Global state
|
| 58 |
+
current_state = None
|
| 59 |
+
current_config = {
|
| 60 |
+
"api_key": None,
|
| 61 |
+
"model": "Qwen/Qwen3-Coder-Next:novita"
|
| 62 |
+
}
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
# Request/Response models
|
| 66 |
+
class StepRequest(BaseModel):
|
| 67 |
+
action: str
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
class AnswerRequest(BaseModel):
|
| 71 |
+
answer: str
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
class AutoRunRequest(BaseModel):
|
| 75 |
+
use_retry: bool = True
|
| 76 |
+
|
| 77 |
+
|
| 78 |
+
class ConfigRequest(BaseModel):
|
| 79 |
+
api_key: str
|
| 80 |
+
model: str = "Qwen/Qwen3-Coder-Next:novita"
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
@app.get("/")
|
| 84 |
+
def read_root():
|
| 85 |
+
"""Root endpoint"""
|
| 86 |
+
return {
|
| 87 |
+
"message": "AI Interview Preparation RL Environment (OpenEnv Compliant)",
|
| 88 |
+
"status": "running",
|
| 89 |
+
"endpoints": ["/reset", "/step", "/state", "/question", "/answer", "/auto-run", "/stats", "/config"]
|
| 90 |
+
}
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
# OpenEnv-compliant endpoints
|
| 94 |
+
|
| 95 |
+
@app.post("/reset")
|
| 96 |
+
def reset():
|
| 97 |
+
"""OpenEnv: Reset environment and get new question"""
|
| 98 |
+
global current_state
|
| 99 |
+
|
| 100 |
+
try:
|
| 101 |
+
current_state = env.reset()
|
| 102 |
+
# Return only the required fields for OpenEnv compliance
|
| 103 |
+
return {
|
| 104 |
+
"question": current_state["question"],
|
| 105 |
+
"difficulty": current_state["difficulty"],
|
| 106 |
+
"attempt": current_state["attempt"]
|
| 107 |
+
}
|
| 108 |
+
except Exception as e:
|
| 109 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 110 |
+
|
| 111 |
+
|
| 112 |
+
@app.post("/step")
|
| 113 |
+
def step(request: StepRequest):
|
| 114 |
+
"""OpenEnv: Submit action and get reward"""
|
| 115 |
+
global current_state
|
| 116 |
+
|
| 117 |
+
if current_state is None:
|
| 118 |
+
raise HTTPException(status_code=400, detail="Environment not initialized. Call /reset first")
|
| 119 |
+
|
| 120 |
+
try:
|
| 121 |
+
result = env.step(request.action)
|
| 122 |
+
|
| 123 |
+
# Get current state
|
| 124 |
+
state_data = env.state()
|
| 125 |
+
|
| 126 |
+
# Return OpenEnv-compliant response
|
| 127 |
+
return {
|
| 128 |
+
"reward": result["reward"],
|
| 129 |
+
"done": result["done"],
|
| 130 |
+
"state": state_data
|
| 131 |
+
}
|
| 132 |
+
except Exception as e:
|
| 133 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 134 |
+
|
| 135 |
+
|
| 136 |
+
@app.get("/state")
|
| 137 |
+
def get_state():
|
| 138 |
+
"""OpenEnv: Get current environment state"""
|
| 139 |
+
if current_state is None:
|
| 140 |
+
raise HTTPException(status_code=400, detail="Environment not initialized. Call /reset first")
|
| 141 |
+
|
| 142 |
+
try:
|
| 143 |
+
state_data = env.state()
|
| 144 |
+
return state_data
|
| 145 |
+
except Exception as e:
|
| 146 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 147 |
+
|
| 148 |
+
|
| 149 |
+
# Legacy endpoints (for backward compatibility with existing frontend)
|
| 150 |
+
|
| 151 |
+
@app.get("/question")
|
| 152 |
+
def get_question():
|
| 153 |
+
"""Get a new interview question (calls env.reset())"""
|
| 154 |
+
global current_state
|
| 155 |
+
|
| 156 |
+
try:
|
| 157 |
+
current_state = env.reset()
|
| 158 |
+
return {
|
| 159 |
+
"status": "success",
|
| 160 |
+
"state": current_state
|
| 161 |
+
}
|
| 162 |
+
except Exception as e:
|
| 163 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
@app.post("/answer")
|
| 167 |
+
def submit_answer(request: AnswerRequest):
|
| 168 |
+
"""Submit an answer and get evaluation (calls env.step())"""
|
| 169 |
+
global current_state
|
| 170 |
+
|
| 171 |
+
if current_state is None:
|
| 172 |
+
raise HTTPException(status_code=400, detail="No active question. Call /question first")
|
| 173 |
+
|
| 174 |
+
try:
|
| 175 |
+
result = env.step(request.answer)
|
| 176 |
+
return {
|
| 177 |
+
"status": "success",
|
| 178 |
+
"result": result
|
| 179 |
+
}
|
| 180 |
+
except Exception as e:
|
| 181 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 182 |
+
|
| 183 |
+
|
| 184 |
+
@app.post("/auto-run")
|
| 185 |
+
def auto_run(request: AutoRunRequest):
|
| 186 |
+
"""Automatic RL episode - agent generates and improves answer"""
|
| 187 |
+
global current_state
|
| 188 |
+
|
| 189 |
+
try:
|
| 190 |
+
# Reset environment
|
| 191 |
+
current_state = env.reset()
|
| 192 |
+
question = current_state["question"]
|
| 193 |
+
|
| 194 |
+
# Agent generates initial answer
|
| 195 |
+
answer = agent.generate_answer(question)
|
| 196 |
+
|
| 197 |
+
# Evaluate
|
| 198 |
+
result = env.step(answer)
|
| 199 |
+
|
| 200 |
+
episode_data = {
|
| 201 |
+
"question": question,
|
| 202 |
+
"difficulty": current_state["difficulty"],
|
| 203 |
+
"attempt_1": {
|
| 204 |
+
"answer": answer,
|
| 205 |
+
"score": result["score"],
|
| 206 |
+
"reward": result["reward"],
|
| 207 |
+
"feedback": result["feedback"]
|
| 208 |
+
}
|
| 209 |
+
}
|
| 210 |
+
|
| 211 |
+
# Retry logic if enabled and reward is low
|
| 212 |
+
if request.use_retry and env.should_retry(result["reward"]):
|
| 213 |
+
# Reset for retry
|
| 214 |
+
current_state = {
|
| 215 |
+
"question": question,
|
| 216 |
+
"keywords": current_state["keywords"],
|
| 217 |
+
"difficulty": current_state["difficulty"]
|
| 218 |
+
}
|
| 219 |
+
env.current_question = {
|
| 220 |
+
"question": question,
|
| 221 |
+
"keywords": current_state["keywords"],
|
| 222 |
+
"difficulty": current_state["difficulty"]
|
| 223 |
+
}
|
| 224 |
+
|
| 225 |
+
# Generate improved answer with feedback
|
| 226 |
+
improved_answer = agent.generate_answer(
|
| 227 |
+
question,
|
| 228 |
+
feedback=result["feedback"],
|
| 229 |
+
missing_keywords=result.get("missing_keywords", [])
|
| 230 |
+
)
|
| 231 |
+
|
| 232 |
+
# Evaluate again
|
| 233 |
+
retry_result = env.step(improved_answer)
|
| 234 |
+
|
| 235 |
+
episode_data["attempt_2"] = {
|
| 236 |
+
"answer": improved_answer,
|
| 237 |
+
"score": retry_result["score"],
|
| 238 |
+
"reward": retry_result["reward"],
|
| 239 |
+
"feedback": retry_result["feedback"]
|
| 240 |
+
}
|
| 241 |
+
|
| 242 |
+
episode_data["improvement"] = retry_result["score"] - result["score"]
|
| 243 |
+
|
| 244 |
+
return {
|
| 245 |
+
"status": "success",
|
| 246 |
+
"episode": episode_data
|
| 247 |
+
}
|
| 248 |
+
|
| 249 |
+
except Exception as e:
|
| 250 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 251 |
+
|
| 252 |
+
|
| 253 |
+
@app.post("/config")
|
| 254 |
+
def update_config(request: ConfigRequest):
|
| 255 |
+
"""Update API configuration (API key and model)"""
|
| 256 |
+
global agent, current_config
|
| 257 |
+
|
| 258 |
+
try:
|
| 259 |
+
# Update configuration
|
| 260 |
+
current_config["api_key"] = request.api_key
|
| 261 |
+
current_config["model"] = request.model
|
| 262 |
+
|
| 263 |
+
# Reinitialize agent with new configuration
|
| 264 |
+
agent = InterviewAgent(mode="api", api_key=request.api_key, model=request.model)
|
| 265 |
+
|
| 266 |
+
# Check if agent has proper client initialization
|
| 267 |
+
has_client = hasattr(agent, 'client') and agent.client is not None
|
| 268 |
+
|
| 269 |
+
return {
|
| 270 |
+
"status": "success",
|
| 271 |
+
"message": "Configuration updated successfully",
|
| 272 |
+
"config": {
|
| 273 |
+
"model": request.model,
|
| 274 |
+
"api_key_set": bool(request.api_key),
|
| 275 |
+
"client_initialized": has_client
|
| 276 |
+
}
|
| 277 |
+
}
|
| 278 |
+
except Exception as e:
|
| 279 |
+
raise HTTPException(status_code=500, detail=f"Configuration error: {str(e)}")
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
@app.get("/config")
|
| 283 |
+
def get_config():
|
| 284 |
+
"""Get current API configuration"""
|
| 285 |
+
has_client = hasattr(agent, 'client') and agent.client is not None
|
| 286 |
+
|
| 287 |
+
return {
|
| 288 |
+
"status": "success",
|
| 289 |
+
"config": {
|
| 290 |
+
"model": current_config["model"],
|
| 291 |
+
"api_key_set": bool(current_config["api_key"]),
|
| 292 |
+
"client_initialized": has_client
|
| 293 |
+
}
|
| 294 |
+
}
|
| 295 |
+
|
| 296 |
+
|
| 297 |
+
@app.get("/stats")
|
| 298 |
+
def get_stats():
|
| 299 |
+
"""Get episode statistics"""
|
| 300 |
+
try:
|
| 301 |
+
history = env.get_history()
|
| 302 |
+
if not history:
|
| 303 |
+
return {
|
| 304 |
+
"status": "success",
|
| 305 |
+
"stats": {
|
| 306 |
+
"total_attempts": 0,
|
| 307 |
+
"average_score": 0,
|
| 308 |
+
"average_reward": 0
|
| 309 |
+
}
|
| 310 |
+
}
|
| 311 |
+
|
| 312 |
+
avg_score = sum(h["score"] for h in history) / len(history)
|
| 313 |
+
avg_reward = sum(h["reward"] for h in history) / len(history)
|
| 314 |
+
|
| 315 |
+
return {
|
| 316 |
+
"status": "success",
|
| 317 |
+
"stats": {
|
| 318 |
+
"total_attempts": len(history),
|
| 319 |
+
"average_score": round(avg_score, 3),
|
| 320 |
+
"average_reward": round(avg_reward, 2),
|
| 321 |
+
"history": history
|
| 322 |
+
}
|
| 323 |
+
}
|
| 324 |
+
except Exception as e:
|
| 325 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 326 |
+
|
| 327 |
+
|
| 328 |
+
if __name__ == "__main__":
|
| 329 |
+
import uvicorn
|
| 330 |
+
port = int(os.getenv("PORT", "8000"))
|
| 331 |
+
print(f"🌐 Starting server on port {port}")
|
| 332 |
+
uvicorn.run(app, host="0.0.0.0", port=port)
|
backend/main_old.py
ADDED
|
@@ -0,0 +1,291 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
FastAPI Backend for AI Interview Preparation RL Environment
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
from fastapi import FastAPI, HTTPException
|
| 6 |
+
from fastapi.middleware.cors import CORSMiddleware
|
| 7 |
+
from pydantic import BaseModel
|
| 8 |
+
from typing import Optional
|
| 9 |
+
import sys
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
from dotenv import load_dotenv
|
| 12 |
+
|
| 13 |
+
# Load environment variables from .env file
|
| 14 |
+
load_dotenv()
|
| 15 |
+
|
| 16 |
+
# Add parent directory to path
|
| 17 |
+
sys.path.append(str(Path(__file__).parent))
|
| 18 |
+
|
| 19 |
+
from env.interview_env import InterviewEnv
|
| 20 |
+
from evaluator.evaluator import Evaluator
|
| 21 |
+
from agent.llm_agent import LLMAgent
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
# Initialize FastAPI app
|
| 25 |
+
app = FastAPI(title="AI Interview Prep RL Environment")
|
| 26 |
+
|
| 27 |
+
# CORS middleware for frontend
|
| 28 |
+
app.add_middleware(
|
| 29 |
+
CORSMiddleware,
|
| 30 |
+
allow_origins=["*"], # In production, specify exact origins
|
| 31 |
+
allow_credentials=True,
|
| 32 |
+
allow_methods=["*"],
|
| 33 |
+
allow_headers=["*"],
|
| 34 |
+
)
|
| 35 |
+
|
| 36 |
+
# Initialize components
|
| 37 |
+
evaluator = Evaluator()
|
| 38 |
+
env = InterviewEnv(evaluator=evaluator)
|
| 39 |
+
agent = LLMAgent()
|
| 40 |
+
|
| 41 |
+
# Global state (in production, use session management)
|
| 42 |
+
current_state = None
|
| 43 |
+
current_config = {
|
| 44 |
+
"api_key": None,
|
| 45 |
+
"model": "Qwen/Qwen3-Coder-Next:novita"
|
| 46 |
+
}
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
# Request/Response models
|
| 50 |
+
class AnswerRequest(BaseModel):
|
| 51 |
+
answer: str
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
class AutoRunRequest(BaseModel):
|
| 55 |
+
use_retry: bool = True
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
class ConfigRequest(BaseModel):
|
| 59 |
+
api_key: str
|
| 60 |
+
model: str = "Qwen/Qwen3-Coder-Next:novita"
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
@app.get("/")
|
| 64 |
+
def read_root():
|
| 65 |
+
"""Root endpoint"""
|
| 66 |
+
return {
|
| 67 |
+
"message": "AI Interview Preparation RL Environment",
|
| 68 |
+
"status": "running",
|
| 69 |
+
"endpoints": ["/question", "/answer", "/auto-run", "/stats", "/config"]
|
| 70 |
+
}
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
@app.get("/question")
|
| 74 |
+
def get_question():
|
| 75 |
+
"""
|
| 76 |
+
Get a new interview question (calls env.reset())
|
| 77 |
+
|
| 78 |
+
Returns:
|
| 79 |
+
{
|
| 80 |
+
"question": str,
|
| 81 |
+
"keywords": list,
|
| 82 |
+
"difficulty": str
|
| 83 |
+
}
|
| 84 |
+
"""
|
| 85 |
+
global current_state
|
| 86 |
+
|
| 87 |
+
try:
|
| 88 |
+
current_state = env.reset()
|
| 89 |
+
return {
|
| 90 |
+
"status": "success",
|
| 91 |
+
"state": current_state
|
| 92 |
+
}
|
| 93 |
+
except Exception as e:
|
| 94 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
@app.post("/answer")
|
| 98 |
+
def submit_answer(request: AnswerRequest):
|
| 99 |
+
"""
|
| 100 |
+
Submit an answer and get evaluation (calls env.step())
|
| 101 |
+
|
| 102 |
+
Args:
|
| 103 |
+
answer: The candidate's answer
|
| 104 |
+
|
| 105 |
+
Returns:
|
| 106 |
+
{
|
| 107 |
+
"reward": int,
|
| 108 |
+
"score": float,
|
| 109 |
+
"feedback": str,
|
| 110 |
+
"missing_keywords": list,
|
| 111 |
+
"done": bool
|
| 112 |
+
}
|
| 113 |
+
"""
|
| 114 |
+
global current_state
|
| 115 |
+
|
| 116 |
+
if current_state is None:
|
| 117 |
+
raise HTTPException(
|
| 118 |
+
status_code=400,
|
| 119 |
+
detail="No active question. Call /question first"
|
| 120 |
+
)
|
| 121 |
+
|
| 122 |
+
try:
|
| 123 |
+
result = env.step(request.answer)
|
| 124 |
+
return {
|
| 125 |
+
"status": "success",
|
| 126 |
+
"result": result
|
| 127 |
+
}
|
| 128 |
+
except Exception as e:
|
| 129 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 130 |
+
|
| 131 |
+
|
| 132 |
+
@app.post("/auto-run")
|
| 133 |
+
def auto_run(request: AutoRunRequest):
|
| 134 |
+
"""
|
| 135 |
+
Automatic RL loop: Get question → Agent generates answer → Evaluate → Optional retry
|
| 136 |
+
|
| 137 |
+
Args:
|
| 138 |
+
use_retry: Whether to retry on low scores
|
| 139 |
+
|
| 140 |
+
Returns:
|
| 141 |
+
Complete episode results with optional retry
|
| 142 |
+
"""
|
| 143 |
+
global current_state
|
| 144 |
+
|
| 145 |
+
try:
|
| 146 |
+
# Reset environment and get question
|
| 147 |
+
current_state = env.reset()
|
| 148 |
+
question = current_state["question"]
|
| 149 |
+
|
| 150 |
+
# Agent generates initial answer
|
| 151 |
+
answer = agent.generate_answer(question)
|
| 152 |
+
|
| 153 |
+
# Evaluate
|
| 154 |
+
result = env.step(answer)
|
| 155 |
+
|
| 156 |
+
episode_data = {
|
| 157 |
+
"question": question,
|
| 158 |
+
"difficulty": current_state["difficulty"],
|
| 159 |
+
"attempt_1": {
|
| 160 |
+
"answer": answer,
|
| 161 |
+
"score": result["score"],
|
| 162 |
+
"reward": result["reward"],
|
| 163 |
+
"feedback": result["feedback"]
|
| 164 |
+
}
|
| 165 |
+
}
|
| 166 |
+
|
| 167 |
+
# Retry logic if enabled and reward is low
|
| 168 |
+
if request.use_retry and env.should_retry(result["reward"]):
|
| 169 |
+
# Reset for retry
|
| 170 |
+
current_state = {
|
| 171 |
+
"question": question,
|
| 172 |
+
"keywords": current_state["keywords"],
|
| 173 |
+
"difficulty": current_state["difficulty"]
|
| 174 |
+
}
|
| 175 |
+
env.current_question = {
|
| 176 |
+
"question": question,
|
| 177 |
+
"keywords": current_state["keywords"],
|
| 178 |
+
"difficulty": current_state["difficulty"]
|
| 179 |
+
}
|
| 180 |
+
|
| 181 |
+
# Generate improved answer with feedback
|
| 182 |
+
improved_answer = agent.generate_answer(question, result["feedback"])
|
| 183 |
+
|
| 184 |
+
# Evaluate again
|
| 185 |
+
retry_result = env.step(improved_answer)
|
| 186 |
+
|
| 187 |
+
episode_data["attempt_2"] = {
|
| 188 |
+
"answer": improved_answer,
|
| 189 |
+
"score": retry_result["score"],
|
| 190 |
+
"reward": retry_result["reward"],
|
| 191 |
+
"feedback": retry_result["feedback"]
|
| 192 |
+
}
|
| 193 |
+
|
| 194 |
+
episode_data["improvement"] = retry_result["score"] - result["score"]
|
| 195 |
+
|
| 196 |
+
return {
|
| 197 |
+
"status": "success",
|
| 198 |
+
"episode": episode_data
|
| 199 |
+
}
|
| 200 |
+
|
| 201 |
+
except Exception as e:
|
| 202 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 203 |
+
|
| 204 |
+
|
| 205 |
+
@app.post("/config")
|
| 206 |
+
def update_config(request: ConfigRequest):
|
| 207 |
+
"""
|
| 208 |
+
Update API configuration (API key and model)
|
| 209 |
+
|
| 210 |
+
Args:
|
| 211 |
+
api_key: Hugging Face API token
|
| 212 |
+
model: Model name/endpoint to use
|
| 213 |
+
|
| 214 |
+
Returns:
|
| 215 |
+
Configuration status
|
| 216 |
+
"""
|
| 217 |
+
global agent, current_config
|
| 218 |
+
|
| 219 |
+
try:
|
| 220 |
+
# Update configuration
|
| 221 |
+
current_config["api_key"] = request.api_key
|
| 222 |
+
current_config["model"] = request.model
|
| 223 |
+
|
| 224 |
+
# Reinitialize agent with new configuration
|
| 225 |
+
agent = LLMAgent(api_key=request.api_key, model=request.model)
|
| 226 |
+
|
| 227 |
+
return {
|
| 228 |
+
"status": "success",
|
| 229 |
+
"message": "Configuration updated successfully",
|
| 230 |
+
"config": {
|
| 231 |
+
"model": request.model,
|
| 232 |
+
"api_key_set": bool(request.api_key),
|
| 233 |
+
"client_initialized": bool(agent.client)
|
| 234 |
+
}
|
| 235 |
+
}
|
| 236 |
+
except Exception as e:
|
| 237 |
+
raise HTTPException(status_code=500, detail=f"Configuration error: {str(e)}")
|
| 238 |
+
|
| 239 |
+
|
| 240 |
+
@app.get("/config")
|
| 241 |
+
def get_config():
|
| 242 |
+
"""
|
| 243 |
+
Get current API configuration
|
| 244 |
+
|
| 245 |
+
Returns:
|
| 246 |
+
Current configuration (without exposing full API key)
|
| 247 |
+
"""
|
| 248 |
+
return {
|
| 249 |
+
"status": "success",
|
| 250 |
+
"config": {
|
| 251 |
+
"model": current_config["model"],
|
| 252 |
+
"api_key_set": bool(current_config["api_key"]),
|
| 253 |
+
"client_initialized": bool(agent.client)
|
| 254 |
+
}
|
| 255 |
+
}
|
| 256 |
+
|
| 257 |
+
|
| 258 |
+
@app.get("/stats")
|
| 259 |
+
def get_stats():
|
| 260 |
+
"""Get episode statistics"""
|
| 261 |
+
try:
|
| 262 |
+
history = env.get_history()
|
| 263 |
+
if not history:
|
| 264 |
+
return {
|
| 265 |
+
"status": "success",
|
| 266 |
+
"stats": {
|
| 267 |
+
"total_attempts": 0,
|
| 268 |
+
"average_score": 0,
|
| 269 |
+
"average_reward": 0
|
| 270 |
+
}
|
| 271 |
+
}
|
| 272 |
+
|
| 273 |
+
avg_score = sum(h["score"] for h in history) / len(history)
|
| 274 |
+
avg_reward = sum(h["reward"] for h in history) / len(history)
|
| 275 |
+
|
| 276 |
+
return {
|
| 277 |
+
"status": "success",
|
| 278 |
+
"stats": {
|
| 279 |
+
"total_attempts": len(history),
|
| 280 |
+
"average_score": round(avg_score, 3),
|
| 281 |
+
"average_reward": round(avg_reward, 2),
|
| 282 |
+
"history": history
|
| 283 |
+
}
|
| 284 |
+
}
|
| 285 |
+
except Exception as e:
|
| 286 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 287 |
+
|
| 288 |
+
|
| 289 |
+
if __name__ == "__main__":
|
| 290 |
+
import uvicorn
|
| 291 |
+
uvicorn.run(app, host="0.0.0.0", port=8000)
|
backend/requirements.txt
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
fastapi==0.104.1
|
| 2 |
+
uvicorn==0.24.0
|
| 3 |
+
pydantic==2.5.0
|
| 4 |
+
openai==1.57.4
|
| 5 |
+
python-multipart==0.0.6
|
| 6 |
+
python-dotenv==1.0.0
|
backend/test_env.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Test script to verify the RL environment works correctly
|
| 3 |
+
Run this before starting the full application
|
| 4 |
+
"""
|
| 5 |
+
|
| 6 |
+
import sys
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
# Add parent directory to path
|
| 10 |
+
sys.path.append(str(Path(__file__).parent))
|
| 11 |
+
|
| 12 |
+
from env.interview_env import InterviewEnv
|
| 13 |
+
from evaluator.evaluator import Evaluator
|
| 14 |
+
from agent.llm_agent import LLMAgent
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
def test_environment():
|
| 18 |
+
"""Test the RL environment"""
|
| 19 |
+
print("=" * 60)
|
| 20 |
+
print("🧪 TESTING RL ENVIRONMENT")
|
| 21 |
+
print("=" * 60)
|
| 22 |
+
|
| 23 |
+
# Initialize components
|
| 24 |
+
evaluator = Evaluator()
|
| 25 |
+
env = InterviewEnv(evaluator=evaluator)
|
| 26 |
+
agent = LLMAgent()
|
| 27 |
+
|
| 28 |
+
print("\n✅ Components initialized successfully\n")
|
| 29 |
+
|
| 30 |
+
# Test 1: Reset environment
|
| 31 |
+
print("📝 Test 1: Environment Reset")
|
| 32 |
+
print("-" * 60)
|
| 33 |
+
state = env.reset()
|
| 34 |
+
print(f"Question: {state['question']}")
|
| 35 |
+
print(f"Difficulty: {state['difficulty']}")
|
| 36 |
+
print(f"Keywords: {', '.join(state['keywords'])}")
|
| 37 |
+
print("✅ Reset successful\n")
|
| 38 |
+
|
| 39 |
+
# Test 2: Evaluate a good answer
|
| 40 |
+
print("📝 Test 2: Evaluate Good Answer")
|
| 41 |
+
print("-" * 60)
|
| 42 |
+
good_answer = " ".join(state['keywords'][:3]) # Include some keywords
|
| 43 |
+
result = env.step(good_answer)
|
| 44 |
+
print(f"Answer: {good_answer}")
|
| 45 |
+
print(f"Score: {result['score']:.2%}")
|
| 46 |
+
print(f"Reward: {result['reward']}")
|
| 47 |
+
print(f"Feedback: {result['feedback']}")
|
| 48 |
+
print("✅ Evaluation successful\n")
|
| 49 |
+
|
| 50 |
+
# Test 3: Generate AI answer
|
| 51 |
+
print("📝 Test 3: AI Agent Answer Generation")
|
| 52 |
+
print("-" * 60)
|
| 53 |
+
env.reset() # Get new question
|
| 54 |
+
question = env.current_question['question']
|
| 55 |
+
ai_answer = agent.generate_answer(question)
|
| 56 |
+
print(f"Question: {question}")
|
| 57 |
+
print(f"AI Answer: {ai_answer}")
|
| 58 |
+
result = env.step(ai_answer)
|
| 59 |
+
print(f"Score: {result['score']:.2%}")
|
| 60 |
+
print(f"Reward: {result['reward']}")
|
| 61 |
+
print(f"Feedback: {result['feedback']}")
|
| 62 |
+
print("✅ AI generation successful\n")
|
| 63 |
+
|
| 64 |
+
# Test 4: Retry mechanism
|
| 65 |
+
print("📝 Test 4: Retry Mechanism")
|
| 66 |
+
print("-" * 60)
|
| 67 |
+
if env.should_retry(result['reward']):
|
| 68 |
+
print("⚠️ Low reward detected - initiating retry")
|
| 69 |
+
improved_answer = agent.generate_answer(question, result['feedback'])
|
| 70 |
+
|
| 71 |
+
# Reset to same question for retry
|
| 72 |
+
env.current_question = {
|
| 73 |
+
"question": question,
|
| 74 |
+
"keywords": state['keywords'],
|
| 75 |
+
"difficulty": state['difficulty']
|
| 76 |
+
}
|
| 77 |
+
|
| 78 |
+
retry_result = env.step(improved_answer)
|
| 79 |
+
print(f"Improved Answer: {improved_answer}")
|
| 80 |
+
print(f"New Score: {retry_result['score']:.2%}")
|
| 81 |
+
print(f"New Reward: {retry_result['reward']}")
|
| 82 |
+
improvement = retry_result['score'] - result['score']
|
| 83 |
+
print(f"Improvement: {improvement:+.2%}")
|
| 84 |
+
print("✅ Retry successful\n")
|
| 85 |
+
else:
|
| 86 |
+
print("✅ Score was good - no retry needed\n")
|
| 87 |
+
|
| 88 |
+
print("=" * 60)
|
| 89 |
+
print("✅ ALL TESTS PASSED!")
|
| 90 |
+
print("=" * 60)
|
| 91 |
+
print("\nYou can now start the FastAPI server with:")
|
| 92 |
+
print(" python main.py")
|
| 93 |
+
|
| 94 |
+
|
| 95 |
+
if __name__ == "__main__":
|
| 96 |
+
test_environment()
|
inference.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
OpenEnv-compliant inference script
|
| 3 |
+
Runs complete episode over all tasks with exact logging format
|
| 4 |
+
"""
|
| 5 |
+
|
| 6 |
+
import os
|
| 7 |
+
import sys
|
| 8 |
+
import json
|
| 9 |
+
from pathlib import Path
|
| 10 |
+
|
| 11 |
+
# Add backend to path
|
| 12 |
+
sys.path.insert(0, str(Path(__file__).parent / "backend"))
|
| 13 |
+
|
| 14 |
+
from app.environment import InterviewEnv
|
| 15 |
+
from app.evaluator import Evaluator
|
| 16 |
+
from app.agent import InterviewAgent
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
def run_inference():
|
| 20 |
+
"""
|
| 21 |
+
Run inference over all tasks with OpenEnv-compliant logging
|
| 22 |
+
|
| 23 |
+
Environment variables:
|
| 24 |
+
- API_BASE_URL: LLM API endpoint
|
| 25 |
+
- MODEL_NAME: Model identifier
|
| 26 |
+
- HF_TOKEN: API authentication token
|
| 27 |
+
"""
|
| 28 |
+
|
| 29 |
+
# Get environment variables
|
| 30 |
+
api_base_url = os.getenv("API_BASE_URL", "https://router.huggingface.co/v1")
|
| 31 |
+
model_name = os.getenv("MODEL_NAME", "Qwen/Qwen3-Coder-Next:novita")
|
| 32 |
+
hf_token = os.getenv("HF_TOKEN")
|
| 33 |
+
|
| 34 |
+
# Initialize components
|
| 35 |
+
evaluator = Evaluator()
|
| 36 |
+
questions_path = Path(__file__).parent / "backend" / "app" / "dataset.json"
|
| 37 |
+
env = InterviewEnv(questions_path=str(questions_path), evaluator=evaluator)
|
| 38 |
+
|
| 39 |
+
# Initialize agent (will use mock mode if no token)
|
| 40 |
+
if hf_token:
|
| 41 |
+
agent = InterviewAgent(mode="api", api_key=hf_token, model=model_name)
|
| 42 |
+
else:
|
| 43 |
+
print("⚠️ No HF_TOKEN found, using mock mode")
|
| 44 |
+
agent = InterviewAgent(mode="mock")
|
| 45 |
+
|
| 46 |
+
# Load all tasks
|
| 47 |
+
with open(questions_path, 'r') as f:
|
| 48 |
+
tasks = json.load(f)
|
| 49 |
+
|
| 50 |
+
print(f"Running inference on {len(tasks)} tasks")
|
| 51 |
+
print(f"API Base URL: {api_base_url}")
|
| 52 |
+
print(f"Model: {model_name}")
|
| 53 |
+
print("-" * 80)
|
| 54 |
+
|
| 55 |
+
# Run inference on each task
|
| 56 |
+
for task_idx, task in enumerate(tasks):
|
| 57 |
+
task_id = f"task_{task_idx}"
|
| 58 |
+
|
| 59 |
+
# Print START marker
|
| 60 |
+
print(f"[START]")
|
| 61 |
+
print(f"task_id={task_id}")
|
| 62 |
+
|
| 63 |
+
# Reset environment to this specific task
|
| 64 |
+
env.current_question = task
|
| 65 |
+
env.episode_history = []
|
| 66 |
+
env.retry_count = 0
|
| 67 |
+
|
| 68 |
+
question = task["question"]
|
| 69 |
+
|
| 70 |
+
# Generate answer
|
| 71 |
+
answer = agent.generate_answer(question)
|
| 72 |
+
|
| 73 |
+
# Print STEP marker with action
|
| 74 |
+
print(f"[STEP]")
|
| 75 |
+
print(f"action={answer}")
|
| 76 |
+
|
| 77 |
+
# Evaluate
|
| 78 |
+
result = env.step(answer)
|
| 79 |
+
reward = result["reward"]
|
| 80 |
+
|
| 81 |
+
# Print reward
|
| 82 |
+
print(f"reward={reward}")
|
| 83 |
+
|
| 84 |
+
# Print END marker
|
| 85 |
+
print(f"[END]")
|
| 86 |
+
print()
|
| 87 |
+
|
| 88 |
+
print("-" * 80)
|
| 89 |
+
print(f"✅ Inference complete: {len(tasks)} tasks processed")
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
if __name__ == "__main__":
|
| 93 |
+
run_inference()
|
openenv.yaml
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: ai-interview-prep-rl-env
|
| 2 |
+
description: Reinforcement Learning environment for AI-powered technical interview preparation with keyword-based evaluation and retry mechanism
|
| 3 |
+
|
| 4 |
+
endpoints:
|
| 5 |
+
reset:
|
| 6 |
+
method: POST
|
| 7 |
+
path: /reset
|
| 8 |
+
description: Reset environment and get new interview question
|
| 9 |
+
response_schema:
|
| 10 |
+
type: object
|
| 11 |
+
properties:
|
| 12 |
+
question:
|
| 13 |
+
type: string
|
| 14 |
+
description: Interview question text
|
| 15 |
+
difficulty:
|
| 16 |
+
type: string
|
| 17 |
+
description: Question difficulty level (easy, medium, hard)
|
| 18 |
+
attempt:
|
| 19 |
+
type: integer
|
| 20 |
+
description: Current attempt number (0 after reset)
|
| 21 |
+
required:
|
| 22 |
+
- question
|
| 23 |
+
- difficulty
|
| 24 |
+
- attempt
|
| 25 |
+
|
| 26 |
+
step:
|
| 27 |
+
method: POST
|
| 28 |
+
path: /step
|
| 29 |
+
description: Submit answer and get evaluation with reward
|
| 30 |
+
request_schema:
|
| 31 |
+
type: object
|
| 32 |
+
properties:
|
| 33 |
+
action:
|
| 34 |
+
type: string
|
| 35 |
+
description: Candidate's answer to the question
|
| 36 |
+
required:
|
| 37 |
+
- action
|
| 38 |
+
response_schema:
|
| 39 |
+
type: object
|
| 40 |
+
properties:
|
| 41 |
+
reward:
|
| 42 |
+
type: number
|
| 43 |
+
description: Reward signal between 0.0 and 1.0
|
| 44 |
+
minimum: 0.0
|
| 45 |
+
maximum: 1.0
|
| 46 |
+
done:
|
| 47 |
+
type: boolean
|
| 48 |
+
description: Whether episode is complete
|
| 49 |
+
state:
|
| 50 |
+
type: object
|
| 51 |
+
properties:
|
| 52 |
+
question:
|
| 53 |
+
type: string
|
| 54 |
+
difficulty:
|
| 55 |
+
type: string
|
| 56 |
+
attempt:
|
| 57 |
+
type: integer
|
| 58 |
+
required:
|
| 59 |
+
- question
|
| 60 |
+
- difficulty
|
| 61 |
+
- attempt
|
| 62 |
+
required:
|
| 63 |
+
- reward
|
| 64 |
+
- done
|
| 65 |
+
- state
|
| 66 |
+
|
| 67 |
+
state:
|
| 68 |
+
method: GET
|
| 69 |
+
path: /state
|
| 70 |
+
description: Get current environment state
|
| 71 |
+
response_schema:
|
| 72 |
+
type: object
|
| 73 |
+
properties:
|
| 74 |
+
question:
|
| 75 |
+
type: string
|
| 76 |
+
description: Current interview question
|
| 77 |
+
difficulty:
|
| 78 |
+
type: string
|
| 79 |
+
description: Question difficulty level
|
| 80 |
+
attempt:
|
| 81 |
+
type: integer
|
| 82 |
+
description: Current attempt number
|
| 83 |
+
required:
|
| 84 |
+
- question
|
| 85 |
+
- difficulty
|
| 86 |
+
- attempt
|
| 87 |
+
|
| 88 |
+
tasks:
|
| 89 |
+
count: 12
|
| 90 |
+
source: backend/app/dataset.json
|
| 91 |
+
description: Technical interview questions covering algorithms, systems, databases, and software design
|
| 92 |
+
|
| 93 |
+
grading:
|
| 94 |
+
type: hybrid
|
| 95 |
+
deterministic: true
|
| 96 |
+
components:
|
| 97 |
+
- keyword_matching: 0.8
|
| 98 |
+
- length_scoring: 0.2
|
| 99 |
+
output_range:
|
| 100 |
+
min: 0.0
|
| 101 |
+
max: 1.0
|
| 102 |
+
|
| 103 |
+
environment:
|
| 104 |
+
type: single_question_episode
|
| 105 |
+
max_retries: 3
|
| 106 |
+
state_representation:
|
| 107 |
+
- question: string
|
| 108 |
+
- difficulty: string
|
| 109 |
+
- attempt: integer
|
| 110 |
+
action_space: text
|
| 111 |
+
reward_range: [0.0, 1.0]
|