Divyatmaj commited on
Commit
cbf1ed7
·
0 Parent(s):

clean initial commit

Browse files
.DS_Store ADDED
Binary file (8.2 kB). View file
 
.gitattributes ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
.gitignore ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ venv/
2
+ __pycache__/
3
+ *.pyc
4
+ *.pyo
5
+ *.pyd
6
+ .env
Dockerfile ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM python:3.10-slim
2
+
3
+ WORKDIR /app
4
+
5
+ COPY . .
6
+
7
+ RUN pip install --no-cache-dir -r backend/requirements.txt
8
+
9
+ EXPOSE 7860
10
+
11
+ CMD ["uvicorn", "backend.main:app", "--host", "0.0.0.0", "--port", "7860"]
README.md ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: Scalar Hackathon Space 1
3
+ emoji: 🌍
4
+ colorFrom: indigo
5
+ colorTo: pink
6
+ sdk: docker
7
+ pinned: false
8
+ ---
9
+
10
+ Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
backend/.DS_Store ADDED
Binary file (10.2 kB). View file
 
backend/app/__init__.py ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ AI Interview Preparation - Core Application Package
3
+ """
4
+
5
+ from .environment import InterviewEnv
6
+ from .evaluator import Evaluator
7
+ from .agent import InterviewAgent
8
+
9
+ __all__ = ['InterviewEnv', 'Evaluator', 'InterviewAgent']
backend/app/agent.py ADDED
@@ -0,0 +1,334 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ AI Agent with Retry and Memory Capabilities
3
+ Uses HuggingFace LLM via OpenAI-compatible API
4
+
5
+ Features:
6
+ - Real LLM integration (HuggingFace Qwen model)
7
+ - Retry mechanism with feedback integration
8
+ - Memory of previous attempts
9
+ - Mock fallback when API unavailable
10
+ """
11
+
12
+ import os
13
+ from typing import List, Dict, Any, Optional
14
+
15
+
16
+ class InterviewAgent:
17
+ """
18
+ AI Agent for Interview Preparation
19
+
20
+ Simulates a candidate that:
21
+ 1. Generates answers to questions
22
+ 2. Receives feedback on answers
23
+ 3. Improves answers using feedback
24
+ 4. Tracks performance over time
25
+
26
+ The agent can use:
27
+ - Mock/simulated answers (for testing)
28
+ - Real LLM API (GPT-4, Claude, etc.)
29
+ """
30
+
31
+ def __init__(self, mode: str = "mock", api_key: str = None, model: str = None):
32
+ """
33
+ Initialize the AI agent
34
+
35
+ Args:
36
+ mode: "mock" for simulated answers or "api" for real LLM
37
+ api_key: API key for HuggingFace (or from HF_TOKEN env variable)
38
+ model: Model name (or from MODEL_NAME env variable)
39
+ """
40
+ self.mode = mode
41
+ self.api_key = api_key or os.getenv("HF_TOKEN")
42
+ self.model = model or os.getenv("MODEL_NAME", "Qwen/Qwen3-Coder-Next:novita")
43
+ self.base_url = os.getenv("API_BASE_URL", "https://router.huggingface.co/v1")
44
+ self.memory = []
45
+ self.current_question_memory = []
46
+ self.client = None
47
+
48
+ # Initialize API client if using real LLM
49
+ if mode == "api" and self.api_key:
50
+ self._init_api_client()
51
+ elif mode == "api":
52
+ print("⚠️ API mode selected but no API key provided. Falling back to mock mode.")
53
+ self.mode = "mock"
54
+
55
+ def _init_api_client(self):
56
+ """Initialize LLM client via OpenAI-compatible API"""
57
+ try:
58
+ from openai import OpenAI
59
+ self.client = OpenAI(
60
+ base_url=self.base_url,
61
+ api_key=self.api_key
62
+ )
63
+ print(f"🔌 API client initialized")
64
+ print(f" Base URL: {self.base_url}")
65
+ print(f" Model: {self.model}")
66
+ except ImportError:
67
+ print("⚠️ openai package not installed. Install with: pip install openai")
68
+ print("⚠️ Falling back to mock mode")
69
+ self.mode = "mock"
70
+ except Exception as e:
71
+ print(f"⚠️ Error initializing API client: {e}")
72
+ print("⚠️ Falling back to mock mode")
73
+ self.mode = "mock"
74
+
75
+ def generate_answer(
76
+ self,
77
+ question: str,
78
+ feedback: Optional[str] = None,
79
+ missing_keywords: Optional[List[str]] = None,
80
+ previous_attempts: Optional[List[Dict]] = None
81
+ ) -> str:
82
+ """
83
+ Generate an answer to the interview question
84
+
85
+ Uses different strategies based on whether this is:
86
+ - Initial attempt: Generate fresh answer
87
+ - Retry attempt: Improve using feedback and missing keywords
88
+
89
+ Args:
90
+ question: The interview question
91
+ feedback: Feedback from previous attempt (if retry)
92
+ missing_keywords: Keywords that were missing (if retry)
93
+ previous_attempts: List of previous attempts (for context)
94
+
95
+ Returns:
96
+ Generated answer string
97
+ """
98
+ if self.mode == "mock":
99
+ return self._generate_mock_answer(
100
+ question,
101
+ feedback,
102
+ missing_keywords,
103
+ previous_attempts
104
+ )
105
+ else:
106
+ return self._generate_api_answer(
107
+ question,
108
+ feedback,
109
+ missing_keywords,
110
+ previous_attempts
111
+ )
112
+
113
+ def _generate_mock_answer(
114
+ self,
115
+ question: str,
116
+ feedback: Optional[str] = None,
117
+ missing_keywords: Optional[List[str]] = None,
118
+ previous_attempts: Optional[List[Dict]] = None
119
+ ) -> str:
120
+ """
121
+ Generate simulated answer (for testing without API costs)
122
+
123
+ Mock Logic:
124
+ - Initial attempt: Returns partial answer (simulates incomplete knowledge)
125
+ - Retry attempt: Adds missing keywords to answer (simulates learning)
126
+
127
+ This allows testing the retry/feedback loop without LLM API calls.
128
+
129
+ Args:
130
+ question: Interview question
131
+ feedback: Previous feedback
132
+ missing_keywords: Keywords to add
133
+ previous_attempts: History
134
+
135
+ Returns:
136
+ Simulated answer
137
+ """
138
+ # Database of mock answers for common topics
139
+ # These are intentionally partial to trigger retry mechanism
140
+ mock_knowledge_base = {
141
+ "binary search": "Binary search is an algorithm that finds items in a sorted array by dividing the search space.",
142
+ "process": "A process is a program in execution. A thread is a unit of execution within a process.",
143
+ "normalization": "Database normalization organizes data to reduce redundancy using normal forms.",
144
+ "oop": "The pillars of OOP include encapsulation and inheritance.",
145
+ "deadlock": "A deadlock is when processes wait for each other's resources.",
146
+ "hashmap": "A hashmap stores key-value pairs using a hash function.",
147
+ "sql": "SQL databases use structured schemas. NoSQL databases are more flexible.",
148
+ "solid": "SOLID includes Single Responsibility and Open-Closed principles.",
149
+ "stack": "Stack uses LIFO. Heap is for dynamic allocation.",
150
+ "rest": "REST is an architectural style for web services using HTTP.",
151
+ "big o": "Big O notation describes algorithm complexity.",
152
+ "cap": "CAP theorem states you can only achieve two of three properties in distributed systems."
153
+ }
154
+
155
+ # Find relevant base answer
156
+ question_lower = question.lower()
157
+ base_answer = None
158
+
159
+ for key, answer in mock_knowledge_base.items():
160
+ if key in question_lower:
161
+ base_answer = answer
162
+ break
163
+
164
+ if base_answer is None:
165
+ base_answer = "This is a complex topic in computer science."
166
+
167
+ # If this is a retry, improve the answer by adding missing keywords
168
+ if feedback and missing_keywords:
169
+ improved_answer = base_answer
170
+
171
+ # Add missing keywords to answer (simulating learning)
172
+ if missing_keywords:
173
+ improved_answer += " Important concepts include: "
174
+ improved_answer += ", ".join(missing_keywords[:4]) + "."
175
+
176
+ # Add some variety
177
+ attempt_count = len(previous_attempts) if previous_attempts else 1
178
+ if attempt_count > 1:
179
+ improved_answer += f" Additionally, this relates to fundamental principles of the domain."
180
+
181
+ return improved_answer
182
+
183
+ # Return initial (incomplete) answer
184
+ return base_answer
185
+
186
+ def _generate_api_answer(
187
+ self,
188
+ question: str,
189
+ feedback: Optional[str] = None,
190
+ missing_keywords: Optional[List[str]] = None,
191
+ previous_attempts: Optional[List[Dict]] = None
192
+ ) -> str:
193
+ """
194
+ Generate answer using HuggingFace LLM API
195
+
196
+ Constructs different prompts based on:
197
+ - Initial attempt: Clear, structured question
198
+ - Retry attempt: Includes feedback and missing concepts
199
+ """
200
+ if not self.client:
201
+ return self._generate_mock_answer(question, feedback, missing_keywords, previous_attempts)
202
+
203
+ # Build prompt
204
+ if feedback and missing_keywords:
205
+ # Retry prompt with feedback
206
+ prompt = f"""You are a technical interview candidate who received feedback on your previous answer.
207
+
208
+ Question: {question}
209
+
210
+ Previous Feedback: {feedback}
211
+
212
+ Missing Concepts: {', '.join(missing_keywords)}
213
+
214
+ Please provide an IMPROVED answer that addresses the feedback and includes the missing concepts. Be clear, concise, and comprehensive."""
215
+ else:
216
+ # Initial prompt
217
+ prompt = f"""You are a technical interview candidate. Answer the following question clearly and comprehensively, covering all key concepts:
218
+
219
+ Question: {question}
220
+
221
+ Provide a well-structured answer:"""
222
+
223
+ try:
224
+ response = self.client.chat.completions.create(
225
+ model=self.model,
226
+ messages=[
227
+ {"role": "system", "content": "You are a knowledgeable technical interview candidate."},
228
+ {"role": "user", "content": prompt}
229
+ ],
230
+ temperature=0.7,
231
+ max_tokens=300
232
+ )
233
+ return response.choices[0].message.content.strip()
234
+ except Exception as e:
235
+ print(f"Error calling HuggingFace API: {e}")
236
+ return self._generate_mock_answer(question, feedback, missing_keywords, previous_attempts)
237
+
238
+ def remember_attempt(
239
+ self,
240
+ question: str,
241
+ answer: str,
242
+ score: float,
243
+ feedback: str,
244
+ attempt: int
245
+ ):
246
+ """
247
+ Store attempt in memory for learning and analysis
248
+
249
+ Memory allows agent to:
250
+ - Track improvement over time
251
+ - Analyze common mistakes
252
+ - Build knowledge base
253
+
254
+ Args:
255
+ question: The question asked
256
+ answer: Answer provided
257
+ score: Score received
258
+ feedback: Feedback received
259
+ attempt: Attempt number
260
+ """
261
+ memory_entry = {
262
+ "question": question,
263
+ "answer": answer,
264
+ "score": score,
265
+ "feedback": feedback,
266
+ "attempt": attempt
267
+ }
268
+
269
+ # Add to both global and question-specific memory
270
+ self.memory.append(memory_entry)
271
+ self.current_question_memory.append(memory_entry)
272
+
273
+ def reset_question_memory(self):
274
+ """
275
+ Clear memory for current question
276
+
277
+ Called when starting a new question/episode.
278
+ Keeps global memory but clears question-specific memory.
279
+ """
280
+ self.current_question_memory = []
281
+
282
+ def get_improvement_stats(self) -> Dict[str, Any]:
283
+ """
284
+ Calculate improvement statistics from memory
285
+
286
+ Returns:
287
+ Dictionary with:
288
+ - total_attempts: Total questions attempted
289
+ - average_score: Average score across all attempts
290
+ - improvement_rate: How much scores improved on retries
291
+ - total_retries: Number of retry attempts made
292
+ """
293
+ if not self.memory:
294
+ return {
295
+ "total_attempts": 0,
296
+ "average_score": 0.0,
297
+ "improvement_rate": 0.0,
298
+ "total_retries": 0
299
+ }
300
+
301
+ scores = [m["score"] for m in self.memory]
302
+ retries = [m for m in self.memory if m["attempt"] > 1]
303
+
304
+ # Calculate improvement rate
305
+ improvement_rate = 0.0
306
+ if len(self.current_question_memory) > 1:
307
+ first_score = self.current_question_memory[0]["score"]
308
+ last_score = self.current_question_memory[-1]["score"]
309
+ improvement_rate = last_score - first_score
310
+
311
+ return {
312
+ "total_attempts": len(self.memory),
313
+ "average_score": sum(scores) / len(scores),
314
+ "improvement_rate": improvement_rate,
315
+ "total_retries": len(retries)
316
+ }
317
+
318
+ def get_memory(self) -> List[Dict[str, Any]]:
319
+ """
320
+ Get complete memory
321
+
322
+ Returns:
323
+ List of all stored attempts
324
+ """
325
+ return self.memory
326
+
327
+ def get_current_question_memory(self) -> List[Dict[str, Any]]:
328
+ """
329
+ Get memory for current question only
330
+
331
+ Returns:
332
+ List of attempts for current question
333
+ """
334
+ return self.current_question_memory
backend/app/dataset.json ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "question": "What is a binary search and what is its time complexity?",
4
+ "keywords": ["sorted array", "divide", "logarithmic", "O(log n)", "middle element", "halving"],
5
+ "difficulty": "easy"
6
+ },
7
+ {
8
+ "question": "Explain the difference between a process and a thread.",
9
+ "keywords": ["process", "thread", "memory space", "lightweight", "shared memory", "independent"],
10
+ "difficulty": "medium"
11
+ },
12
+ {
13
+ "question": "What is database normalization and why is it important?",
14
+ "keywords": ["redundancy", "normal forms", "dependency", "data integrity", "anomalies"],
15
+ "difficulty": "medium"
16
+ },
17
+ {
18
+ "question": "Explain the four pillars of Object-Oriented Programming.",
19
+ "keywords": ["encapsulation", "abstraction", "inheritance", "polymorphism"],
20
+ "difficulty": "easy"
21
+ },
22
+ {
23
+ "question": "What is a deadlock in operating systems and how can it be prevented?",
24
+ "keywords": ["mutual exclusion", "hold and wait", "circular wait", "prevention", "avoidance", "resources"],
25
+ "difficulty": "hard"
26
+ },
27
+ {
28
+ "question": "Explain how a hashmap works internally.",
29
+ "keywords": ["hash function", "buckets", "collision", "linked list", "O(1)", "array"],
30
+ "difficulty": "medium"
31
+ },
32
+ {
33
+ "question": "What is the difference between SQL and NoSQL databases?",
34
+ "keywords": ["structured", "schema", "scalability", "ACID", "flexible", "horizontal scaling"],
35
+ "difficulty": "easy"
36
+ },
37
+ {
38
+ "question": "Describe the SOLID principles in software design.",
39
+ "keywords": ["single responsibility", "open-closed", "liskov substitution", "interface segregation", "dependency inversion"],
40
+ "difficulty": "hard"
41
+ },
42
+ {
43
+ "question": "What is the difference between stack and heap memory?",
44
+ "keywords": ["static", "dynamic", "LIFO", "allocation", "scope", "lifetime"],
45
+ "difficulty": "medium"
46
+ },
47
+ {
48
+ "question": "Explain what a RESTful API is and its key constraints.",
49
+ "keywords": ["stateless", "client-server", "cacheable", "uniform interface", "HTTP methods", "resources"],
50
+ "difficulty": "medium"
51
+ },
52
+ {
53
+ "question": "What is Big O notation and why is it important?",
54
+ "keywords": ["time complexity", "space complexity", "worst case", "algorithm efficiency", "growth rate"],
55
+ "difficulty": "easy"
56
+ },
57
+ {
58
+ "question": "Explain the CAP theorem in distributed systems.",
59
+ "keywords": ["consistency", "availability", "partition tolerance", "distributed", "trade-off"],
60
+ "difficulty": "hard"
61
+ }
62
+ ]
backend/app/environment.py ADDED
@@ -0,0 +1,281 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ AI Interview Preparation Environment
3
+ OpenAI Gym-style RL Environment for Interview Practice
4
+
5
+ This module implements the core environment following the standard RL interface:
6
+ - reset() -> returns initial state (question)
7
+ - step(action) -> executes action and returns (reward, state, done, info)
8
+
9
+ Features:
10
+ - Random question selection from dataset
11
+ - Episode history tracking
12
+ - Configurable retry thresholds
13
+ - Support for multi-turn episodes (extensible)
14
+ """
15
+
16
+ import json
17
+ import random
18
+ from typing import Dict, Any, List
19
+ from pathlib import Path
20
+
21
+
22
+ class InterviewEnv:
23
+ """
24
+ Interview Preparation Environment
25
+
26
+ This environment simulates a technical interview scenario where:
27
+ - State: Interview question with metadata (keywords, difficulty)
28
+ - Action: Candidate's answer (text string)
29
+ - Reward: Score-based reward computed by evaluator
30
+
31
+ Design Decisions:
32
+ - Single question per episode (can be extended to multi-question)
33
+ - Stateful: maintains current question and episode history
34
+ - Delegated evaluation: uses injected evaluator for scoring
35
+ """
36
+
37
+ def __init__(self, questions_path: str = None, evaluator=None):
38
+ """
39
+ Initialize the interview environment
40
+
41
+ Args:
42
+ questions_path: Path to questions.json (defaults to ../data/questions.json)
43
+ evaluator: Evaluator instance for scoring answers (must implement evaluate())
44
+ """
45
+ # Set default path if not provided
46
+ if questions_path is None:
47
+ questions_path = Path(__file__).parent.parent / "data" / "questions.json"
48
+
49
+ self.questions_path = questions_path
50
+ self.evaluator = evaluator
51
+ self.questions = self._load_questions()
52
+
53
+ # State tracking
54
+ self.current_question = None
55
+ self.episode_history = [] # Stores all attempts in current episode
56
+ self.retry_count = 0
57
+ self.max_retries = 3 # Configurable retry limit
58
+
59
+ def _load_questions(self) -> List[Dict[str, Any]]:
60
+ """
61
+ Load questions from JSON dataset
62
+
63
+ Returns:
64
+ List of question dictionaries with structure:
65
+ {
66
+ "question": str,
67
+ "keywords": List[str],
68
+ "difficulty": str
69
+ }
70
+ """
71
+ try:
72
+ with open(self.questions_path, 'r') as f:
73
+ questions = json.load(f)
74
+ print(f"✅ Loaded {len(questions)} questions from {self.questions_path}")
75
+ return questions
76
+ except FileNotFoundError:
77
+ print(f"❌ Questions file not found: {self.questions_path}")
78
+ return []
79
+ except json.JSONDecodeError:
80
+ print(f"❌ Invalid JSON in questions file")
81
+ return []
82
+
83
+ def reset(self) -> Dict[str, Any]:
84
+ """
85
+ Reset environment to initial state
86
+
87
+ Selects a new random question and clears episode history.
88
+ This follows the Gym API convention.
89
+
90
+ Returns:
91
+ state: Dictionary containing:
92
+ - question: The interview question text
93
+ - keywords: Expected keywords for evaluation
94
+ - difficulty: Question difficulty level
95
+ - attempt: Current attempt number (always 0 after reset)
96
+ """
97
+ # Select random question from dataset
98
+ if not self.questions:
99
+ raise ValueError("No questions loaded. Check questions.json")
100
+
101
+ self.current_question = random.choice(self.questions)
102
+ self.episode_history = []
103
+ self.retry_count = 0
104
+
105
+ # Return state representation
106
+ state = {
107
+ "question": self.current_question["question"],
108
+ "keywords": self.current_question["keywords"],
109
+ "difficulty": self.current_question["difficulty"],
110
+ "attempt": 0
111
+ }
112
+
113
+ print(f"\n{'='*60}")
114
+ print(f"📝 NEW QUESTION [{state['difficulty'].upper()}]")
115
+ print(f"{'='*60}")
116
+ print(f"{state['question']}")
117
+ print(f"{'='*60}\n")
118
+
119
+ return state
120
+
121
+ def step(self, action: str) -> Dict[str, Any]:
122
+ """
123
+ Execute one environment step
124
+
125
+ Takes the candidate's answer, evaluates it, and returns feedback.
126
+ This is the core of the RL loop.
127
+
128
+ Args:
129
+ action: The candidate's answer (string)
130
+
131
+ Returns:
132
+ Dictionary containing:
133
+ - reward: Integer reward signal (+10, +5, 0, -5)
134
+ - score: Float score (0.0 to 1.0)
135
+ - feedback: Structured feedback message
136
+ - matched_keywords: List of correctly mentioned keywords
137
+ - missing_keywords: List of missed keywords
138
+ - done: Boolean indicating if episode is complete
139
+ - attempt: Current attempt number
140
+
141
+ Raises:
142
+ ValueError: If environment not initialized (call reset() first)
143
+ """
144
+ # Validation
145
+ if self.current_question is None:
146
+ raise ValueError("Environment not initialized. Call reset() first.")
147
+
148
+ if self.evaluator is None:
149
+ raise ValueError("Evaluator not set. Pass evaluator to constructor.")
150
+
151
+ # Evaluate the answer using injected evaluator
152
+ evaluation = self.evaluator.evaluate(
153
+ answer=action,
154
+ keywords=self.current_question["keywords"]
155
+ )
156
+
157
+ # Extract evaluation results
158
+ score = evaluation["score"]
159
+ feedback = evaluation["feedback"]
160
+ matched_keywords = evaluation["matched_keywords"]
161
+ missing_keywords = evaluation["missing_keywords"]
162
+
163
+ # Compute reward signal
164
+ reward = self.evaluator.compute_reward(score)
165
+
166
+ # Increment retry counter
167
+ self.retry_count += 1
168
+
169
+ # Store in episode history for memory/analysis
170
+ self.episode_history.append({
171
+ "attempt": self.retry_count,
172
+ "answer": action,
173
+ "score": score,
174
+ "reward": reward,
175
+ "matched_keywords": matched_keywords,
176
+ "missing_keywords": missing_keywords,
177
+ "feedback": feedback
178
+ })
179
+
180
+ # Determine if episode is done
181
+ # Done if: high score OR max retries reached
182
+ done = (reward >= 0.9) or (self.retry_count >= self.max_retries)
183
+
184
+ # Return full step information
185
+ return {
186
+ "reward": reward,
187
+ "score": score,
188
+ "feedback": feedback,
189
+ "matched_keywords": matched_keywords,
190
+ "missing_keywords": missing_keywords,
191
+ "done": done,
192
+ "attempt": self.retry_count,
193
+ "question": self.current_question["question"]
194
+ }
195
+
196
+ def should_retry(self, reward: float) -> bool:
197
+ """
198
+ Determine if agent should attempt to improve answer
199
+
200
+ Args:
201
+ reward: Current reward value (0.0 to 1.0)
202
+
203
+ Returns:
204
+ True if agent should retry (low reward and retries available)
205
+ """
206
+ can_retry = self.retry_count < self.max_retries
207
+ needs_retry = reward < 0.7 # Threshold for improvement
208
+
209
+ return can_retry and needs_retry
210
+
211
+ def get_history(self) -> List[Dict[str, Any]]:
212
+ """
213
+ Get complete episode history
214
+
215
+ Returns:
216
+ List of all attempts in current episode with full details
217
+ """
218
+ return self.episode_history
219
+
220
+ def get_stats(self) -> Dict[str, Any]:
221
+ """
222
+ Get episode statistics
223
+
224
+ Returns:
225
+ Dictionary with:
226
+ - total_attempts: Number of attempts made
227
+ - best_score: Highest score achieved
228
+ - improvement: Score improvement from first to last
229
+ - final_reward: Final reward value
230
+ """
231
+ if not self.episode_history:
232
+ return {
233
+ "total_attempts": 0,
234
+ "best_score": 0.0,
235
+ "improvement": 0.0,
236
+ "final_reward": 0
237
+ }
238
+
239
+ scores = [h["score"] for h in self.episode_history]
240
+ rewards = [h["reward"] for h in self.episode_history]
241
+
242
+ return {
243
+ "total_attempts": len(self.episode_history),
244
+ "best_score": max(scores),
245
+ "improvement": scores[-1] - scores[0] if len(scores) > 1 else 0.0,
246
+ "final_reward": rewards[-1]
247
+ }
248
+
249
+ def set_max_retries(self, max_retries: int):
250
+ """
251
+ Configure maximum retry attempts
252
+
253
+ Args:
254
+ max_retries: Maximum number of attempts allowed
255
+ """
256
+ self.max_retries = max_retries
257
+
258
+ def state(self) -> Dict[str, Any]:
259
+ """
260
+ Get current environment state
261
+
262
+ OpenEnv-compliant state getter.
263
+ Returns current question and attempt information.
264
+
265
+ Returns:
266
+ Dictionary containing:
267
+ - question: Current question text
268
+ - difficulty: Question difficulty level
269
+ - attempt: Current attempt number
270
+
271
+ Raises:
272
+ ValueError: If environment not initialized (call reset() first)
273
+ """
274
+ if self.current_question is None:
275
+ raise ValueError("Environment not initialized. Call reset() first.")
276
+
277
+ return {
278
+ "question": self.current_question["question"],
279
+ "difficulty": self.current_question["difficulty"],
280
+ "attempt": self.retry_count
281
+ }
backend/app/evaluator.py ADDED
@@ -0,0 +1,310 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Advanced Evaluator with Hybrid Scoring System
3
+ Evaluates interview answers using multiple metrics
4
+
5
+ Features:
6
+ - Keyword-based scoring (primary)
7
+ - AI-based scoring (placeholder/mock - extensible)
8
+ - Structured feedback generation
9
+ - Reward computation for RL
10
+ - Text normalization for robust matching
11
+
12
+ Design Philosophy:
13
+ - Modular: Easy to swap keyword logic or add LLM judge
14
+ - Extensible: Hybrid scoring allows future AI integration
15
+ - Explainable: Returns detailed feedback and matched/missing keywords
16
+ """
17
+
18
+ import re
19
+ from typing import Dict, List, Any, Tuple
20
+
21
+
22
+ class Evaluator:
23
+ """
24
+ Hybrid Answer Evaluator
25
+
26
+ Evaluates answers using:
27
+ 1. Keyword matching (primary metric)
28
+ 2. AI scoring (placeholder for LLM-as-judge)
29
+
30
+ Final score = 0.7 * keyword_score + 0.3 * ai_score
31
+
32
+ This allows:
33
+ - Fast, deterministic evaluation (keywords)
34
+ - Future semantic understanding (AI judge)
35
+ - Explainable results (specific missing concepts)
36
+ """
37
+
38
+ def __init__(self, keyword_weight: float = 0.8, ai_weight: float = 0.2):
39
+ """
40
+ Initialize evaluator with scoring weights
41
+
42
+ Args:
43
+ keyword_weight: Weight for keyword-based score (default 0.8)
44
+ ai_weight: Weight for length-based score (default 0.2)
45
+
46
+ Weights should sum to 1.0 for normalized scoring
47
+ """
48
+ self.keyword_weight = keyword_weight
49
+ self.ai_weight = ai_weight
50
+
51
+ # Validate weights
52
+ if not abs((keyword_weight + ai_weight) - 1.0) < 0.01:
53
+ print(f"⚠️ Warning: Weights don't sum to 1.0 ({keyword_weight + ai_weight})")
54
+
55
+ def normalize_text(self, text: str) -> str:
56
+ """
57
+ Normalize text for robust keyword matching
58
+
59
+ Steps:
60
+ 1. Convert to lowercase
61
+ 2. Remove special characters (keep alphanumeric, spaces, hyphens, parens)
62
+ 3. Collapse multiple spaces
63
+
64
+ Args:
65
+ text: Raw text string
66
+
67
+ Returns:
68
+ Normalized text string
69
+
70
+ Examples:
71
+ "O(log n)" -> "o(log n)"
72
+ "Process!!!" -> "process"
73
+ "REST API" -> "rest api"
74
+ """
75
+ text = text.lower()
76
+
77
+ # Keep alphanumeric, spaces, hyphens, and parentheses
78
+ # This preserves "O(log n)", "client-server", etc.
79
+ text = re.sub(r'[^a-z0-9\s\-\(\)]', ' ', text)
80
+
81
+ # Collapse multiple spaces
82
+ text = ' '.join(text.split())
83
+
84
+ return text
85
+
86
+ def evaluate_keywords(self, answer: str, keywords: List[str]) -> Tuple[float, List[str], List[str]]:
87
+ """
88
+ Evaluate answer based on keyword coverage
89
+
90
+ This is the primary evaluation metric.
91
+
92
+ Args:
93
+ answer: Candidate's answer text
94
+ keywords: List of expected keywords/phrases
95
+
96
+ Returns:
97
+ Tuple of (score, matched_keywords, missing_keywords)
98
+ - score: Proportion of keywords found (0.0 to 1.0)
99
+ - matched_keywords: Keywords found in answer
100
+ - missing_keywords: Keywords not found in answer
101
+
102
+ Algorithm:
103
+ 1. Normalize both answer and keywords
104
+ 2. Check if each keyword appears in answer (substring match)
105
+ 3. Calculate score as ratio: matched / total
106
+ """
107
+ if not answer or not answer.strip():
108
+ return 0.0, [], keywords
109
+
110
+ normalized_answer = self.normalize_text(answer)
111
+ matched_keywords = []
112
+ missing_keywords = []
113
+
114
+ # Check each keyword
115
+ for keyword in keywords:
116
+ normalized_keyword = self.normalize_text(keyword)
117
+
118
+ # Use substring matching for flexibility
119
+ # "O(log n)" matches "time complexity is O(log n)"
120
+ if normalized_keyword in normalized_answer:
121
+ matched_keywords.append(keyword)
122
+ else:
123
+ missing_keywords.append(keyword)
124
+
125
+ # Calculate proportional score
126
+ if len(keywords) == 0:
127
+ keyword_score = 0.5 # Neutral score if no keywords defined
128
+ else:
129
+ keyword_score = len(matched_keywords) / len(keywords)
130
+
131
+ return keyword_score, matched_keywords, missing_keywords
132
+
133
+ def evaluate_ai(self, answer: str, question: str = None) -> float:
134
+ """
135
+ Length-based evaluation (deterministic)
136
+
137
+ DETERMINISTIC implementation for OpenEnv compliance.
138
+ Scores based on answer length relative to ideal length.
139
+
140
+ Current implementation:
141
+ - Deterministic scoring based on answer length
142
+ - No randomness - same input always gives same output
143
+ - Considers completeness via length heuristic
144
+
145
+ Args:
146
+ answer: Candidate's answer
147
+ question: Original question (for context)
148
+
149
+ Returns:
150
+ Length score between 0.0 and 1.0
151
+ """
152
+ # DETERMINISTIC IMPLEMENTATION for OpenEnv compliance
153
+
154
+ # Ideal answer length: 100-300 characters
155
+ # Score based on how close to ideal range
156
+ if not answer or len(answer.strip()) < 10:
157
+ return 0.2
158
+
159
+ answer_length = len(answer.strip())
160
+
161
+ # Optimal range: 100-300 characters
162
+ if 100 <= answer_length <= 300:
163
+ return 1.0
164
+ elif answer_length < 100:
165
+ # Scale from 0.5 to 1.0 as length approaches 100
166
+ return 0.5 + 0.5 * (answer_length / 100.0)
167
+ else:
168
+ # Penalize excessively long answers (diminishing returns)
169
+ excess = answer_length - 300
170
+ penalty = min(excess / 500.0, 0.3) # Max 0.3 penalty
171
+ return max(0.7, 1.0 - penalty)
172
+
173
+
174
+
175
+ def evaluate(self, answer: str, keywords: List[str], question: str = None) -> Dict[str, Any]:
176
+ """
177
+ Complete hybrid evaluation
178
+
179
+ Combines keyword-based and AI-based scoring with configurable weights.
180
+
181
+ Args:
182
+ answer: Candidate's answer
183
+ keywords: Expected keywords for this question
184
+ question: Original question (optional, for AI judge)
185
+
186
+ Returns:
187
+ Dictionary containing:
188
+ - score: Final hybrid score (0.0 to 1.0)
189
+ - keyword_score: Keyword-based score
190
+ - ai_score: AI-based score
191
+ - matched_keywords: Keywords found
192
+ - missing_keywords: Keywords not found
193
+ - feedback: Human-readable feedback string
194
+ """
195
+ # Evaluate using both methods
196
+ keyword_score, matched, missing = self.evaluate_keywords(answer, keywords)
197
+ ai_score = self.evaluate_ai(answer, question)
198
+
199
+ # Compute weighted hybrid score
200
+ final_score = (self.keyword_weight * keyword_score) + (self.ai_weight * ai_score)
201
+
202
+ # Generate structured feedback
203
+ feedback = self._generate_feedback(
204
+ score=final_score,
205
+ keyword_score=keyword_score,
206
+ matched=matched,
207
+ missing=missing
208
+ )
209
+
210
+ return {
211
+ "score": final_score,
212
+ "keyword_score": keyword_score,
213
+ "ai_score": ai_score,
214
+ "matched_keywords": matched,
215
+ "missing_keywords": missing,
216
+ "feedback": feedback
217
+ }
218
+
219
+ def _generate_feedback(
220
+ self,
221
+ score: float,
222
+ keyword_score: float,
223
+ matched: List[str],
224
+ missing: List[str]
225
+ ) -> str:
226
+ """
227
+ Generate structured, actionable feedback
228
+
229
+ Feedback is designed to be:
230
+ - Clear and specific
231
+ - Actionable (tells what to add)
232
+ - Encouraging (positive framing)
233
+ - Machine-parseable (for agent improvement)
234
+
235
+ Args:
236
+ score: Final score
237
+ keyword_score: Keyword coverage score
238
+ matched: Matched keywords
239
+ missing: Missing keywords
240
+
241
+ Returns:
242
+ Formatted feedback string
243
+ """
244
+ total_keywords = len(matched) + len(missing)
245
+
246
+ # Score-based feedback tier
247
+ if score >= 0.8:
248
+ base = f"✅ Excellent answer! "
249
+ base += f"Covered {len(matched)}/{total_keywords} key concepts."
250
+ if matched:
251
+ base += f"\n ✓ Mentioned: {', '.join(matched)}"
252
+ return base
253
+
254
+ elif score >= 0.5:
255
+ base = f"👍 Good answer. "
256
+ base += f"Covered {len(matched)}/{total_keywords} concepts."
257
+ if matched:
258
+ base += f"\n ✓ Covered: {', '.join(matched)}"
259
+ if missing:
260
+ # Show up to 3 missing keywords for actionable feedback
261
+ base += f"\n ⚠ Consider adding: {', '.join(missing[:3])}"
262
+ return base
263
+
264
+ elif score >= 0.3:
265
+ base = f"⚠️ Partial answer. "
266
+ base += f"Only {len(matched)}/{total_keywords} concepts covered."
267
+ if matched:
268
+ base += f"\n ✓ Included: {', '.join(matched)}"
269
+ if missing:
270
+ base += f"\n ✗ Missing: {', '.join(missing)}"
271
+ return base
272
+
273
+ else:
274
+ base = f"❌ Weak answer. "
275
+ base += f"Missing {len(missing)}/{total_keywords} critical concepts."
276
+ if missing:
277
+ base += f"\n ✗ Must include: {', '.join(missing)}"
278
+ base += "\n 💡 Tip: Cover fundamental concepts first"
279
+ return base
280
+
281
+ def compute_reward(self, score: float) -> float:
282
+ """
283
+ Convert score to RL reward signal
284
+
285
+ OpenEnv-compliant reward function:
286
+ - Reward is IDENTICAL to score (0.0 to 1.0 range)
287
+ - Deterministic: same score always gives same reward
288
+ - Continuous: allows fine-grained learning signals
289
+
290
+ Args:
291
+ score: Evaluation score (0.0 to 1.0)
292
+
293
+ Returns:
294
+ Float reward value (0.0 to 1.0)
295
+ """
296
+ # OpenEnv compliance: reward = score (0.0 to 1.0)
297
+ return float(score)
298
+
299
+ def set_weights(self, keyword_weight: float, ai_weight: float):
300
+ """
301
+ Update scoring weights
302
+
303
+ Allows runtime adjustment of hybrid scoring balance.
304
+
305
+ Args:
306
+ keyword_weight: New keyword weight
307
+ ai_weight: New AI weight
308
+ """
309
+ self.keyword_weight = keyword_weight
310
+ self.ai_weight = ai_weight
backend/main.py ADDED
@@ -0,0 +1,332 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ FastAPI Backend for AI Interview Preparation RL Environment
3
+ OpenEnv-compliant API implementation
4
+ """
5
+
6
+ import os
7
+ from fastapi import FastAPI, HTTPException
8
+ from fastapi.middleware.cors import CORSMiddleware
9
+ from pydantic import BaseModel
10
+ from typing import Optional
11
+ from pathlib import Path
12
+ from dotenv import load_dotenv
13
+
14
+ # Load environment variables
15
+ load_dotenv()
16
+
17
+ print("🚀 Server starting...")
18
+
19
+ # Import from app package
20
+ from app.environment import InterviewEnv
21
+ from app.evaluator import Evaluator
22
+ from app.agent import InterviewAgent
23
+
24
+
25
+ # Initialize FastAPI app
26
+ app = FastAPI(title="AI Interview Prep RL Environment")
27
+
28
+ # CORS middleware for frontend
29
+ app.add_middleware(
30
+ CORSMiddleware,
31
+ allow_origins=["*"],
32
+ allow_credentials=True,
33
+ allow_methods=["*"],
34
+ allow_headers=["*"],
35
+ )
36
+
37
+ # Initialize components
38
+ evaluator = Evaluator()
39
+
40
+ # Resolve dataset path — works in both local (backend/) and Docker (/app/) contexts
41
+ questions_path = Path(__file__).resolve().parent / "app" / "dataset.json"
42
+ print(f"📂 Dataset path: {questions_path}")
43
+ print(f"📂 Dataset exists: {questions_path.exists()}")
44
+
45
+ env = InterviewEnv(questions_path=str(questions_path), evaluator=evaluator)
46
+
47
+ # Graceful agent init — don't crash server if API key is missing
48
+ try:
49
+ agent = InterviewAgent(mode="api") # Will use HF_TOKEN from .env
50
+ print(f"🤖 Agent initialized in mode: {agent.mode}")
51
+ except Exception as e:
52
+ print(f"⚠️ Agent init failed ({e}), using mock mode")
53
+ agent = InterviewAgent(mode="mock")
54
+
55
+ print("✅ All components initialized successfully")
56
+
57
+ # Global state
58
+ current_state = None
59
+ current_config = {
60
+ "api_key": None,
61
+ "model": "Qwen/Qwen3-Coder-Next:novita"
62
+ }
63
+
64
+
65
+ # Request/Response models
66
+ class StepRequest(BaseModel):
67
+ action: str
68
+
69
+
70
+ class AnswerRequest(BaseModel):
71
+ answer: str
72
+
73
+
74
+ class AutoRunRequest(BaseModel):
75
+ use_retry: bool = True
76
+
77
+
78
+ class ConfigRequest(BaseModel):
79
+ api_key: str
80
+ model: str = "Qwen/Qwen3-Coder-Next:novita"
81
+
82
+
83
+ @app.get("/")
84
+ def read_root():
85
+ """Root endpoint"""
86
+ return {
87
+ "message": "AI Interview Preparation RL Environment (OpenEnv Compliant)",
88
+ "status": "running",
89
+ "endpoints": ["/reset", "/step", "/state", "/question", "/answer", "/auto-run", "/stats", "/config"]
90
+ }
91
+
92
+
93
+ # OpenEnv-compliant endpoints
94
+
95
+ @app.post("/reset")
96
+ def reset():
97
+ """OpenEnv: Reset environment and get new question"""
98
+ global current_state
99
+
100
+ try:
101
+ current_state = env.reset()
102
+ # Return only the required fields for OpenEnv compliance
103
+ return {
104
+ "question": current_state["question"],
105
+ "difficulty": current_state["difficulty"],
106
+ "attempt": current_state["attempt"]
107
+ }
108
+ except Exception as e:
109
+ raise HTTPException(status_code=500, detail=str(e))
110
+
111
+
112
+ @app.post("/step")
113
+ def step(request: StepRequest):
114
+ """OpenEnv: Submit action and get reward"""
115
+ global current_state
116
+
117
+ if current_state is None:
118
+ raise HTTPException(status_code=400, detail="Environment not initialized. Call /reset first")
119
+
120
+ try:
121
+ result = env.step(request.action)
122
+
123
+ # Get current state
124
+ state_data = env.state()
125
+
126
+ # Return OpenEnv-compliant response
127
+ return {
128
+ "reward": result["reward"],
129
+ "done": result["done"],
130
+ "state": state_data
131
+ }
132
+ except Exception as e:
133
+ raise HTTPException(status_code=500, detail=str(e))
134
+
135
+
136
+ @app.get("/state")
137
+ def get_state():
138
+ """OpenEnv: Get current environment state"""
139
+ if current_state is None:
140
+ raise HTTPException(status_code=400, detail="Environment not initialized. Call /reset first")
141
+
142
+ try:
143
+ state_data = env.state()
144
+ return state_data
145
+ except Exception as e:
146
+ raise HTTPException(status_code=500, detail=str(e))
147
+
148
+
149
+ # Legacy endpoints (for backward compatibility with existing frontend)
150
+
151
+ @app.get("/question")
152
+ def get_question():
153
+ """Get a new interview question (calls env.reset())"""
154
+ global current_state
155
+
156
+ try:
157
+ current_state = env.reset()
158
+ return {
159
+ "status": "success",
160
+ "state": current_state
161
+ }
162
+ except Exception as e:
163
+ raise HTTPException(status_code=500, detail=str(e))
164
+
165
+
166
+ @app.post("/answer")
167
+ def submit_answer(request: AnswerRequest):
168
+ """Submit an answer and get evaluation (calls env.step())"""
169
+ global current_state
170
+
171
+ if current_state is None:
172
+ raise HTTPException(status_code=400, detail="No active question. Call /question first")
173
+
174
+ try:
175
+ result = env.step(request.answer)
176
+ return {
177
+ "status": "success",
178
+ "result": result
179
+ }
180
+ except Exception as e:
181
+ raise HTTPException(status_code=500, detail=str(e))
182
+
183
+
184
+ @app.post("/auto-run")
185
+ def auto_run(request: AutoRunRequest):
186
+ """Automatic RL episode - agent generates and improves answer"""
187
+ global current_state
188
+
189
+ try:
190
+ # Reset environment
191
+ current_state = env.reset()
192
+ question = current_state["question"]
193
+
194
+ # Agent generates initial answer
195
+ answer = agent.generate_answer(question)
196
+
197
+ # Evaluate
198
+ result = env.step(answer)
199
+
200
+ episode_data = {
201
+ "question": question,
202
+ "difficulty": current_state["difficulty"],
203
+ "attempt_1": {
204
+ "answer": answer,
205
+ "score": result["score"],
206
+ "reward": result["reward"],
207
+ "feedback": result["feedback"]
208
+ }
209
+ }
210
+
211
+ # Retry logic if enabled and reward is low
212
+ if request.use_retry and env.should_retry(result["reward"]):
213
+ # Reset for retry
214
+ current_state = {
215
+ "question": question,
216
+ "keywords": current_state["keywords"],
217
+ "difficulty": current_state["difficulty"]
218
+ }
219
+ env.current_question = {
220
+ "question": question,
221
+ "keywords": current_state["keywords"],
222
+ "difficulty": current_state["difficulty"]
223
+ }
224
+
225
+ # Generate improved answer with feedback
226
+ improved_answer = agent.generate_answer(
227
+ question,
228
+ feedback=result["feedback"],
229
+ missing_keywords=result.get("missing_keywords", [])
230
+ )
231
+
232
+ # Evaluate again
233
+ retry_result = env.step(improved_answer)
234
+
235
+ episode_data["attempt_2"] = {
236
+ "answer": improved_answer,
237
+ "score": retry_result["score"],
238
+ "reward": retry_result["reward"],
239
+ "feedback": retry_result["feedback"]
240
+ }
241
+
242
+ episode_data["improvement"] = retry_result["score"] - result["score"]
243
+
244
+ return {
245
+ "status": "success",
246
+ "episode": episode_data
247
+ }
248
+
249
+ except Exception as e:
250
+ raise HTTPException(status_code=500, detail=str(e))
251
+
252
+
253
+ @app.post("/config")
254
+ def update_config(request: ConfigRequest):
255
+ """Update API configuration (API key and model)"""
256
+ global agent, current_config
257
+
258
+ try:
259
+ # Update configuration
260
+ current_config["api_key"] = request.api_key
261
+ current_config["model"] = request.model
262
+
263
+ # Reinitialize agent with new configuration
264
+ agent = InterviewAgent(mode="api", api_key=request.api_key, model=request.model)
265
+
266
+ # Check if agent has proper client initialization
267
+ has_client = hasattr(agent, 'client') and agent.client is not None
268
+
269
+ return {
270
+ "status": "success",
271
+ "message": "Configuration updated successfully",
272
+ "config": {
273
+ "model": request.model,
274
+ "api_key_set": bool(request.api_key),
275
+ "client_initialized": has_client
276
+ }
277
+ }
278
+ except Exception as e:
279
+ raise HTTPException(status_code=500, detail=f"Configuration error: {str(e)}")
280
+
281
+
282
+ @app.get("/config")
283
+ def get_config():
284
+ """Get current API configuration"""
285
+ has_client = hasattr(agent, 'client') and agent.client is not None
286
+
287
+ return {
288
+ "status": "success",
289
+ "config": {
290
+ "model": current_config["model"],
291
+ "api_key_set": bool(current_config["api_key"]),
292
+ "client_initialized": has_client
293
+ }
294
+ }
295
+
296
+
297
+ @app.get("/stats")
298
+ def get_stats():
299
+ """Get episode statistics"""
300
+ try:
301
+ history = env.get_history()
302
+ if not history:
303
+ return {
304
+ "status": "success",
305
+ "stats": {
306
+ "total_attempts": 0,
307
+ "average_score": 0,
308
+ "average_reward": 0
309
+ }
310
+ }
311
+
312
+ avg_score = sum(h["score"] for h in history) / len(history)
313
+ avg_reward = sum(h["reward"] for h in history) / len(history)
314
+
315
+ return {
316
+ "status": "success",
317
+ "stats": {
318
+ "total_attempts": len(history),
319
+ "average_score": round(avg_score, 3),
320
+ "average_reward": round(avg_reward, 2),
321
+ "history": history
322
+ }
323
+ }
324
+ except Exception as e:
325
+ raise HTTPException(status_code=500, detail=str(e))
326
+
327
+
328
+ if __name__ == "__main__":
329
+ import uvicorn
330
+ port = int(os.getenv("PORT", "8000"))
331
+ print(f"🌐 Starting server on port {port}")
332
+ uvicorn.run(app, host="0.0.0.0", port=port)
backend/main_old.py ADDED
@@ -0,0 +1,291 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ FastAPI Backend for AI Interview Preparation RL Environment
3
+ """
4
+
5
+ from fastapi import FastAPI, HTTPException
6
+ from fastapi.middleware.cors import CORSMiddleware
7
+ from pydantic import BaseModel
8
+ from typing import Optional
9
+ import sys
10
+ from pathlib import Path
11
+ from dotenv import load_dotenv
12
+
13
+ # Load environment variables from .env file
14
+ load_dotenv()
15
+
16
+ # Add parent directory to path
17
+ sys.path.append(str(Path(__file__).parent))
18
+
19
+ from env.interview_env import InterviewEnv
20
+ from evaluator.evaluator import Evaluator
21
+ from agent.llm_agent import LLMAgent
22
+
23
+
24
+ # Initialize FastAPI app
25
+ app = FastAPI(title="AI Interview Prep RL Environment")
26
+
27
+ # CORS middleware for frontend
28
+ app.add_middleware(
29
+ CORSMiddleware,
30
+ allow_origins=["*"], # In production, specify exact origins
31
+ allow_credentials=True,
32
+ allow_methods=["*"],
33
+ allow_headers=["*"],
34
+ )
35
+
36
+ # Initialize components
37
+ evaluator = Evaluator()
38
+ env = InterviewEnv(evaluator=evaluator)
39
+ agent = LLMAgent()
40
+
41
+ # Global state (in production, use session management)
42
+ current_state = None
43
+ current_config = {
44
+ "api_key": None,
45
+ "model": "Qwen/Qwen3-Coder-Next:novita"
46
+ }
47
+
48
+
49
+ # Request/Response models
50
+ class AnswerRequest(BaseModel):
51
+ answer: str
52
+
53
+
54
+ class AutoRunRequest(BaseModel):
55
+ use_retry: bool = True
56
+
57
+
58
+ class ConfigRequest(BaseModel):
59
+ api_key: str
60
+ model: str = "Qwen/Qwen3-Coder-Next:novita"
61
+
62
+
63
+ @app.get("/")
64
+ def read_root():
65
+ """Root endpoint"""
66
+ return {
67
+ "message": "AI Interview Preparation RL Environment",
68
+ "status": "running",
69
+ "endpoints": ["/question", "/answer", "/auto-run", "/stats", "/config"]
70
+ }
71
+
72
+
73
+ @app.get("/question")
74
+ def get_question():
75
+ """
76
+ Get a new interview question (calls env.reset())
77
+
78
+ Returns:
79
+ {
80
+ "question": str,
81
+ "keywords": list,
82
+ "difficulty": str
83
+ }
84
+ """
85
+ global current_state
86
+
87
+ try:
88
+ current_state = env.reset()
89
+ return {
90
+ "status": "success",
91
+ "state": current_state
92
+ }
93
+ except Exception as e:
94
+ raise HTTPException(status_code=500, detail=str(e))
95
+
96
+
97
+ @app.post("/answer")
98
+ def submit_answer(request: AnswerRequest):
99
+ """
100
+ Submit an answer and get evaluation (calls env.step())
101
+
102
+ Args:
103
+ answer: The candidate's answer
104
+
105
+ Returns:
106
+ {
107
+ "reward": int,
108
+ "score": float,
109
+ "feedback": str,
110
+ "missing_keywords": list,
111
+ "done": bool
112
+ }
113
+ """
114
+ global current_state
115
+
116
+ if current_state is None:
117
+ raise HTTPException(
118
+ status_code=400,
119
+ detail="No active question. Call /question first"
120
+ )
121
+
122
+ try:
123
+ result = env.step(request.answer)
124
+ return {
125
+ "status": "success",
126
+ "result": result
127
+ }
128
+ except Exception as e:
129
+ raise HTTPException(status_code=500, detail=str(e))
130
+
131
+
132
+ @app.post("/auto-run")
133
+ def auto_run(request: AutoRunRequest):
134
+ """
135
+ Automatic RL loop: Get question → Agent generates answer → Evaluate → Optional retry
136
+
137
+ Args:
138
+ use_retry: Whether to retry on low scores
139
+
140
+ Returns:
141
+ Complete episode results with optional retry
142
+ """
143
+ global current_state
144
+
145
+ try:
146
+ # Reset environment and get question
147
+ current_state = env.reset()
148
+ question = current_state["question"]
149
+
150
+ # Agent generates initial answer
151
+ answer = agent.generate_answer(question)
152
+
153
+ # Evaluate
154
+ result = env.step(answer)
155
+
156
+ episode_data = {
157
+ "question": question,
158
+ "difficulty": current_state["difficulty"],
159
+ "attempt_1": {
160
+ "answer": answer,
161
+ "score": result["score"],
162
+ "reward": result["reward"],
163
+ "feedback": result["feedback"]
164
+ }
165
+ }
166
+
167
+ # Retry logic if enabled and reward is low
168
+ if request.use_retry and env.should_retry(result["reward"]):
169
+ # Reset for retry
170
+ current_state = {
171
+ "question": question,
172
+ "keywords": current_state["keywords"],
173
+ "difficulty": current_state["difficulty"]
174
+ }
175
+ env.current_question = {
176
+ "question": question,
177
+ "keywords": current_state["keywords"],
178
+ "difficulty": current_state["difficulty"]
179
+ }
180
+
181
+ # Generate improved answer with feedback
182
+ improved_answer = agent.generate_answer(question, result["feedback"])
183
+
184
+ # Evaluate again
185
+ retry_result = env.step(improved_answer)
186
+
187
+ episode_data["attempt_2"] = {
188
+ "answer": improved_answer,
189
+ "score": retry_result["score"],
190
+ "reward": retry_result["reward"],
191
+ "feedback": retry_result["feedback"]
192
+ }
193
+
194
+ episode_data["improvement"] = retry_result["score"] - result["score"]
195
+
196
+ return {
197
+ "status": "success",
198
+ "episode": episode_data
199
+ }
200
+
201
+ except Exception as e:
202
+ raise HTTPException(status_code=500, detail=str(e))
203
+
204
+
205
+ @app.post("/config")
206
+ def update_config(request: ConfigRequest):
207
+ """
208
+ Update API configuration (API key and model)
209
+
210
+ Args:
211
+ api_key: Hugging Face API token
212
+ model: Model name/endpoint to use
213
+
214
+ Returns:
215
+ Configuration status
216
+ """
217
+ global agent, current_config
218
+
219
+ try:
220
+ # Update configuration
221
+ current_config["api_key"] = request.api_key
222
+ current_config["model"] = request.model
223
+
224
+ # Reinitialize agent with new configuration
225
+ agent = LLMAgent(api_key=request.api_key, model=request.model)
226
+
227
+ return {
228
+ "status": "success",
229
+ "message": "Configuration updated successfully",
230
+ "config": {
231
+ "model": request.model,
232
+ "api_key_set": bool(request.api_key),
233
+ "client_initialized": bool(agent.client)
234
+ }
235
+ }
236
+ except Exception as e:
237
+ raise HTTPException(status_code=500, detail=f"Configuration error: {str(e)}")
238
+
239
+
240
+ @app.get("/config")
241
+ def get_config():
242
+ """
243
+ Get current API configuration
244
+
245
+ Returns:
246
+ Current configuration (without exposing full API key)
247
+ """
248
+ return {
249
+ "status": "success",
250
+ "config": {
251
+ "model": current_config["model"],
252
+ "api_key_set": bool(current_config["api_key"]),
253
+ "client_initialized": bool(agent.client)
254
+ }
255
+ }
256
+
257
+
258
+ @app.get("/stats")
259
+ def get_stats():
260
+ """Get episode statistics"""
261
+ try:
262
+ history = env.get_history()
263
+ if not history:
264
+ return {
265
+ "status": "success",
266
+ "stats": {
267
+ "total_attempts": 0,
268
+ "average_score": 0,
269
+ "average_reward": 0
270
+ }
271
+ }
272
+
273
+ avg_score = sum(h["score"] for h in history) / len(history)
274
+ avg_reward = sum(h["reward"] for h in history) / len(history)
275
+
276
+ return {
277
+ "status": "success",
278
+ "stats": {
279
+ "total_attempts": len(history),
280
+ "average_score": round(avg_score, 3),
281
+ "average_reward": round(avg_reward, 2),
282
+ "history": history
283
+ }
284
+ }
285
+ except Exception as e:
286
+ raise HTTPException(status_code=500, detail=str(e))
287
+
288
+
289
+ if __name__ == "__main__":
290
+ import uvicorn
291
+ uvicorn.run(app, host="0.0.0.0", port=8000)
backend/requirements.txt ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ fastapi==0.104.1
2
+ uvicorn==0.24.0
3
+ pydantic==2.5.0
4
+ openai==1.57.4
5
+ python-multipart==0.0.6
6
+ python-dotenv==1.0.0
backend/test_env.py ADDED
@@ -0,0 +1,96 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Test script to verify the RL environment works correctly
3
+ Run this before starting the full application
4
+ """
5
+
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ # Add parent directory to path
10
+ sys.path.append(str(Path(__file__).parent))
11
+
12
+ from env.interview_env import InterviewEnv
13
+ from evaluator.evaluator import Evaluator
14
+ from agent.llm_agent import LLMAgent
15
+
16
+
17
+ def test_environment():
18
+ """Test the RL environment"""
19
+ print("=" * 60)
20
+ print("🧪 TESTING RL ENVIRONMENT")
21
+ print("=" * 60)
22
+
23
+ # Initialize components
24
+ evaluator = Evaluator()
25
+ env = InterviewEnv(evaluator=evaluator)
26
+ agent = LLMAgent()
27
+
28
+ print("\n✅ Components initialized successfully\n")
29
+
30
+ # Test 1: Reset environment
31
+ print("📝 Test 1: Environment Reset")
32
+ print("-" * 60)
33
+ state = env.reset()
34
+ print(f"Question: {state['question']}")
35
+ print(f"Difficulty: {state['difficulty']}")
36
+ print(f"Keywords: {', '.join(state['keywords'])}")
37
+ print("✅ Reset successful\n")
38
+
39
+ # Test 2: Evaluate a good answer
40
+ print("📝 Test 2: Evaluate Good Answer")
41
+ print("-" * 60)
42
+ good_answer = " ".join(state['keywords'][:3]) # Include some keywords
43
+ result = env.step(good_answer)
44
+ print(f"Answer: {good_answer}")
45
+ print(f"Score: {result['score']:.2%}")
46
+ print(f"Reward: {result['reward']}")
47
+ print(f"Feedback: {result['feedback']}")
48
+ print("✅ Evaluation successful\n")
49
+
50
+ # Test 3: Generate AI answer
51
+ print("📝 Test 3: AI Agent Answer Generation")
52
+ print("-" * 60)
53
+ env.reset() # Get new question
54
+ question = env.current_question['question']
55
+ ai_answer = agent.generate_answer(question)
56
+ print(f"Question: {question}")
57
+ print(f"AI Answer: {ai_answer}")
58
+ result = env.step(ai_answer)
59
+ print(f"Score: {result['score']:.2%}")
60
+ print(f"Reward: {result['reward']}")
61
+ print(f"Feedback: {result['feedback']}")
62
+ print("✅ AI generation successful\n")
63
+
64
+ # Test 4: Retry mechanism
65
+ print("📝 Test 4: Retry Mechanism")
66
+ print("-" * 60)
67
+ if env.should_retry(result['reward']):
68
+ print("⚠️ Low reward detected - initiating retry")
69
+ improved_answer = agent.generate_answer(question, result['feedback'])
70
+
71
+ # Reset to same question for retry
72
+ env.current_question = {
73
+ "question": question,
74
+ "keywords": state['keywords'],
75
+ "difficulty": state['difficulty']
76
+ }
77
+
78
+ retry_result = env.step(improved_answer)
79
+ print(f"Improved Answer: {improved_answer}")
80
+ print(f"New Score: {retry_result['score']:.2%}")
81
+ print(f"New Reward: {retry_result['reward']}")
82
+ improvement = retry_result['score'] - result['score']
83
+ print(f"Improvement: {improvement:+.2%}")
84
+ print("✅ Retry successful\n")
85
+ else:
86
+ print("✅ Score was good - no retry needed\n")
87
+
88
+ print("=" * 60)
89
+ print("✅ ALL TESTS PASSED!")
90
+ print("=" * 60)
91
+ print("\nYou can now start the FastAPI server with:")
92
+ print(" python main.py")
93
+
94
+
95
+ if __name__ == "__main__":
96
+ test_environment()
inference.py ADDED
@@ -0,0 +1,93 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ OpenEnv-compliant inference script
3
+ Runs complete episode over all tasks with exact logging format
4
+ """
5
+
6
+ import os
7
+ import sys
8
+ import json
9
+ from pathlib import Path
10
+
11
+ # Add backend to path
12
+ sys.path.insert(0, str(Path(__file__).parent / "backend"))
13
+
14
+ from app.environment import InterviewEnv
15
+ from app.evaluator import Evaluator
16
+ from app.agent import InterviewAgent
17
+
18
+
19
+ def run_inference():
20
+ """
21
+ Run inference over all tasks with OpenEnv-compliant logging
22
+
23
+ Environment variables:
24
+ - API_BASE_URL: LLM API endpoint
25
+ - MODEL_NAME: Model identifier
26
+ - HF_TOKEN: API authentication token
27
+ """
28
+
29
+ # Get environment variables
30
+ api_base_url = os.getenv("API_BASE_URL", "https://router.huggingface.co/v1")
31
+ model_name = os.getenv("MODEL_NAME", "Qwen/Qwen3-Coder-Next:novita")
32
+ hf_token = os.getenv("HF_TOKEN")
33
+
34
+ # Initialize components
35
+ evaluator = Evaluator()
36
+ questions_path = Path(__file__).parent / "backend" / "app" / "dataset.json"
37
+ env = InterviewEnv(questions_path=str(questions_path), evaluator=evaluator)
38
+
39
+ # Initialize agent (will use mock mode if no token)
40
+ if hf_token:
41
+ agent = InterviewAgent(mode="api", api_key=hf_token, model=model_name)
42
+ else:
43
+ print("⚠️ No HF_TOKEN found, using mock mode")
44
+ agent = InterviewAgent(mode="mock")
45
+
46
+ # Load all tasks
47
+ with open(questions_path, 'r') as f:
48
+ tasks = json.load(f)
49
+
50
+ print(f"Running inference on {len(tasks)} tasks")
51
+ print(f"API Base URL: {api_base_url}")
52
+ print(f"Model: {model_name}")
53
+ print("-" * 80)
54
+
55
+ # Run inference on each task
56
+ for task_idx, task in enumerate(tasks):
57
+ task_id = f"task_{task_idx}"
58
+
59
+ # Print START marker
60
+ print(f"[START]")
61
+ print(f"task_id={task_id}")
62
+
63
+ # Reset environment to this specific task
64
+ env.current_question = task
65
+ env.episode_history = []
66
+ env.retry_count = 0
67
+
68
+ question = task["question"]
69
+
70
+ # Generate answer
71
+ answer = agent.generate_answer(question)
72
+
73
+ # Print STEP marker with action
74
+ print(f"[STEP]")
75
+ print(f"action={answer}")
76
+
77
+ # Evaluate
78
+ result = env.step(answer)
79
+ reward = result["reward"]
80
+
81
+ # Print reward
82
+ print(f"reward={reward}")
83
+
84
+ # Print END marker
85
+ print(f"[END]")
86
+ print()
87
+
88
+ print("-" * 80)
89
+ print(f"✅ Inference complete: {len(tasks)} tasks processed")
90
+
91
+
92
+ if __name__ == "__main__":
93
+ run_inference()
openenv.yaml ADDED
@@ -0,0 +1,111 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ name: ai-interview-prep-rl-env
2
+ description: Reinforcement Learning environment for AI-powered technical interview preparation with keyword-based evaluation and retry mechanism
3
+
4
+ endpoints:
5
+ reset:
6
+ method: POST
7
+ path: /reset
8
+ description: Reset environment and get new interview question
9
+ response_schema:
10
+ type: object
11
+ properties:
12
+ question:
13
+ type: string
14
+ description: Interview question text
15
+ difficulty:
16
+ type: string
17
+ description: Question difficulty level (easy, medium, hard)
18
+ attempt:
19
+ type: integer
20
+ description: Current attempt number (0 after reset)
21
+ required:
22
+ - question
23
+ - difficulty
24
+ - attempt
25
+
26
+ step:
27
+ method: POST
28
+ path: /step
29
+ description: Submit answer and get evaluation with reward
30
+ request_schema:
31
+ type: object
32
+ properties:
33
+ action:
34
+ type: string
35
+ description: Candidate's answer to the question
36
+ required:
37
+ - action
38
+ response_schema:
39
+ type: object
40
+ properties:
41
+ reward:
42
+ type: number
43
+ description: Reward signal between 0.0 and 1.0
44
+ minimum: 0.0
45
+ maximum: 1.0
46
+ done:
47
+ type: boolean
48
+ description: Whether episode is complete
49
+ state:
50
+ type: object
51
+ properties:
52
+ question:
53
+ type: string
54
+ difficulty:
55
+ type: string
56
+ attempt:
57
+ type: integer
58
+ required:
59
+ - question
60
+ - difficulty
61
+ - attempt
62
+ required:
63
+ - reward
64
+ - done
65
+ - state
66
+
67
+ state:
68
+ method: GET
69
+ path: /state
70
+ description: Get current environment state
71
+ response_schema:
72
+ type: object
73
+ properties:
74
+ question:
75
+ type: string
76
+ description: Current interview question
77
+ difficulty:
78
+ type: string
79
+ description: Question difficulty level
80
+ attempt:
81
+ type: integer
82
+ description: Current attempt number
83
+ required:
84
+ - question
85
+ - difficulty
86
+ - attempt
87
+
88
+ tasks:
89
+ count: 12
90
+ source: backend/app/dataset.json
91
+ description: Technical interview questions covering algorithms, systems, databases, and software design
92
+
93
+ grading:
94
+ type: hybrid
95
+ deterministic: true
96
+ components:
97
+ - keyword_matching: 0.8
98
+ - length_scoring: 0.2
99
+ output_range:
100
+ min: 0.0
101
+ max: 1.0
102
+
103
+ environment:
104
+ type: single_question_episode
105
+ max_retries: 3
106
+ state_representation:
107
+ - question: string
108
+ - difficulty: string
109
+ - attempt: integer
110
+ action_space: text
111
+ reward_range: [0.0, 1.0]