jatin gyass commited on
Commit
4db2d34
Β·
1 Parent(s): 2eef9ea

update for the new web data source

Browse files
backend/agents/critic.py CHANGED
@@ -16,11 +16,12 @@ errors compared to self-evaluation by the same model.
16
  """
17
  from __future__ import annotations
18
  import json
 
19
  from datetime import datetime, timezone
20
 
21
- from langchain_google_genai import ChatGoogleGenerativeAI
22
  from langchain_core.messages import SystemMessage, HumanMessage
23
 
 
24
  from ..state.graph_state import (
25
  WorkflowState, TaskStatus, AgentRole, make_agent_event
26
  )
@@ -29,71 +30,16 @@ from ..core.logger import get_logger
29
 
30
  log = get_logger(__name__)
31
 
32
- CRITIC_SYSTEM = """You are a rigorous quality critic for AI task execution. Your job is to evaluate whether a task was completed correctly and completely.
33
-
34
- ## Evaluation criteria
35
- 1. **Completeness** (0–25): Did the agent address ALL aspects of the task?
36
- 2. **Accuracy** (0–25): Are the facts/calculations correct? Are there hallucinations?
37
- 3. **Usefulness** (0–25): Is the output actually helpful for the user's goal?
38
- 4. **Quality** (0–25): Is the output well-structured, clear, and actionable?
39
-
40
- ## Decision thresholds
41
- - Score β‰₯ 80: APPROVE β€” task well completed
42
- - Score 60–79: APPROVE with suggestions
43
- - Score < 60: REQUEST_REPLAN β€” major issues need fixing
44
-
45
- ## Output format (JSON only)
46
- {
47
- "thinking": "Your detailed evaluation reasoning",
48
- "score": 0-100,
49
- "decision": "approve" | "request_replan",
50
- "critique": "Specific issues found (if any)",
51
- "suggestions": ["suggestion 1", "suggestion 2"],
52
- "completeness_score": 0-25,
53
- "accuracy_score": 0-25,
54
- "usefulness_score": 0-25,
55
- "quality_score": 0-25
56
- }
57
-
58
- ## Critical rules
59
- - Be HONEST and SPECIFIC β€” vague critiques don't help
60
- - If output is good, approve it β€” don't invent problems
61
- - Only request replan for REAL deficiencies
62
- - Point to SPECIFIC missing or incorrect elements"""
63
-
64
-
65
- CRITIC_TOOL = {
66
- "type": "function",
67
- "function": {
68
- "name": "submit_critique",
69
- "description": "Submit quality evaluation and decision",
70
- "parameters": {
71
- "type": "object",
72
- "required": ["thinking", "score", "decision", "critique"],
73
- "properties": {
74
- "thinking": {"type": "string"},
75
- "score": {"type": "integer", "minimum": 0, "maximum": 100},
76
- "decision": {"type": "string", "enum": ["approve", "request_replan"]},
77
- "critique": {"type": "string"},
78
- "suggestions": {"type": "array", "items": {"type": "string"}},
79
- "completeness_score": {"type": "integer", "minimum": 0, "maximum": 25},
80
- "accuracy_score": {"type": "integer", "minimum": 0, "maximum": 25},
81
- "usefulness_score": {"type": "integer", "minimum": 0, "maximum": 25},
82
- "quality_score": {"type": "integer", "minimum": 0, "maximum": 25},
83
- },
84
- },
85
- },
86
- }
87
 
88
 
89
  def critic_node(state: WorkflowState) -> WorkflowState:
90
- """LangGraph node β€” runs the Critic agent for reflection."""
91
  settings = get_settings()
92
- llm = ChatGoogleGenerativeAI(
93
- model=settings.critic_model,
94
- temperature=0.1,
95
- google_api_key=settings.google_api_key,
96
- )
97
 
98
  log.info("Critic running", task_id=state["task_id"])
99
 
@@ -121,45 +67,38 @@ def critic_node(state: WorkflowState) -> WorkflowState:
121
  result_str = json.dumps(result, default=str)
122
  results_summary[step_id] = result_str[:400]
123
 
124
- eval_context = f"""
125
- ORIGINAL TASK:
126
- {state["task"]}
127
-
128
- EXECUTION PLAN ({len(completed_steps)}/{len(state["plan"])} steps completed):
129
- {chr(10).join(f" [{s['status']}] {s['title']}: {s['description'][:100]}" for s in state["plan"])}
130
-
131
- FAILED STEPS: {len(failed_steps)}
132
- {chr(10).join(f" - {s['title']}: {s.get('error', '')}" for s in failed_steps) if failed_steps else " None"}
133
-
134
- STEP RESULTS:
135
- {json.dumps(results_summary, indent=2)[:2000]}
136
-
137
- FINAL OUTPUT:
138
- {state.get("final_output", "NOT YET GENERATED")[:1500]}
139
- """
140
 
141
  try:
 
142
  response = llm.invoke(
143
  [SystemMessage(content=CRITIC_SYSTEM), HumanMessage(content=eval_context)],
144
- tools=[CRITIC_TOOL],
145
  )
146
 
147
  tokens = response.usage_metadata.get("total_tokens", 0) if response.usage_metadata else 0
148
- tool_call = response.tool_calls[0] if response.tool_calls else None
149
-
150
- # Gemini Flash Lite sometimes returns JSON text instead of a tool call β€” parse it as fallback
151
- if not tool_call:
152
- text = response.content or ""
153
- # Strip markdown code fences if present
154
- stripped = text.strip().removeprefix("```json").removeprefix("```").removesuffix("```").strip()
155
- try:
156
- parsed = json.loads(stripped)
157
- tool_call = {"args": parsed}
158
- log.info("Critic: parsed JSON text fallback")
159
- except Exception:
160
- raise ValueError(f"No critique tool call and JSON parse failed. Response: {text[:200]}")
161
-
162
- c = tool_call["args"]
 
 
163
  score = int(c.get("score", 70))
164
  decision = c.get("decision", "approve")
165
  critique = c.get("critique", "")
 
16
  """
17
  from __future__ import annotations
18
  import json
19
+ import re
20
  from datetime import datetime, timezone
21
 
 
22
  from langchain_core.messages import SystemMessage, HumanMessage
23
 
24
+ from ..core.llm import get_llm
25
  from ..state.graph_state import (
26
  WorkflowState, TaskStatus, AgentRole, make_agent_event
27
  )
 
30
 
31
  log = get_logger(__name__)
32
 
33
+ CRITIC_SYSTEM = """Evaluate task completion. Score 0-100. Approve if score>=60, else request_replan.
34
+
35
+ You MUST respond with valid JSON only - no extra text, no markdown fences:
36
+ {"thinking":"one line","score":75,"decision":"approve","critique":"none","suggestions":[]}"""
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
37
 
38
 
39
  def critic_node(state: WorkflowState) -> WorkflowState:
40
+ """LangGraph node - runs the Critic agent for reflection."""
41
  settings = get_settings()
42
+ llm = get_llm("critic", temperature=0.1)
 
 
 
 
43
 
44
  log.info("Critic running", task_id=state["task_id"])
45
 
 
67
  result_str = json.dumps(result, default=str)
68
  results_summary[step_id] = result_str[:400]
69
 
70
+ steps_summary = ", ".join(f"[{s['status']}]{s['title']}" for s in state["plan"])
71
+ eval_context = (
72
+ f"TASK: {state['task']}\n"
73
+ f"STEPS: {steps_summary}\n"
74
+ f"FAILED: {len(failed_steps)}\n"
75
+ f"OUTPUT: {str(state.get('final_output', 'none'))[:600]}"
76
+ )
 
 
 
 
 
 
 
 
 
77
 
78
  try:
79
+ # No tool calling β€” plain JSON text is more reliable across providers
80
  response = llm.invoke(
81
  [SystemMessage(content=CRITIC_SYSTEM), HumanMessage(content=eval_context)],
 
82
  )
83
 
84
  tokens = response.usage_metadata.get("total_tokens", 0) if response.usage_metadata else 0
85
+ text = response.content or ""
86
+
87
+ # Strip markdown fences and extract JSON
88
+ stripped = text.strip()
89
+ for fence in ("```json", "```"):
90
+ stripped = stripped.removeprefix(fence)
91
+ stripped = stripped.removesuffix("```").strip()
92
+
93
+ # Find first {...} block in case model adds prose
94
+ match = re.search(r'\{.*\}', stripped, re.DOTALL)
95
+ if match:
96
+ stripped = match.group(0)
97
+
98
+ try:
99
+ c = json.loads(stripped)
100
+ except Exception:
101
+ raise ValueError(f"Critic JSON parse failed. Response: {text[:200]}")
102
  score = int(c.get("score", 70))
103
  decision = c.get("decision", "approve")
104
  critique = c.get("critique", "")
backend/agents/executor.py CHANGED
@@ -17,9 +17,9 @@ import json
17
  import time
18
  from datetime import datetime, timezone
19
 
20
- from langchain_google_genai import ChatGoogleGenerativeAI
21
  from langchain_core.messages import SystemMessage, HumanMessage
22
 
 
23
  from ..state.graph_state import (
24
  WorkflowState, TaskStatus, StepStatus, AgentRole, make_agent_event
25
  )
@@ -29,30 +29,10 @@ from ..core.logger import get_logger
29
 
30
  log = get_logger(__name__)
31
 
32
- EXECUTOR_SYSTEM = """You are a precise task executor. You execute ONE step at a time using the available tools.
33
 
34
- ## Your job
35
- 1. Understand exactly what the current step requires
36
- 2. Choose the RIGHT tool and RIGHT arguments
37
- 3. If tool results are insufficient, try a different approach
38
- 4. Be specific β€” vague tool inputs produce vague results
39
-
40
- ## Key rules
41
- - Read the step description CAREFULLY
42
- - Use context from previous step results when relevant
43
- - For "synthesize" steps: write_file first, then provide the complete answer
44
- - If a tool returns an error, try a different approach
45
- - Output only what is needed β€” do not hallucinate results
46
-
47
- ## Response format
48
- Always call a tool. Do not respond without calling a tool."""
49
-
50
- SYNTHESIZE_SYSTEM = """You are a synthesis agent. Your job is to produce the FINAL comprehensive answer.
51
-
52
- Using all the results collected during task execution, write a complete, well-structured response to the original task.
53
-
54
- Format your response clearly. Use headers, bullet points, or numbered lists where appropriate.
55
- Be thorough but concise. Cite specific findings where relevant."""
56
 
57
 
58
  def _build_context(state: WorkflowState) -> str:
@@ -105,11 +85,7 @@ def executor_node(state: WorkflowState) -> WorkflowState:
105
  else:
106
  updated_plan.append(s)
107
 
108
- llm = ChatGoogleGenerativeAI(
109
- model=settings.executor_model,
110
- temperature=0.1,
111
- google_api_key=settings.google_api_key,
112
- )
113
 
114
  context = _build_context(state)
115
  user_msg = (
@@ -187,11 +163,7 @@ async def _safe_tool_call(name: str, args: dict) -> dict:
187
 
188
  def _handle_synthesize(state: WorkflowState, step: dict, settings) -> WorkflowState:
189
  """Handle the special synthesize step β€” produces final answer."""
190
- llm = ChatGoogleGenerativeAI(
191
- model=settings.executor_model,
192
- temperature=0.3,
193
- google_api_key=settings.google_api_key,
194
- )
195
 
196
  context = _build_context(state)
197
  response = llm.invoke([
 
17
  import time
18
  from datetime import datetime, timezone
19
 
 
20
  from langchain_core.messages import SystemMessage, HumanMessage
21
 
22
+ from ..core.llm import get_llm
23
  from ..state.graph_state import (
24
  WorkflowState, TaskStatus, StepStatus, AgentRole, make_agent_event
25
  )
 
29
 
30
  log = get_logger(__name__)
31
 
32
+ EXECUTOR_SYSTEM = """Execute ONE step using the available tools. Choose the right tool and args. Always call a tool."""
33
 
34
+ SYNTHESIZE_SYSTEM = """Write the final answer to the task using the collected results. Be concise and clear.
35
+ Use plain text only - no LaTeX, no $\\boxed{}, no math notation. Write numbers as regular text (e.g. "1040" not "$\\boxed{1040}$")."""
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36
 
37
 
38
  def _build_context(state: WorkflowState) -> str:
 
85
  else:
86
  updated_plan.append(s)
87
 
88
+ llm = get_llm("executor", temperature=0.1)
 
 
 
 
89
 
90
  context = _build_context(state)
91
  user_msg = (
 
163
 
164
  def _handle_synthesize(state: WorkflowState, step: dict, settings) -> WorkflowState:
165
  """Handle the special synthesize step β€” produces final answer."""
166
+ llm = get_llm("executor", temperature=0.3)
 
 
 
 
167
 
168
  context = _build_context(state)
169
  response = llm.invoke([
backend/agents/memory_agent.py CHANGED
@@ -14,12 +14,8 @@ Memory types:
14
  """
15
  from __future__ import annotations
16
  import asyncio
17
- import json
18
  from datetime import datetime, timezone
19
 
20
- from langchain_google_genai import ChatGoogleGenerativeAI
21
- from langchain_core.messages import SystemMessage, HumanMessage
22
-
23
  from ..state.graph_state import WorkflowState, AgentRole, make_agent_event, make_memory_entry
24
  from ..memory.memory_store import long_term, short_term
25
  from ..core.config import get_settings
@@ -27,31 +23,6 @@ from ..core.logger import get_logger
27
 
28
  log = get_logger(__name__)
29
 
30
- MEMORY_RETRIEVE_SYSTEM = """You are a memory retrieval assistant. Given a task, identify what past knowledge would be most relevant.
31
-
32
- Output a JSON list of search queries to retrieve relevant memories:
33
- {
34
- "queries": ["query 1", "query 2", "query 3"]
35
- }
36
-
37
- Focus on: similar past tasks, relevant domain knowledge, successful strategies."""
38
-
39
- MEMORY_STORE_SYSTEM = """You are a memory extraction assistant. Given a completed task execution, extract valuable learnings to remember for the future.
40
-
41
- Output a JSON list of memories to store:
42
- {
43
- "memories": [
44
- {
45
- "content": "The memory content β€” be specific and actionable",
46
- "memory_type": "episodic|semantic|procedural",
47
- "importance": 0.0-1.0,
48
- "tags": ["tag1", "tag2"]
49
- }
50
- ]
51
- }
52
-
53
- Extract: key findings, successful approaches, useful facts, things to avoid."""
54
-
55
 
56
  def memory_retrieve_node(state: WorkflowState) -> WorkflowState:
57
  """Run at start of workflow β€” retrieve relevant memories."""
@@ -67,19 +38,8 @@ def memory_retrieve_node(state: WorkflowState) -> WorkflowState:
67
  log.info("Restored from short-term cache")
68
  return {**state, **cached_state, "task_id": state["task_id"]}
69
 
70
- try:
71
- llm = ChatGoogleGenerativeAI(
72
- model=settings.memory_model, temperature=0,
73
- google_api_key=settings.google_api_key,
74
- )
75
- response = llm.invoke([
76
- SystemMessage(content=MEMORY_RETRIEVE_SYSTEM),
77
- HumanMessage(content=f"Task: {state['task']}"),
78
- ])
79
- parsed = json.loads(response.content)
80
- queries = parsed.get("queries", [state["task"][:50]])
81
- except Exception:
82
- queries = [state["task"][:50]]
83
 
84
  # Retrieve from long-term memory
85
  all_memories = []
@@ -133,30 +93,7 @@ def memory_store_node(state: WorkflowState) -> WorkflowState:
133
  except Exception as e:
134
  log.warning("Memory store failed", error=str(e))
135
 
136
- # Also extract memories from execution via LLM (if task completed well)
137
- if (state.get("quality_score", 0) or 0) >= 70 and state.get("final_output"):
138
- try:
139
- llm = ChatGoogleGenerativeAI(
140
- model=settings.memory_model, temperature=0,
141
- google_api_key=settings.google_api_key,
142
- )
143
- context = f"""Task: {state['task']}
144
- Final output (first 800 chars): {str(state.get('final_output', ''))[:800]}
145
- Quality score: {state.get('quality_score', 0)}
146
- Steps taken: {', '.join(s['title'] for s in state['plan'] if s['status']=='done')}"""
147
-
148
- response = llm.invoke([
149
- SystemMessage(content=MEMORY_STORE_SYSTEM),
150
- HumanMessage(content=context),
151
- ])
152
-
153
- parsed = json.loads(response.content)
154
- for mem in parsed.get("memories", [])[:4]:
155
- asyncio.run(long_term.store(mem, task_id=state["task_id"]))
156
- memories_stored += 1
157
-
158
- except Exception as e:
159
- log.warning("Auto-memory extraction failed", error=str(e))
160
 
161
  log.info("Memories stored", count=memories_stored)
162
 
 
14
  """
15
  from __future__ import annotations
16
  import asyncio
 
17
  from datetime import datetime, timezone
18
 
 
 
 
19
  from ..state.graph_state import WorkflowState, AgentRole, make_agent_event, make_memory_entry
20
  from ..memory.memory_store import long_term, short_term
21
  from ..core.config import get_settings
 
23
 
24
  log = get_logger(__name__)
25
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
26
 
27
  def memory_retrieve_node(state: WorkflowState) -> WorkflowState:
28
  """Run at start of workflow β€” retrieve relevant memories."""
 
38
  log.info("Restored from short-term cache")
39
  return {**state, **cached_state, "task_id": state["task_id"]}
40
 
41
+ # Use task text directly β€” no LLM call needed to generate queries
42
+ queries = [state["task"][:80]]
 
 
 
 
 
 
 
 
 
 
 
43
 
44
  # Retrieve from long-term memory
45
  all_memories = []
 
93
  except Exception as e:
94
  log.warning("Memory store failed", error=str(e))
95
 
96
+ # No extra LLM call β€” critic already added memory entries to new_memories
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
97
 
98
  log.info("Memories stored", count=memories_stored)
99
 
backend/agents/planner.py CHANGED
@@ -17,70 +17,32 @@ import json
17
  import uuid
18
  from datetime import datetime, timezone
19
 
20
- from langchain_google_genai import ChatGoogleGenerativeAI
21
  from langchain_core.messages import SystemMessage, HumanMessage
22
 
 
23
  from ..state.graph_state import (
24
  WorkflowState, TaskStatus, AgentRole,
25
  make_plan_step, make_agent_event,
26
  )
27
- from ..core.config import get_settings
28
  from ..core.logger import get_logger
29
 
30
  log = get_logger(__name__)
31
 
32
- PLANNER_SYSTEM = """You are a strategic task planner AI. Your job is to decompose complex tasks into clear, executable steps.
33
-
34
- ## Your responsibilities
35
- 1. Analyze the task and identify sub-goals
36
- 2. Create a MINIMAL but COMPLETE plan (3–8 steps max)
37
- 3. Assign the right tool to each step
38
- 4. Specify dependencies between steps
39
- 5. Be concrete β€” vague steps fail
40
-
41
- ## Available tools for steps
42
- - web_search: Search for information online
43
- - fetch_url: Read content from a specific URL
44
- - calculate: Perform mathematical calculations
45
- - run_python: Execute Python code for data processing
46
- - write_file: Save results or drafts to files
47
- - read_file: Read previously saved content
48
- - get_datetime: Get current date/time
49
- - synthesize: Combine all results into final answer (ALWAYS last step)
50
-
51
- ## Output format (JSON)
52
- Return ONLY a JSON object:
53
- {
54
- "thinking": "Brief analysis of the task",
55
- "plan": [
56
- {
57
- "step_id": "step_1",
58
- "title": "Short title",
59
- "description": "What to do and why",
60
- "tool": "tool_name",
61
- "depends_on": []
62
- }
63
- ],
64
- "estimated_complexity": "low|medium|high"
65
- }
66
 
67
- ## Rules
68
- - First step depends_on: []
69
- - Later steps list their prerequisites in depends_on
70
- - Always end with a "synthesize" step that depends on all prior steps
71
- - If replanning after critique, address the specific issues raised
72
- - NEVER include steps for things already completed"""
73
 
 
 
74
 
75
- REPLAN_ADDITION = """
76
- ## Replanning context
77
- Previous plan failed or received critique. Address these issues:
78
- {critique}
79
 
80
- Completed steps so far:
81
- {completed_steps}
82
 
83
- Do NOT repeat completed steps. Only plan remaining/corrective actions."""
 
 
 
84
 
85
 
86
  PLANNING_TOOL = {
@@ -119,12 +81,9 @@ def planner_node(state: WorkflowState) -> WorkflowState:
119
  LangGraph node β€” runs the Planner agent.
120
  Called at start and whenever Critic requests replanning.
121
  """
 
122
  settings = get_settings()
123
- llm = ChatGoogleGenerativeAI(
124
- model=settings.planner_model,
125
- temperature=0.2,
126
- google_api_key=settings.google_api_key,
127
- )
128
 
129
  log.info("Planner running", task_id=state["task_id"], iteration=state["iteration"])
130
 
@@ -158,7 +117,7 @@ def planner_node(state: WorkflowState) -> WorkflowState:
158
  response = llm.invoke(
159
  [SystemMessage(content=system_content), HumanMessage(content=user_msg)],
160
  tools=[PLANNING_TOOL],
161
- tool_choice="submit_plan",
162
  )
163
 
164
  tool_call = response.tool_calls[0] if response.tool_calls else None
@@ -168,9 +127,10 @@ def planner_node(state: WorkflowState) -> WorkflowState:
168
  plan_data = tool_call["args"]
169
  thinking = plan_data.get("thinking", "")
170
 
171
- # Convert to plan steps
172
  new_plan = []
173
- for step in plan_data.get("plan", []):
 
174
  new_plan.append(make_plan_step(
175
  step_id=step.get("step_id", f"step_{len(new_plan)+1}"),
176
  title=step.get("title", ""),
 
17
  import uuid
18
  from datetime import datetime, timezone
19
 
 
20
  from langchain_core.messages import SystemMessage, HumanMessage
21
 
22
+ from ..core.llm import get_llm
23
  from ..state.graph_state import (
24
  WorkflowState, TaskStatus, AgentRole,
25
  make_plan_step, make_agent_event,
26
  )
 
27
  from ..core.logger import get_logger
28
 
29
  log = get_logger(__name__)
30
 
31
+ PLANNER_SYSTEM = """You are a task planner. Decompose the task into 2-4 steps MAX.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
32
 
33
+ Tools: web_search, fetch_url, calculate, run_python, write_file, read_file, get_datetime, synthesize
34
+ Always end with a synthesize step.
 
 
 
 
35
 
36
+ Return ONLY JSON:
37
+ {"thinking":"one line","plan":[{"step_id":"step_1","title":"short","description":"what to do","tool":"tool_name","depends_on":[]}],"estimated_complexity":"low|medium|high"}
38
 
39
+ Rules: minimal steps, no redundancy, synthesize last, depends_on lists prerequisite step_ids."""
 
 
 
40
 
 
 
41
 
42
+ REPLAN_ADDITION = """
43
+ Replanning β€” fix these issues: {critique}
44
+ Already done: {completed_steps}
45
+ Only plan remaining steps."""
46
 
47
 
48
  PLANNING_TOOL = {
 
81
  LangGraph node β€” runs the Planner agent.
82
  Called at start and whenever Critic requests replanning.
83
  """
84
+ from ..core.config import get_settings
85
  settings = get_settings()
86
+ llm = get_llm("planner", temperature=0.2)
 
 
 
 
87
 
88
  log.info("Planner running", task_id=state["task_id"], iteration=state["iteration"])
89
 
 
117
  response = llm.invoke(
118
  [SystemMessage(content=system_content), HumanMessage(content=user_msg)],
119
  tools=[PLANNING_TOOL],
120
+ tool_choice={"type": "function", "function": {"name": "submit_plan"}},
121
  )
122
 
123
  tool_call = response.tool_calls[0] if response.tool_calls else None
 
127
  plan_data = tool_call["args"]
128
  thinking = plan_data.get("thinking", "")
129
 
130
+ # Convert to plan steps (hard cap)
131
  new_plan = []
132
+ raw_steps = plan_data.get("plan", [])[:settings.max_plan_steps]
133
+ for step in raw_steps:
134
  new_plan.append(make_plan_step(
135
  step_id=step.get("step_id", f"step_{len(new_plan)+1}"),
136
  title=step.get("title", ""),
backend/core/config.py CHANGED
@@ -1,38 +1,56 @@
1
- """backend/core/config.py"""
2
  from functools import lru_cache
3
- from typing import Literal
4
  from pydantic_settings import BaseSettings, SettingsConfigDict
5
 
6
 
7
  class Settings(BaseSettings):
8
  model_config = SettingsConfigDict(env_file=".env", extra="ignore", case_sensitive=False)
9
 
10
- # LLM
11
  google_api_key: str = ""
12
- planner_model: str = "gemini-2.5-flash"
13
- executor_model: str = "gemini-2.5-flash"
14
- critic_model: str = "gemini-2.5-flash"
15
- memory_model: str = "gemini-2.5-flash"
16
 
17
- # Redis
18
- redis_url: str = "redis://localhost:6379"
19
- redis_ttl: int = 86400 # 24h default
20
-
21
- # DB
22
- database_url: str = "sqlite+aiosqlite:///./agent_system.db"
23
-
24
- # App
25
  app_env: str = "development"
26
  app_host: str = "0.0.0.0"
27
  app_port: int = 8000
28
  log_level: str = "INFO"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
29
 
30
- # Agent config
31
- max_iterations: int = 10 # Max planner loops
32
- max_retries: int = 3 # Per-step retry limit
33
- step_timeout: int = 120 # Seconds per step
34
- enable_reflection: bool = True # Critic reflection pass
35
- enable_memory: bool = True # Memory persistence
 
 
 
 
 
 
 
 
36
 
37
  @property
38
  def is_dev(self) -> bool:
 
1
+ """backend/core/config.py β€” All non-secret config lives here. Secrets come from .env."""
2
  from functools import lru_cache
 
3
  from pydantic_settings import BaseSettings, SettingsConfigDict
4
 
5
 
6
  class Settings(BaseSettings):
7
  model_config = SettingsConfigDict(env_file=".env", extra="ignore", case_sensitive=False)
8
 
9
+ # ── Secrets (from .env) ───────────────────────────────────────────────────
10
  google_api_key: str = ""
11
+ groq_api_key: str = ""
12
+ serper_api_key: str = ""
 
 
13
 
14
+ # ── App ───────────────────────────────────────────────────────────────────
 
 
 
 
 
 
 
15
  app_env: str = "development"
16
  app_host: str = "0.0.0.0"
17
  app_port: int = 8000
18
  log_level: str = "INFO"
19
+ frontend_url: str = "http://localhost:5500"
20
+
21
+ # ── Database ──────────────────────────────────────────────────────────────
22
+ database_url: str = "sqlite+aiosqlite:///./agent_system.db"
23
+
24
+ # ── Redis ─────────────────────────────────────────────────────────────────
25
+ redis_host: str = "localhost"
26
+ redis_port: int = 6379
27
+ redis_db: int = 0
28
+ redis_url: str = "redis://localhost:6379"
29
+ redis_ttl: int = 86400
30
+
31
+ # ── LLM provider: "gemini" or "groq" ─────────────────────────────────────
32
+ llm_provider: str = "groq"
33
+
34
+ # Gemini models
35
+ planner_model: str = "gemini-2.5-flash"
36
+ executor_model: str = "gemini-2.5-flash"
37
+ critic_model: str = "gemini-2.5-flash"
38
+ memory_model: str = "gemini-2.5-flash"
39
 
40
+ # Groq models
41
+ groq_planner_model: str = "llama-3.3-70b-versatile"
42
+ groq_executor_model: str = "llama-3.3-70b-versatile"
43
+ groq_critic_model: str = "llama-3.1-8b-instant"
44
+ groq_memory_model: str = "llama-3.1-8b-instant"
45
+
46
+ # ── Agent behaviour ───────────────────────────────────────────────────────
47
+ max_iterations: int = 3
48
+ max_retries: int = 1
49
+ step_timeout: int = 60
50
+ enable_reflection: bool = True
51
+ enable_memory: bool = True
52
+ max_plan_steps: int = 4
53
+ max_output_tokens: int = 512
54
 
55
  @property
56
  def is_dev(self) -> bool:
backend/core/llm.py ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """backend/core/llm.py β€” LLM factory supporting Gemini and Grok."""
2
+ from __future__ import annotations
3
+
4
+
5
+ def get_llm(role: str, temperature: float = 0.1):
6
+ """
7
+ Return the right LLM based on LLM_PROVIDER in .env.
8
+ role: "planner" | "executor" | "critic" | "memory"
9
+ """
10
+ from .config import get_settings
11
+ settings = get_settings()
12
+
13
+ max_tokens = settings.max_output_tokens
14
+
15
+ if settings.llm_provider == "groq":
16
+ from langchain_groq import ChatGroq
17
+
18
+ model_map = {
19
+ "planner": settings.groq_planner_model,
20
+ "executor": settings.groq_executor_model,
21
+ "critic": settings.groq_critic_model,
22
+ "memory": settings.groq_memory_model,
23
+ }
24
+ return ChatGroq(
25
+ model=model_map.get(role, "llama-3.3-70b-versatile"),
26
+ temperature=temperature,
27
+ api_key=settings.groq_api_key,
28
+ max_tokens=max_tokens,
29
+ )
30
+
31
+ else: # gemini (default)
32
+ from langchain_google_genai import ChatGoogleGenerativeAI
33
+
34
+ model_map = {
35
+ "planner": settings.planner_model,
36
+ "executor": settings.executor_model,
37
+ "critic": settings.critic_model,
38
+ "memory": settings.memory_model,
39
+ }
40
+ return ChatGoogleGenerativeAI(
41
+ model=model_map.get(role, settings.planner_model),
42
+ temperature=temperature,
43
+ google_api_key=settings.google_api_key,
44
+ max_output_tokens=max_tokens,
45
+ )
backend/tools/registry.py CHANGED
@@ -33,22 +33,45 @@ def err(message: str, details: str = "") -> dict:
33
  # ── Web search ────────────────────────────────────────────────────────────────
34
 
35
  async def web_search(query: str, max_results: int = 5) -> dict:
36
- """Search the web using DuckDuckGo DDGS (avoids rate-limit issues of DuckDuckGoSearchRun)."""
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
37
  try:
38
  from duckduckgo_search import DDGS
39
  loop = asyncio.get_event_loop()
40
 
41
  def _search():
42
  with DDGS() as ddgs:
43
- results = list(ddgs.text(query, max_results=int(max_results)))
44
- return results
45
 
46
  results = await loop.run_in_executor(None, _search)
47
  formatted = "\n\n".join(
48
  f"**{r.get('title', '')}**\n{r.get('href', '')}\n{r.get('body', '')}"
49
  for r in results
50
  )
51
- return ok({"query": query, "results": formatted[:3000], "count": len(results)})
52
  except Exception as e:
53
  return err(f"Search failed: {e}", str(e))
54
 
@@ -124,28 +147,41 @@ def list_files() -> dict:
124
 
125
  def run_python(code: str) -> dict:
126
  """
127
- Execute Python code in a restricted sandbox.
128
- Returns stdout output (captured) or error.
129
- PRODUCTION NOTE: Use Docker sandbox / e2b in real deployment.
130
  """
131
- import io, sys, contextlib
132
- # Block dangerous imports
133
- blocked = ["os", "sys", "subprocess", "socket", "requests", "open", "__import__"]
134
  for b in blocked:
135
- if f"import {b}" in code or f"({b}" in code:
136
  return err(f"Blocked: '{b}' not allowed in sandbox")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
137
 
138
  stdout_capture = io.StringIO()
139
  try:
140
  with contextlib.redirect_stdout(stdout_capture):
141
- exec(code, {"__builtins__": {"print": print, "range": range,
142
- "len": len, "str": str, "int": int,
143
- "float": float, "list": list, "dict": dict,
144
- "enumerate": enumerate, "zip": zip,
145
- "sorted": sorted, "sum": sum, "min": min,
146
- "max": max, "abs": abs, "round": round}})
147
  output = stdout_capture.getvalue()
148
- return ok({"output": output[:2000], "code": code[:500]})
149
  except Exception as e:
150
  return err(f"Execution error: {type(e).__name__}: {e}", code[:200])
151
 
 
33
  # ── Web search ────────────────────────────────────────────────────────────────
34
 
35
  async def web_search(query: str, max_results: int = 5) -> dict:
36
+ """Search via Serper (Google), falling back to DuckDuckGo if unavailable."""
37
+ from ..core.config import get_settings
38
+ settings = get_settings()
39
+
40
+ # ── Primary: Serper (Google Search API) ──────────────────────────────────
41
+ if settings.serper_api_key:
42
+ try:
43
+ async with httpx.AsyncClient(timeout=10) as c:
44
+ r = await c.post(
45
+ "https://google.serper.dev/search",
46
+ headers={"X-API-KEY": settings.serper_api_key, "Content-Type": "application/json"},
47
+ json={"q": query, "num": int(max_results)},
48
+ )
49
+ if r.status_code == 200:
50
+ data = r.json()
51
+ items = data.get("organic", [])
52
+ formatted = "\n\n".join(
53
+ f"**{item.get('title', '')}**\n{item.get('link', '')}\n{item.get('snippet', '')}"
54
+ for item in items
55
+ )
56
+ return ok({"query": query, "results": formatted[:3000], "count": len(items), "source": "serper"})
57
+ except Exception:
58
+ pass # fall through to DuckDuckGo
59
+
60
+ # ── Fallback: DuckDuckGo ──────────────────────────────────────────────────
61
  try:
62
  from duckduckgo_search import DDGS
63
  loop = asyncio.get_event_loop()
64
 
65
  def _search():
66
  with DDGS() as ddgs:
67
+ return list(ddgs.text(query, max_results=int(max_results)))
 
68
 
69
  results = await loop.run_in_executor(None, _search)
70
  formatted = "\n\n".join(
71
  f"**{r.get('title', '')}**\n{r.get('href', '')}\n{r.get('body', '')}"
72
  for r in results
73
  )
74
+ return ok({"query": query, "results": formatted[:3000], "count": len(results), "source": "duckduckgo"})
75
  except Exception as e:
76
  return err(f"Search failed: {e}", str(e))
77
 
 
147
 
148
  def run_python(code: str) -> dict:
149
  """
150
+ Execute Python code in a sandbox with common data-science libs available.
151
+ Blocks truly dangerous ops (os, subprocess, socket, open).
 
152
  """
153
+ import io, contextlib, builtins
154
+ # Block dangerous modules only
155
+ blocked = ["subprocess", "socket", "requests"]
156
  for b in blocked:
157
+ if f"import {b}" in code:
158
  return err(f"Blocked: '{b}' not allowed in sandbox")
159
+ if "open(" in code:
160
+ return err("Blocked: file 'open()' not allowed in sandbox")
161
+
162
+ # Allow standard builtins + safe data libs
163
+ import math, json
164
+ from datetime import datetime as _dt
165
+ safe_globals = {
166
+ "__builtins__": builtins, # full builtins so import works
167
+ "math": math,
168
+ "json": json,
169
+ "datetime": _dt,
170
+ }
171
+ # Try to inject optional libs if installed
172
+ for lib in ["pandas", "numpy", "matplotlib"]:
173
+ try:
174
+ import importlib
175
+ safe_globals[lib.split(".")[0]] = importlib.import_module(lib)
176
+ except ImportError:
177
+ pass
178
 
179
  stdout_capture = io.StringIO()
180
  try:
181
  with contextlib.redirect_stdout(stdout_capture):
182
+ exec(code, safe_globals)
 
 
 
 
 
183
  output = stdout_capture.getvalue()
184
+ return ok({"output": output[:2000] or "(no output)", "code": code[:500]})
185
  except Exception as e:
186
  return err(f"Execution error: {type(e).__name__}: {e}", code[:200])
187
 
check_api.py ADDED
@@ -0,0 +1,79 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Check Gemini API key, available models, and rate limits."""
2
+ import asyncio, os, sys
3
+ from dotenv import load_dotenv
4
+ import httpx
5
+
6
+ load_dotenv()
7
+ KEY = os.getenv("GOOGLE_API_KEY", "")
8
+
9
+ BASE = "https://generativelanguage.googleapis.com/v1beta"
10
+ MODELS_TO_TEST = [
11
+ "gemini-2.5-flash",
12
+ "gemini-2.5-pro",
13
+ "gemini-2.0-flash",
14
+ "gemini-1.5-flash",
15
+ "gemini-1.5-pro",
16
+ ]
17
+
18
+
19
+ async def main():
20
+ if not KEY:
21
+ print("[X] GOOGLE_API_KEY not set in .env")
22
+ sys.exit(1)
23
+ print(f"Key: ...{KEY[-6:]}\n")
24
+
25
+ async with httpx.AsyncClient(timeout=20) as c:
26
+
27
+ # 1) List all available models
28
+ print("=" * 55)
29
+ print("AVAILABLE MODELS")
30
+ print("=" * 55)
31
+ r = await c.get(f"{BASE}/models?key={KEY}")
32
+ if r.status_code != 200:
33
+ print(f"[X] Could not list models: {r.status_code} {r.text[:200]}")
34
+ else:
35
+ models = r.json().get("models", [])
36
+ gen_models = [m for m in models if "generateContent" in m.get("supportedGenerationMethods", [])]
37
+ for m in gen_models:
38
+ name = m["name"].replace("models/", "")
39
+ limit = m.get("description", "")[:60]
40
+ print(f" {name:<35} rpm={m.get('rpmLimit','?'):>6} tpm={m.get('tpmLimit','?'):>10}")
41
+
42
+ # 2) Test each target model with a tiny call
43
+ print("\n" + "=" * 55)
44
+ print("MODEL PING TEST")
45
+ print("=" * 55)
46
+ body = {"contents": [{"parts": [{"text": "Say OK"}]}],
47
+ "generationConfig": {"maxOutputTokens": 5}}
48
+ for model in MODELS_TO_TEST:
49
+ url = f"{BASE}/models/{model}:generateContent?key={KEY}"
50
+ r = await c.post(url, json=body)
51
+ if r.status_code == 200:
52
+ status = "[OK]"
53
+ elif r.status_code == 429:
54
+ status = "[429 rate limit]"
55
+ elif r.status_code == 404:
56
+ status = "[404 not found]"
57
+ elif r.status_code == 403:
58
+ status = "[403 denied]"
59
+ else:
60
+ status = f"[{r.status_code}]"
61
+ print(f" {model:<30} {status}")
62
+
63
+ # 3) Quota info from a real call on the working model
64
+ print("\n" + "=" * 55)
65
+ print("RATE LIMITS (from API metadata)")
66
+ print("=" * 55)
67
+ r = await c.get(f"{BASE}/models/gemini-2.5-flash?key={KEY}")
68
+ if r.status_code == 200:
69
+ m = r.json()
70
+ print(f" Model : {m.get('displayName','')}")
71
+ print(f" Input limit : {m.get('inputTokenLimit','?'):,} tokens")
72
+ print(f" Output limit : {m.get('outputTokenLimit','?'):,} tokens")
73
+ print(f" RPM limit : {m.get('rpmLimit', 'see https://ai.dev/rate-limit')}")
74
+ print(f" TPM limit : {m.get('tpmLimit', 'see https://ai.dev/rate-limit')}")
75
+ print(f"\n Free tier caps: 15 RPM / 1,000,000 TPM / 1,500 RPD")
76
+ print(f" Full limits : https://ai.google.dev/gemini-api/docs/rate-limits")
77
+
78
+
79
+ asyncio.run(main())
frontend/app.js CHANGED
@@ -1,5 +1,7 @@
1
  /* ─── Config ──────────────────────────────────────────────────────────────── */
2
- const API_BASE = 'http://localhost:8000';
 
 
3
 
4
  /* ─── State ───────────────────────────────────────────────────────────────── */
5
  const state = {
@@ -397,9 +399,14 @@ async function runBatch(task) {
397
  }
398
 
399
  /* ─── Show output ────────────────────────────────────────────────────────── */
 
 
 
 
 
400
  function showOutput(text, score) {
401
  outputCard.style.display = '';
402
- outputContent.textContent = text;
403
  if (score != null) {
404
  const s = Math.round(score);
405
  qualityBadge.textContent = `${s}/100`;
 
1
  /* ─── Config ──────────────────────────────────────────────────────────────── */
2
+ // Empty string = same origin (works on HF Spaces and any deployment).
3
+ // Falls back to localhost only when running locally on a different port.
4
+ const API_BASE = window.location.hostname === 'localhost' ? 'http://localhost:8000' : '';
5
 
6
  /* ─── State ───────────────────────────────────────────────────────────────── */
7
  const state = {
 
399
  }
400
 
401
  /* ─── Show output ────────────────────────────────────────────────────────── */
402
+ function cleanOutput(text) {
403
+ // Strip LaTeX boxing that Llama sometimes outputs
404
+ return (text || '').replace(/\$\\boxed\{([^}]+)\}\$/g, '$1').replace(/\\boxed\{([^}]+)\}/g, '$1');
405
+ }
406
+
407
  function showOutput(text, score) {
408
  outputCard.style.display = '';
409
+ outputContent.textContent = cleanOutput(text);
410
  if (score != null) {
411
  const s = Math.round(score);
412
  qualityBadge.textContent = `${s}/100`;
frontend/index.html CHANGED
@@ -22,7 +22,7 @@
22
  </svg>
23
  <span class="logo-text">Multi-Agent System</span>
24
  </div>
25
- <span class="powered-by">Powered by <span class="gemini-badge">Gemini 2.5 Flash Lite</span></span>
26
  </div>
27
  <div class="header-right">
28
  <div class="health-indicator" id="healthDot" title="Backend status"></div>
 
22
  </svg>
23
  <span class="logo-text">Multi-Agent System</span>
24
  </div>
25
+ <span class="powered-by">Powered by <span class="gemini-badge">Groq Β· Llama 3.3 70B</span></span>
26
  </div>
27
  <div class="header-right">
28
  <div class="health-indicator" id="healthDot" title="Backend status"></div>
frontend/style.css CHANGED
@@ -139,15 +139,20 @@ body {
139
  /* ─── Main Layout ─────────────────────────────────────────────────────────── */
140
  .layout {
141
  display: grid;
142
- grid-template-columns: 300px 1fr 270px;
 
143
  gap: 0;
144
  flex: 1;
 
 
145
  overflow: hidden;
146
  }
147
 
148
  .panel {
149
  display: flex;
150
  flex-direction: column;
 
 
151
  overflow: hidden;
152
  background: var(--bg-panel);
153
  }
@@ -168,6 +173,9 @@ body {
168
 
169
  .panel-right {
170
  border-left: 1px solid var(--border);
 
 
 
171
  overflow-y: auto;
172
  }
173
 
@@ -327,7 +335,7 @@ textarea::placeholder { color: var(--text-dim); }
327
  .btn-icon-arrow { flex-shrink: 0; }
328
 
329
  /* ─── Left Panel: History ─────────────────────────────────────────────────── */
330
- .history-section { padding: 16px; flex: 1; }
331
 
332
  .task-history { display: flex; flex-direction: column; gap: 6px; }
333
 
@@ -415,13 +423,13 @@ textarea::placeholder { color: var(--text-dim); }
415
  .status-badge.failed { background: rgba(239,68,68,0.15); color: var(--red); border-color: rgba(239,68,68,0.3); }
416
 
417
  /* ─── Agent Graph ─────────────────────────────────────────────────────────── */
418
- .agent-graph { padding: 20px 16px; overflow-x: auto; }
419
 
420
  .graph-track {
421
  display: flex;
422
  align-items: center;
423
  gap: 0;
424
- min-width: fit-content;
425
  margin: 0 auto;
426
  }
427
 
@@ -430,11 +438,12 @@ textarea::placeholder { color: var(--text-dim); }
430
  flex-direction: column;
431
  align-items: center;
432
  gap: 4px;
433
- padding: 12px 14px;
434
  background: var(--bg-input);
435
  border: 1.5px solid var(--border);
436
  border-radius: var(--radius);
437
- min-width: 80px;
 
438
  transition: all var(--transition);
439
  cursor: default;
440
  }
@@ -464,9 +473,13 @@ textarea::placeholder { color: var(--text-dim); }
464
  font-size: 9px;
465
  font-weight: 600;
466
  text-transform: uppercase;
467
- letter-spacing: 0.05em;
468
- padding: 1px 6px;
469
  border-radius: 99px;
 
 
 
 
470
  }
471
  .node-state.idle { background: var(--bg-card); color: var(--text-dim); }
472
  .node-state.running { background: rgba(124,58,237,0.2); color: var(--purple-light); animation: blink 1s step-start infinite; }
@@ -571,7 +584,8 @@ textarea::placeholder { color: var(--text-dim); }
571
  color: var(--text);
572
  white-space: pre-wrap;
573
  word-break: break-word;
574
- max-height: 400px;
 
575
  overflow-y: auto;
576
  }
577
 
@@ -579,10 +593,10 @@ textarea::placeholder { color: var(--text-dim); }
579
  .events-section {
580
  padding: 14px;
581
  border-bottom: 1px solid var(--border);
582
- flex: 1;
583
  display: flex;
584
  flex-direction: column;
585
- min-height: 0;
 
586
  }
587
 
588
  .btn-ghost-sm {
@@ -605,7 +619,6 @@ textarea::placeholder { color: var(--text-dim); }
605
  flex-direction: column;
606
  gap: 4px;
607
  min-height: 0;
608
- max-height: 300px;
609
  }
610
 
611
  .event-item {
@@ -638,7 +651,7 @@ textarea::placeholder { color: var(--text-dim); }
638
  .event-msg { font-size: 11px; color: var(--text-muted); line-height: 1.4; }
639
 
640
  /* ─── Right Panel: Metrics ────────────────────────────────────────────────── */
641
- .metrics-section { padding: 14px; border-bottom: 1px solid var(--border); }
642
 
643
  .metrics-grid {
644
  display: grid;
@@ -668,7 +681,7 @@ textarea::placeholder { color: var(--text-dim); }
668
  .metric-label { font-size: 9px; color: var(--text-dim); text-transform: uppercase; letter-spacing: 0.06em; }
669
 
670
  /* ─── Right Panel: Tools ──────────────────────────────────────────────────── */
671
- .tools-section { padding: 14px; }
672
 
673
  .tool-list { display: flex; flex-direction: column; gap: 3px; }
674
  .tool-item {
@@ -718,12 +731,17 @@ textarea::placeholder { color: var(--text-dim); }
718
  }
719
 
720
  /* ─── Responsive ──────────────────────────────────────────────────────────── */
721
- @media (max-width: 1100px) {
722
- .layout { grid-template-columns: 260px 1fr 240px; }
 
 
 
 
 
723
  }
724
- @media (max-width: 860px) {
725
  .layout { grid-template-columns: 1fr; }
726
- .panel-left, .panel-right { display: none; }
727
  body { overflow: auto; }
728
  .panel-center { min-height: 100vh; }
729
  }
 
139
  /* ─── Main Layout ─────────────────────────────────────────────────────────── */
140
  .layout {
141
  display: grid;
142
+ grid-template-columns: 280px 1fr 260px;
143
+ grid-template-rows: 1fr;
144
  gap: 0;
145
  flex: 1;
146
+ min-height: 0;
147
+ height: 0; /* forces grid rows to respect flex parent height */
148
  overflow: hidden;
149
  }
150
 
151
  .panel {
152
  display: flex;
153
  flex-direction: column;
154
+ min-height: 0;
155
+ height: 100%;
156
  overflow: hidden;
157
  background: var(--bg-panel);
158
  }
 
173
 
174
  .panel-right {
175
  border-left: 1px solid var(--border);
176
+ display: flex;
177
+ flex-direction: column;
178
+ min-height: 0;
179
  overflow-y: auto;
180
  }
181
 
 
335
  .btn-icon-arrow { flex-shrink: 0; }
336
 
337
  /* ─── Left Panel: History ─────────────────────────────────────────────────── */
338
+ .history-section { padding: 16px; flex: 1; overflow-y: auto; min-height: 0; }
339
 
340
  .task-history { display: flex; flex-direction: column; gap: 6px; }
341
 
 
423
  .status-badge.failed { background: rgba(239,68,68,0.15); color: var(--red); border-color: rgba(239,68,68,0.3); }
424
 
425
  /* ─── Agent Graph ─────────────────────────────────────────────────────────── */
426
+ .agent-graph { padding: 16px; overflow-x: auto; }
427
 
428
  .graph-track {
429
  display: flex;
430
  align-items: center;
431
  gap: 0;
432
+ min-width: max-content;
433
  margin: 0 auto;
434
  }
435
 
 
438
  flex-direction: column;
439
  align-items: center;
440
  gap: 4px;
441
+ padding: 10px 12px;
442
  background: var(--bg-input);
443
  border: 1.5px solid var(--border);
444
  border-radius: var(--radius);
445
+ min-width: 88px;
446
+ width: 88px;
447
  transition: all var(--transition);
448
  cursor: default;
449
  }
 
473
  font-size: 9px;
474
  font-weight: 600;
475
  text-transform: uppercase;
476
+ letter-spacing: 0.04em;
477
+ padding: 2px 6px;
478
  border-radius: 99px;
479
+ white-space: nowrap;
480
+ max-width: 100%;
481
+ overflow: hidden;
482
+ text-overflow: ellipsis;
483
  }
484
  .node-state.idle { background: var(--bg-card); color: var(--text-dim); }
485
  .node-state.running { background: rgba(124,58,237,0.2); color: var(--purple-light); animation: blink 1s step-start infinite; }
 
584
  color: var(--text);
585
  white-space: pre-wrap;
586
  word-break: break-word;
587
+ min-height: 80px;
588
+ max-height: 600px;
589
  overflow-y: auto;
590
  }
591
 
 
593
  .events-section {
594
  padding: 14px;
595
  border-bottom: 1px solid var(--border);
 
596
  display: flex;
597
  flex-direction: column;
598
+ min-height: 200px;
599
+ max-height: 45vh;
600
  }
601
 
602
  .btn-ghost-sm {
 
619
  flex-direction: column;
620
  gap: 4px;
621
  min-height: 0;
 
622
  }
623
 
624
  .event-item {
 
651
  .event-msg { font-size: 11px; color: var(--text-muted); line-height: 1.4; }
652
 
653
  /* ─── Right Panel: Metrics ────────────────────────────────────────────────── */
654
+ .metrics-section { padding: 14px; border-bottom: 1px solid var(--border); flex-shrink: 0; }
655
 
656
  .metrics-grid {
657
  display: grid;
 
681
  .metric-label { font-size: 9px; color: var(--text-dim); text-transform: uppercase; letter-spacing: 0.06em; }
682
 
683
  /* ─── Right Panel: Tools ──────────────────────────────────────────────────── */
684
+ .tools-section { padding: 14px; flex-shrink: 0; }
685
 
686
  .tool-list { display: flex; flex-direction: column; gap: 3px; }
687
  .tool-item {
 
731
  }
732
 
733
  /* ─── Responsive ──────────────────────────────────────────────────────────── */
734
+ @media (max-width: 1200px) {
735
+ .layout { grid-template-columns: 240px 1fr 220px; }
736
+ .agent-node { min-width: 78px; width: 78px; }
737
+ }
738
+ @media (max-width: 960px) {
739
+ .layout { grid-template-columns: 220px 1fr; }
740
+ .panel-right { display: none; }
741
  }
742
+ @media (max-width: 680px) {
743
  .layout { grid-template-columns: 1fr; }
744
+ .panel-left { display: none; }
745
  body { overflow: auto; }
746
  .panel-center { min-height: 100vh; }
747
  }
test.py ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Quick smoke test for the multi-agent system.
2
+
3
+ Checks:
4
+ 1. GOOGLE_API_KEY is set and reachable
5
+ 2. Core imports work (state, tools, orchestrator)
6
+ 3. WorkflowState initialisation is correct
7
+ 4. Routing logic behaves as expected
8
+
9
+ Run:
10
+ .venv/Scripts/python test.py
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import asyncio
15
+ import os
16
+ import sys
17
+
18
+ try:
19
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace")
20
+ except Exception:
21
+ pass
22
+
23
+ from dotenv import load_dotenv
24
+
25
+ load_dotenv()
26
+
27
+ SEP = "=" * 60
28
+
29
+
30
+ def check_env() -> bool:
31
+ key = os.getenv("GOOGLE_API_KEY", "")
32
+ print(SEP)
33
+ print("GOOGLE_API_KEY set :", bool(key), f"(...{key[-4:]})" if key else "")
34
+ print("PLANNER_MODEL :", os.getenv("PLANNER_MODEL", "gemini-2.5-flash"))
35
+ print("DATABASE_URL :", os.getenv("DATABASE_URL", "(not set)"))
36
+ print("REDIS_URL :", os.getenv("REDIS_URL", "(not set)"))
37
+ print(SEP)
38
+ if not key:
39
+ print("[X] No GOOGLE_API_KEY found β€” set it in .env")
40
+ return False
41
+ return True
42
+
43
+
44
+ async def check_gemini(key: str) -> bool:
45
+ import httpx
46
+
47
+ model = os.getenv("PLANNER_MODEL", "gemini-2.5-flash")
48
+ url = (
49
+ f"https://generativelanguage.googleapis.com/v1beta/models/"
50
+ f"{model}:generateContent?key={key}"
51
+ )
52
+ body = {"contents": [{"parts": [{"text": "Reply with exactly: API key works"}]}]}
53
+ async with httpx.AsyncClient(timeout=15) as c:
54
+ r = await c.post(url, json=body)
55
+
56
+ print(f"\n[1] Gemini live call -> HTTP {r.status_code}")
57
+ if r.status_code >= 300:
58
+ print(" error:", r.text[:400])
59
+ print("\n[X] Common causes:")
60
+ print(" 400 API_KEY_INVALID -> wrong/expired key")
61
+ print(" 403 PERMISSION_DENIED -> 'Generative Language API' not enabled,")
62
+ print(" OR Google project access denied β€”")
63
+ print(" create a new key at https://aistudio.google.com/apikey")
64
+ print(" 404 model not found -> check PLANNER_MODEL in .env")
65
+ print(" 429 -> rate/quota limit, try again shortly")
66
+ return False
67
+
68
+ try:
69
+ reply = r.json()["candidates"][0]["content"]["parts"][0]["text"].strip()
70
+ except Exception:
71
+ reply = r.text[:200]
72
+ print(" model reply:", reply)
73
+ print("[OK] Gemini API key is working.\n")
74
+ return True
75
+
76
+
77
+ def check_imports() -> bool:
78
+ print("[2] Checking imports ...")
79
+ try:
80
+ from backend.state.graph_state import ( # noqa: F401
81
+ WorkflowState, TaskStatus, StepStatus, AgentRole,
82
+ create_initial_state, make_plan_step, make_agent_event,
83
+ )
84
+ from backend.agents.orchestrator import ( # noqa: F401
85
+ build_workflow, route_after_executor,
86
+ route_after_critic, route_after_planner,
87
+ )
88
+ print(" backend.state OK")
89
+ print(" backend.orchestrator OK")
90
+ except ImportError as e:
91
+ print(f" [X] Import failed: {e}")
92
+ return False
93
+
94
+ try:
95
+ from backend.tools.registry import execute_tool, calculate # noqa: F401
96
+ print(" backend.tools OK")
97
+ except ImportError as e:
98
+ print(f" [X] tools import failed: {e}")
99
+ return False
100
+
101
+ print("[OK] All imports succeeded.\n")
102
+ return True
103
+
104
+
105
+ def check_state() -> bool:
106
+ print("[3] WorkflowState smoke test ...")
107
+ from backend.state.graph_state import (
108
+ create_initial_state, TaskStatus, make_plan_step, StepStatus,
109
+ )
110
+
111
+ state = create_initial_state("Write a hello-world script")
112
+ assert state["task"] == "Write a hello-world script"
113
+ assert state["status"] == TaskStatus.PENDING
114
+ assert state["plan"] == []
115
+ assert state["iteration"] == 0
116
+ assert state["total_tokens"] == 0
117
+ assert len(state["events"]) == 1
118
+
119
+ step = make_plan_step("s1", "Write code", "Create hello.py", tool="run_python")
120
+ assert step["status"] == StepStatus.PENDING
121
+ assert step["attempts"] == 0
122
+
123
+ print(" initial state OK")
124
+ print(" make_plan_step OK")
125
+ print("[OK] State checks passed.\n")
126
+ return True
127
+
128
+
129
+ def check_routing() -> bool:
130
+ print("[4] Routing logic smoke test ...")
131
+ from backend.state.graph_state import create_initial_state, TaskStatus, StepStatus
132
+ from backend.agents.orchestrator import (
133
+ route_after_executor, route_after_critic, route_after_planner,
134
+ )
135
+
136
+ def _state(status, plan=None, needs_replanning=False):
137
+ s = create_initial_state("test")
138
+ s["status"] = status
139
+ s["plan"] = plan or []
140
+ s["needs_replanning"] = needs_replanning
141
+ return s
142
+
143
+ # executor β†’ critic when REFLECTING
144
+ assert route_after_executor(_state(TaskStatus.REFLECTING)) == "critic"
145
+ # executor β†’ self when steps pending
146
+ assert route_after_executor(
147
+ _state(TaskStatus.EXECUTING, [{"step_id": "s1", "status": StepStatus.PENDING}])
148
+ ) == "executor"
149
+ # executor β†’ end on FAILED
150
+ assert route_after_executor(_state(TaskStatus.FAILED)) == "end"
151
+ # critic β†’ planner on needs_replanning
152
+ assert route_after_critic(_state(TaskStatus.PLANNING, needs_replanning=True)) == "planner"
153
+ # critic β†’ memory_store when approved
154
+ assert route_after_critic(_state(TaskStatus.COMPLETED)) == "memory_store"
155
+ # planner β†’ end with empty plan
156
+ assert route_after_planner(_state(TaskStatus.EXECUTING, plan=[])) == "end"
157
+ # planner β†’ executor with valid plan
158
+ assert route_after_planner(
159
+ _state(TaskStatus.EXECUTING, [{"step_id": "s1", "status": StepStatus.PENDING}])
160
+ ) == "executor"
161
+
162
+ print(" route_after_executor OK")
163
+ print(" route_after_critic OK")
164
+ print(" route_after_planner OK")
165
+ print("[OK] Routing checks passed.\n")
166
+ return True
167
+
168
+
169
+ async def main() -> None:
170
+ ok = check_env()
171
+ if not ok:
172
+ sys.exit(1)
173
+
174
+ key = os.getenv("GOOGLE_API_KEY", "")
175
+ gemini_ok = await check_gemini(key)
176
+
177
+ imports_ok = check_imports()
178
+ state_ok = check_state() if imports_ok else False
179
+ routing_ok = check_routing() if imports_ok else False
180
+
181
+ print(SEP)
182
+ print("Summary")
183
+ print(SEP)
184
+ print(" Gemini API :", "[OK]" if gemini_ok else "[FAIL]")
185
+ print(" Imports :", "[OK]" if imports_ok else "[FAIL]")
186
+ print(" State :", "[OK]" if state_ok else "[FAIL]")
187
+ print(" Routing :", "[OK]" if routing_ok else "[FAIL]")
188
+ print(SEP)
189
+
190
+ system_ok = imports_ok and state_ok and routing_ok
191
+ if system_ok and gemini_ok:
192
+ print("\nAll checks passed. The system is ready.")
193
+ elif system_ok:
194
+ print("\nSystem checks passed. Fix the Gemini API key to enable LLM calls.")
195
+ else:
196
+ print("\nSystem checks failed. See above for details.")
197
+ sys.exit(1)
198
+
199
+
200
+ if __name__ == "__main__":
201
+ asyncio.run(main())