Spaces:
Sleeping
Sleeping
jatin gyass commited on
Commit Β·
4db2d34
1
Parent(s): 2eef9ea
update for the new web data source
Browse files- backend/agents/critic.py +33 -94
- backend/agents/executor.py +6 -34
- backend/agents/memory_agent.py +3 -66
- backend/agents/planner.py +17 -57
- backend/core/config.py +39 -21
- backend/core/llm.py +45 -0
- backend/tools/registry.py +54 -18
- check_api.py +79 -0
- frontend/app.js +9 -2
- frontend/index.html +1 -1
- frontend/style.css +36 -18
- test.py +201 -0
backend/agents/critic.py
CHANGED
|
@@ -16,11 +16,12 @@ errors compared to self-evaluation by the same model.
|
|
| 16 |
"""
|
| 17 |
from __future__ import annotations
|
| 18 |
import json
|
|
|
|
| 19 |
from datetime import datetime, timezone
|
| 20 |
|
| 21 |
-
from langchain_google_genai import ChatGoogleGenerativeAI
|
| 22 |
from langchain_core.messages import SystemMessage, HumanMessage
|
| 23 |
|
|
|
|
| 24 |
from ..state.graph_state import (
|
| 25 |
WorkflowState, TaskStatus, AgentRole, make_agent_event
|
| 26 |
)
|
|
@@ -29,71 +30,16 @@ from ..core.logger import get_logger
|
|
| 29 |
|
| 30 |
log = get_logger(__name__)
|
| 31 |
|
| 32 |
-
CRITIC_SYSTEM = """
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
2. **Accuracy** (0β25): Are the facts/calculations correct? Are there hallucinations?
|
| 37 |
-
3. **Usefulness** (0β25): Is the output actually helpful for the user's goal?
|
| 38 |
-
4. **Quality** (0β25): Is the output well-structured, clear, and actionable?
|
| 39 |
-
|
| 40 |
-
## Decision thresholds
|
| 41 |
-
- Score β₯ 80: APPROVE β task well completed
|
| 42 |
-
- Score 60β79: APPROVE with suggestions
|
| 43 |
-
- Score < 60: REQUEST_REPLAN β major issues need fixing
|
| 44 |
-
|
| 45 |
-
## Output format (JSON only)
|
| 46 |
-
{
|
| 47 |
-
"thinking": "Your detailed evaluation reasoning",
|
| 48 |
-
"score": 0-100,
|
| 49 |
-
"decision": "approve" | "request_replan",
|
| 50 |
-
"critique": "Specific issues found (if any)",
|
| 51 |
-
"suggestions": ["suggestion 1", "suggestion 2"],
|
| 52 |
-
"completeness_score": 0-25,
|
| 53 |
-
"accuracy_score": 0-25,
|
| 54 |
-
"usefulness_score": 0-25,
|
| 55 |
-
"quality_score": 0-25
|
| 56 |
-
}
|
| 57 |
-
|
| 58 |
-
## Critical rules
|
| 59 |
-
- Be HONEST and SPECIFIC β vague critiques don't help
|
| 60 |
-
- If output is good, approve it β don't invent problems
|
| 61 |
-
- Only request replan for REAL deficiencies
|
| 62 |
-
- Point to SPECIFIC missing or incorrect elements"""
|
| 63 |
-
|
| 64 |
-
|
| 65 |
-
CRITIC_TOOL = {
|
| 66 |
-
"type": "function",
|
| 67 |
-
"function": {
|
| 68 |
-
"name": "submit_critique",
|
| 69 |
-
"description": "Submit quality evaluation and decision",
|
| 70 |
-
"parameters": {
|
| 71 |
-
"type": "object",
|
| 72 |
-
"required": ["thinking", "score", "decision", "critique"],
|
| 73 |
-
"properties": {
|
| 74 |
-
"thinking": {"type": "string"},
|
| 75 |
-
"score": {"type": "integer", "minimum": 0, "maximum": 100},
|
| 76 |
-
"decision": {"type": "string", "enum": ["approve", "request_replan"]},
|
| 77 |
-
"critique": {"type": "string"},
|
| 78 |
-
"suggestions": {"type": "array", "items": {"type": "string"}},
|
| 79 |
-
"completeness_score": {"type": "integer", "minimum": 0, "maximum": 25},
|
| 80 |
-
"accuracy_score": {"type": "integer", "minimum": 0, "maximum": 25},
|
| 81 |
-
"usefulness_score": {"type": "integer", "minimum": 0, "maximum": 25},
|
| 82 |
-
"quality_score": {"type": "integer", "minimum": 0, "maximum": 25},
|
| 83 |
-
},
|
| 84 |
-
},
|
| 85 |
-
},
|
| 86 |
-
}
|
| 87 |
|
| 88 |
|
| 89 |
def critic_node(state: WorkflowState) -> WorkflowState:
|
| 90 |
-
"""LangGraph node
|
| 91 |
settings = get_settings()
|
| 92 |
-
llm =
|
| 93 |
-
model=settings.critic_model,
|
| 94 |
-
temperature=0.1,
|
| 95 |
-
google_api_key=settings.google_api_key,
|
| 96 |
-
)
|
| 97 |
|
| 98 |
log.info("Critic running", task_id=state["task_id"])
|
| 99 |
|
|
@@ -121,45 +67,38 @@ def critic_node(state: WorkflowState) -> WorkflowState:
|
|
| 121 |
result_str = json.dumps(result, default=str)
|
| 122 |
results_summary[step_id] = result_str[:400]
|
| 123 |
|
| 124 |
-
|
| 125 |
-
|
| 126 |
-
{state[
|
| 127 |
-
|
| 128 |
-
|
| 129 |
-
{
|
| 130 |
-
|
| 131 |
-
FAILED STEPS: {len(failed_steps)}
|
| 132 |
-
{chr(10).join(f" - {s['title']}: {s.get('error', '')}" for s in failed_steps) if failed_steps else " None"}
|
| 133 |
-
|
| 134 |
-
STEP RESULTS:
|
| 135 |
-
{json.dumps(results_summary, indent=2)[:2000]}
|
| 136 |
-
|
| 137 |
-
FINAL OUTPUT:
|
| 138 |
-
{state.get("final_output", "NOT YET GENERATED")[:1500]}
|
| 139 |
-
"""
|
| 140 |
|
| 141 |
try:
|
|
|
|
| 142 |
response = llm.invoke(
|
| 143 |
[SystemMessage(content=CRITIC_SYSTEM), HumanMessage(content=eval_context)],
|
| 144 |
-
tools=[CRITIC_TOOL],
|
| 145 |
)
|
| 146 |
|
| 147 |
tokens = response.usage_metadata.get("total_tokens", 0) if response.usage_metadata else 0
|
| 148 |
-
|
| 149 |
-
|
| 150 |
-
#
|
| 151 |
-
|
| 152 |
-
|
| 153 |
-
|
| 154 |
-
|
| 155 |
-
|
| 156 |
-
|
| 157 |
-
|
| 158 |
-
|
| 159 |
-
|
| 160 |
-
|
| 161 |
-
|
| 162 |
-
|
|
|
|
|
|
|
| 163 |
score = int(c.get("score", 70))
|
| 164 |
decision = c.get("decision", "approve")
|
| 165 |
critique = c.get("critique", "")
|
|
|
|
| 16 |
"""
|
| 17 |
from __future__ import annotations
|
| 18 |
import json
|
| 19 |
+
import re
|
| 20 |
from datetime import datetime, timezone
|
| 21 |
|
|
|
|
| 22 |
from langchain_core.messages import SystemMessage, HumanMessage
|
| 23 |
|
| 24 |
+
from ..core.llm import get_llm
|
| 25 |
from ..state.graph_state import (
|
| 26 |
WorkflowState, TaskStatus, AgentRole, make_agent_event
|
| 27 |
)
|
|
|
|
| 30 |
|
| 31 |
log = get_logger(__name__)
|
| 32 |
|
| 33 |
+
CRITIC_SYSTEM = """Evaluate task completion. Score 0-100. Approve if score>=60, else request_replan.
|
| 34 |
+
|
| 35 |
+
You MUST respond with valid JSON only - no extra text, no markdown fences:
|
| 36 |
+
{"thinking":"one line","score":75,"decision":"approve","critique":"none","suggestions":[]}"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
|
| 38 |
|
| 39 |
def critic_node(state: WorkflowState) -> WorkflowState:
|
| 40 |
+
"""LangGraph node - runs the Critic agent for reflection."""
|
| 41 |
settings = get_settings()
|
| 42 |
+
llm = get_llm("critic", temperature=0.1)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 43 |
|
| 44 |
log.info("Critic running", task_id=state["task_id"])
|
| 45 |
|
|
|
|
| 67 |
result_str = json.dumps(result, default=str)
|
| 68 |
results_summary[step_id] = result_str[:400]
|
| 69 |
|
| 70 |
+
steps_summary = ", ".join(f"[{s['status']}]{s['title']}" for s in state["plan"])
|
| 71 |
+
eval_context = (
|
| 72 |
+
f"TASK: {state['task']}\n"
|
| 73 |
+
f"STEPS: {steps_summary}\n"
|
| 74 |
+
f"FAILED: {len(failed_steps)}\n"
|
| 75 |
+
f"OUTPUT: {str(state.get('final_output', 'none'))[:600]}"
|
| 76 |
+
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 77 |
|
| 78 |
try:
|
| 79 |
+
# No tool calling β plain JSON text is more reliable across providers
|
| 80 |
response = llm.invoke(
|
| 81 |
[SystemMessage(content=CRITIC_SYSTEM), HumanMessage(content=eval_context)],
|
|
|
|
| 82 |
)
|
| 83 |
|
| 84 |
tokens = response.usage_metadata.get("total_tokens", 0) if response.usage_metadata else 0
|
| 85 |
+
text = response.content or ""
|
| 86 |
+
|
| 87 |
+
# Strip markdown fences and extract JSON
|
| 88 |
+
stripped = text.strip()
|
| 89 |
+
for fence in ("```json", "```"):
|
| 90 |
+
stripped = stripped.removeprefix(fence)
|
| 91 |
+
stripped = stripped.removesuffix("```").strip()
|
| 92 |
+
|
| 93 |
+
# Find first {...} block in case model adds prose
|
| 94 |
+
match = re.search(r'\{.*\}', stripped, re.DOTALL)
|
| 95 |
+
if match:
|
| 96 |
+
stripped = match.group(0)
|
| 97 |
+
|
| 98 |
+
try:
|
| 99 |
+
c = json.loads(stripped)
|
| 100 |
+
except Exception:
|
| 101 |
+
raise ValueError(f"Critic JSON parse failed. Response: {text[:200]}")
|
| 102 |
score = int(c.get("score", 70))
|
| 103 |
decision = c.get("decision", "approve")
|
| 104 |
critique = c.get("critique", "")
|
backend/agents/executor.py
CHANGED
|
@@ -17,9 +17,9 @@ import json
|
|
| 17 |
import time
|
| 18 |
from datetime import datetime, timezone
|
| 19 |
|
| 20 |
-
from langchain_google_genai import ChatGoogleGenerativeAI
|
| 21 |
from langchain_core.messages import SystemMessage, HumanMessage
|
| 22 |
|
|
|
|
| 23 |
from ..state.graph_state import (
|
| 24 |
WorkflowState, TaskStatus, StepStatus, AgentRole, make_agent_event
|
| 25 |
)
|
|
@@ -29,30 +29,10 @@ from ..core.logger import get_logger
|
|
| 29 |
|
| 30 |
log = get_logger(__name__)
|
| 31 |
|
| 32 |
-
EXECUTOR_SYSTEM = """
|
| 33 |
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
2. Choose the RIGHT tool and RIGHT arguments
|
| 37 |
-
3. If tool results are insufficient, try a different approach
|
| 38 |
-
4. Be specific β vague tool inputs produce vague results
|
| 39 |
-
|
| 40 |
-
## Key rules
|
| 41 |
-
- Read the step description CAREFULLY
|
| 42 |
-
- Use context from previous step results when relevant
|
| 43 |
-
- For "synthesize" steps: write_file first, then provide the complete answer
|
| 44 |
-
- If a tool returns an error, try a different approach
|
| 45 |
-
- Output only what is needed β do not hallucinate results
|
| 46 |
-
|
| 47 |
-
## Response format
|
| 48 |
-
Always call a tool. Do not respond without calling a tool."""
|
| 49 |
-
|
| 50 |
-
SYNTHESIZE_SYSTEM = """You are a synthesis agent. Your job is to produce the FINAL comprehensive answer.
|
| 51 |
-
|
| 52 |
-
Using all the results collected during task execution, write a complete, well-structured response to the original task.
|
| 53 |
-
|
| 54 |
-
Format your response clearly. Use headers, bullet points, or numbered lists where appropriate.
|
| 55 |
-
Be thorough but concise. Cite specific findings where relevant."""
|
| 56 |
|
| 57 |
|
| 58 |
def _build_context(state: WorkflowState) -> str:
|
|
@@ -105,11 +85,7 @@ def executor_node(state: WorkflowState) -> WorkflowState:
|
|
| 105 |
else:
|
| 106 |
updated_plan.append(s)
|
| 107 |
|
| 108 |
-
llm =
|
| 109 |
-
model=settings.executor_model,
|
| 110 |
-
temperature=0.1,
|
| 111 |
-
google_api_key=settings.google_api_key,
|
| 112 |
-
)
|
| 113 |
|
| 114 |
context = _build_context(state)
|
| 115 |
user_msg = (
|
|
@@ -187,11 +163,7 @@ async def _safe_tool_call(name: str, args: dict) -> dict:
|
|
| 187 |
|
| 188 |
def _handle_synthesize(state: WorkflowState, step: dict, settings) -> WorkflowState:
|
| 189 |
"""Handle the special synthesize step β produces final answer."""
|
| 190 |
-
llm =
|
| 191 |
-
model=settings.executor_model,
|
| 192 |
-
temperature=0.3,
|
| 193 |
-
google_api_key=settings.google_api_key,
|
| 194 |
-
)
|
| 195 |
|
| 196 |
context = _build_context(state)
|
| 197 |
response = llm.invoke([
|
|
|
|
| 17 |
import time
|
| 18 |
from datetime import datetime, timezone
|
| 19 |
|
|
|
|
| 20 |
from langchain_core.messages import SystemMessage, HumanMessage
|
| 21 |
|
| 22 |
+
from ..core.llm import get_llm
|
| 23 |
from ..state.graph_state import (
|
| 24 |
WorkflowState, TaskStatus, StepStatus, AgentRole, make_agent_event
|
| 25 |
)
|
|
|
|
| 29 |
|
| 30 |
log = get_logger(__name__)
|
| 31 |
|
| 32 |
+
EXECUTOR_SYSTEM = """Execute ONE step using the available tools. Choose the right tool and args. Always call a tool."""
|
| 33 |
|
| 34 |
+
SYNTHESIZE_SYSTEM = """Write the final answer to the task using the collected results. Be concise and clear.
|
| 35 |
+
Use plain text only - no LaTeX, no $\\boxed{}, no math notation. Write numbers as regular text (e.g. "1040" not "$\\boxed{1040}$")."""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 36 |
|
| 37 |
|
| 38 |
def _build_context(state: WorkflowState) -> str:
|
|
|
|
| 85 |
else:
|
| 86 |
updated_plan.append(s)
|
| 87 |
|
| 88 |
+
llm = get_llm("executor", temperature=0.1)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 89 |
|
| 90 |
context = _build_context(state)
|
| 91 |
user_msg = (
|
|
|
|
| 163 |
|
| 164 |
def _handle_synthesize(state: WorkflowState, step: dict, settings) -> WorkflowState:
|
| 165 |
"""Handle the special synthesize step β produces final answer."""
|
| 166 |
+
llm = get_llm("executor", temperature=0.3)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 167 |
|
| 168 |
context = _build_context(state)
|
| 169 |
response = llm.invoke([
|
backend/agents/memory_agent.py
CHANGED
|
@@ -14,12 +14,8 @@ Memory types:
|
|
| 14 |
"""
|
| 15 |
from __future__ import annotations
|
| 16 |
import asyncio
|
| 17 |
-
import json
|
| 18 |
from datetime import datetime, timezone
|
| 19 |
|
| 20 |
-
from langchain_google_genai import ChatGoogleGenerativeAI
|
| 21 |
-
from langchain_core.messages import SystemMessage, HumanMessage
|
| 22 |
-
|
| 23 |
from ..state.graph_state import WorkflowState, AgentRole, make_agent_event, make_memory_entry
|
| 24 |
from ..memory.memory_store import long_term, short_term
|
| 25 |
from ..core.config import get_settings
|
|
@@ -27,31 +23,6 @@ from ..core.logger import get_logger
|
|
| 27 |
|
| 28 |
log = get_logger(__name__)
|
| 29 |
|
| 30 |
-
MEMORY_RETRIEVE_SYSTEM = """You are a memory retrieval assistant. Given a task, identify what past knowledge would be most relevant.
|
| 31 |
-
|
| 32 |
-
Output a JSON list of search queries to retrieve relevant memories:
|
| 33 |
-
{
|
| 34 |
-
"queries": ["query 1", "query 2", "query 3"]
|
| 35 |
-
}
|
| 36 |
-
|
| 37 |
-
Focus on: similar past tasks, relevant domain knowledge, successful strategies."""
|
| 38 |
-
|
| 39 |
-
MEMORY_STORE_SYSTEM = """You are a memory extraction assistant. Given a completed task execution, extract valuable learnings to remember for the future.
|
| 40 |
-
|
| 41 |
-
Output a JSON list of memories to store:
|
| 42 |
-
{
|
| 43 |
-
"memories": [
|
| 44 |
-
{
|
| 45 |
-
"content": "The memory content β be specific and actionable",
|
| 46 |
-
"memory_type": "episodic|semantic|procedural",
|
| 47 |
-
"importance": 0.0-1.0,
|
| 48 |
-
"tags": ["tag1", "tag2"]
|
| 49 |
-
}
|
| 50 |
-
]
|
| 51 |
-
}
|
| 52 |
-
|
| 53 |
-
Extract: key findings, successful approaches, useful facts, things to avoid."""
|
| 54 |
-
|
| 55 |
|
| 56 |
def memory_retrieve_node(state: WorkflowState) -> WorkflowState:
|
| 57 |
"""Run at start of workflow β retrieve relevant memories."""
|
|
@@ -67,19 +38,8 @@ def memory_retrieve_node(state: WorkflowState) -> WorkflowState:
|
|
| 67 |
log.info("Restored from short-term cache")
|
| 68 |
return {**state, **cached_state, "task_id": state["task_id"]}
|
| 69 |
|
| 70 |
-
|
| 71 |
-
|
| 72 |
-
model=settings.memory_model, temperature=0,
|
| 73 |
-
google_api_key=settings.google_api_key,
|
| 74 |
-
)
|
| 75 |
-
response = llm.invoke([
|
| 76 |
-
SystemMessage(content=MEMORY_RETRIEVE_SYSTEM),
|
| 77 |
-
HumanMessage(content=f"Task: {state['task']}"),
|
| 78 |
-
])
|
| 79 |
-
parsed = json.loads(response.content)
|
| 80 |
-
queries = parsed.get("queries", [state["task"][:50]])
|
| 81 |
-
except Exception:
|
| 82 |
-
queries = [state["task"][:50]]
|
| 83 |
|
| 84 |
# Retrieve from long-term memory
|
| 85 |
all_memories = []
|
|
@@ -133,30 +93,7 @@ def memory_store_node(state: WorkflowState) -> WorkflowState:
|
|
| 133 |
except Exception as e:
|
| 134 |
log.warning("Memory store failed", error=str(e))
|
| 135 |
|
| 136 |
-
#
|
| 137 |
-
if (state.get("quality_score", 0) or 0) >= 70 and state.get("final_output"):
|
| 138 |
-
try:
|
| 139 |
-
llm = ChatGoogleGenerativeAI(
|
| 140 |
-
model=settings.memory_model, temperature=0,
|
| 141 |
-
google_api_key=settings.google_api_key,
|
| 142 |
-
)
|
| 143 |
-
context = f"""Task: {state['task']}
|
| 144 |
-
Final output (first 800 chars): {str(state.get('final_output', ''))[:800]}
|
| 145 |
-
Quality score: {state.get('quality_score', 0)}
|
| 146 |
-
Steps taken: {', '.join(s['title'] for s in state['plan'] if s['status']=='done')}"""
|
| 147 |
-
|
| 148 |
-
response = llm.invoke([
|
| 149 |
-
SystemMessage(content=MEMORY_STORE_SYSTEM),
|
| 150 |
-
HumanMessage(content=context),
|
| 151 |
-
])
|
| 152 |
-
|
| 153 |
-
parsed = json.loads(response.content)
|
| 154 |
-
for mem in parsed.get("memories", [])[:4]:
|
| 155 |
-
asyncio.run(long_term.store(mem, task_id=state["task_id"]))
|
| 156 |
-
memories_stored += 1
|
| 157 |
-
|
| 158 |
-
except Exception as e:
|
| 159 |
-
log.warning("Auto-memory extraction failed", error=str(e))
|
| 160 |
|
| 161 |
log.info("Memories stored", count=memories_stored)
|
| 162 |
|
|
|
|
| 14 |
"""
|
| 15 |
from __future__ import annotations
|
| 16 |
import asyncio
|
|
|
|
| 17 |
from datetime import datetime, timezone
|
| 18 |
|
|
|
|
|
|
|
|
|
|
| 19 |
from ..state.graph_state import WorkflowState, AgentRole, make_agent_event, make_memory_entry
|
| 20 |
from ..memory.memory_store import long_term, short_term
|
| 21 |
from ..core.config import get_settings
|
|
|
|
| 23 |
|
| 24 |
log = get_logger(__name__)
|
| 25 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 26 |
|
| 27 |
def memory_retrieve_node(state: WorkflowState) -> WorkflowState:
|
| 28 |
"""Run at start of workflow β retrieve relevant memories."""
|
|
|
|
| 38 |
log.info("Restored from short-term cache")
|
| 39 |
return {**state, **cached_state, "task_id": state["task_id"]}
|
| 40 |
|
| 41 |
+
# Use task text directly β no LLM call needed to generate queries
|
| 42 |
+
queries = [state["task"][:80]]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 43 |
|
| 44 |
# Retrieve from long-term memory
|
| 45 |
all_memories = []
|
|
|
|
| 93 |
except Exception as e:
|
| 94 |
log.warning("Memory store failed", error=str(e))
|
| 95 |
|
| 96 |
+
# No extra LLM call β critic already added memory entries to new_memories
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 97 |
|
| 98 |
log.info("Memories stored", count=memories_stored)
|
| 99 |
|
backend/agents/planner.py
CHANGED
|
@@ -17,70 +17,32 @@ import json
|
|
| 17 |
import uuid
|
| 18 |
from datetime import datetime, timezone
|
| 19 |
|
| 20 |
-
from langchain_google_genai import ChatGoogleGenerativeAI
|
| 21 |
from langchain_core.messages import SystemMessage, HumanMessage
|
| 22 |
|
|
|
|
| 23 |
from ..state.graph_state import (
|
| 24 |
WorkflowState, TaskStatus, AgentRole,
|
| 25 |
make_plan_step, make_agent_event,
|
| 26 |
)
|
| 27 |
-
from ..core.config import get_settings
|
| 28 |
from ..core.logger import get_logger
|
| 29 |
|
| 30 |
log = get_logger(__name__)
|
| 31 |
|
| 32 |
-
PLANNER_SYSTEM = """You are a
|
| 33 |
-
|
| 34 |
-
## Your responsibilities
|
| 35 |
-
1. Analyze the task and identify sub-goals
|
| 36 |
-
2. Create a MINIMAL but COMPLETE plan (3β8 steps max)
|
| 37 |
-
3. Assign the right tool to each step
|
| 38 |
-
4. Specify dependencies between steps
|
| 39 |
-
5. Be concrete β vague steps fail
|
| 40 |
-
|
| 41 |
-
## Available tools for steps
|
| 42 |
-
- web_search: Search for information online
|
| 43 |
-
- fetch_url: Read content from a specific URL
|
| 44 |
-
- calculate: Perform mathematical calculations
|
| 45 |
-
- run_python: Execute Python code for data processing
|
| 46 |
-
- write_file: Save results or drafts to files
|
| 47 |
-
- read_file: Read previously saved content
|
| 48 |
-
- get_datetime: Get current date/time
|
| 49 |
-
- synthesize: Combine all results into final answer (ALWAYS last step)
|
| 50 |
-
|
| 51 |
-
## Output format (JSON)
|
| 52 |
-
Return ONLY a JSON object:
|
| 53 |
-
{
|
| 54 |
-
"thinking": "Brief analysis of the task",
|
| 55 |
-
"plan": [
|
| 56 |
-
{
|
| 57 |
-
"step_id": "step_1",
|
| 58 |
-
"title": "Short title",
|
| 59 |
-
"description": "What to do and why",
|
| 60 |
-
"tool": "tool_name",
|
| 61 |
-
"depends_on": []
|
| 62 |
-
}
|
| 63 |
-
],
|
| 64 |
-
"estimated_complexity": "low|medium|high"
|
| 65 |
-
}
|
| 66 |
|
| 67 |
-
|
| 68 |
-
|
| 69 |
-
- Later steps list their prerequisites in depends_on
|
| 70 |
-
- Always end with a "synthesize" step that depends on all prior steps
|
| 71 |
-
- If replanning after critique, address the specific issues raised
|
| 72 |
-
- NEVER include steps for things already completed"""
|
| 73 |
|
|
|
|
|
|
|
| 74 |
|
| 75 |
-
|
| 76 |
-
## Replanning context
|
| 77 |
-
Previous plan failed or received critique. Address these issues:
|
| 78 |
-
{critique}
|
| 79 |
|
| 80 |
-
Completed steps so far:
|
| 81 |
-
{completed_steps}
|
| 82 |
|
| 83 |
-
|
|
|
|
|
|
|
|
|
|
| 84 |
|
| 85 |
|
| 86 |
PLANNING_TOOL = {
|
|
@@ -119,12 +81,9 @@ def planner_node(state: WorkflowState) -> WorkflowState:
|
|
| 119 |
LangGraph node β runs the Planner agent.
|
| 120 |
Called at start and whenever Critic requests replanning.
|
| 121 |
"""
|
|
|
|
| 122 |
settings = get_settings()
|
| 123 |
-
llm =
|
| 124 |
-
model=settings.planner_model,
|
| 125 |
-
temperature=0.2,
|
| 126 |
-
google_api_key=settings.google_api_key,
|
| 127 |
-
)
|
| 128 |
|
| 129 |
log.info("Planner running", task_id=state["task_id"], iteration=state["iteration"])
|
| 130 |
|
|
@@ -158,7 +117,7 @@ def planner_node(state: WorkflowState) -> WorkflowState:
|
|
| 158 |
response = llm.invoke(
|
| 159 |
[SystemMessage(content=system_content), HumanMessage(content=user_msg)],
|
| 160 |
tools=[PLANNING_TOOL],
|
| 161 |
-
tool_choice="submit_plan",
|
| 162 |
)
|
| 163 |
|
| 164 |
tool_call = response.tool_calls[0] if response.tool_calls else None
|
|
@@ -168,9 +127,10 @@ def planner_node(state: WorkflowState) -> WorkflowState:
|
|
| 168 |
plan_data = tool_call["args"]
|
| 169 |
thinking = plan_data.get("thinking", "")
|
| 170 |
|
| 171 |
-
# Convert to plan steps
|
| 172 |
new_plan = []
|
| 173 |
-
|
|
|
|
| 174 |
new_plan.append(make_plan_step(
|
| 175 |
step_id=step.get("step_id", f"step_{len(new_plan)+1}"),
|
| 176 |
title=step.get("title", ""),
|
|
|
|
| 17 |
import uuid
|
| 18 |
from datetime import datetime, timezone
|
| 19 |
|
|
|
|
| 20 |
from langchain_core.messages import SystemMessage, HumanMessage
|
| 21 |
|
| 22 |
+
from ..core.llm import get_llm
|
| 23 |
from ..state.graph_state import (
|
| 24 |
WorkflowState, TaskStatus, AgentRole,
|
| 25 |
make_plan_step, make_agent_event,
|
| 26 |
)
|
|
|
|
| 27 |
from ..core.logger import get_logger
|
| 28 |
|
| 29 |
log = get_logger(__name__)
|
| 30 |
|
| 31 |
+
PLANNER_SYSTEM = """You are a task planner. Decompose the task into 2-4 steps MAX.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
|
| 33 |
+
Tools: web_search, fetch_url, calculate, run_python, write_file, read_file, get_datetime, synthesize
|
| 34 |
+
Always end with a synthesize step.
|
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
|
| 36 |
+
Return ONLY JSON:
|
| 37 |
+
{"thinking":"one line","plan":[{"step_id":"step_1","title":"short","description":"what to do","tool":"tool_name","depends_on":[]}],"estimated_complexity":"low|medium|high"}
|
| 38 |
|
| 39 |
+
Rules: minimal steps, no redundancy, synthesize last, depends_on lists prerequisite step_ids."""
|
|
|
|
|
|
|
|
|
|
| 40 |
|
|
|
|
|
|
|
| 41 |
|
| 42 |
+
REPLAN_ADDITION = """
|
| 43 |
+
Replanning β fix these issues: {critique}
|
| 44 |
+
Already done: {completed_steps}
|
| 45 |
+
Only plan remaining steps."""
|
| 46 |
|
| 47 |
|
| 48 |
PLANNING_TOOL = {
|
|
|
|
| 81 |
LangGraph node β runs the Planner agent.
|
| 82 |
Called at start and whenever Critic requests replanning.
|
| 83 |
"""
|
| 84 |
+
from ..core.config import get_settings
|
| 85 |
settings = get_settings()
|
| 86 |
+
llm = get_llm("planner", temperature=0.2)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 87 |
|
| 88 |
log.info("Planner running", task_id=state["task_id"], iteration=state["iteration"])
|
| 89 |
|
|
|
|
| 117 |
response = llm.invoke(
|
| 118 |
[SystemMessage(content=system_content), HumanMessage(content=user_msg)],
|
| 119 |
tools=[PLANNING_TOOL],
|
| 120 |
+
tool_choice={"type": "function", "function": {"name": "submit_plan"}},
|
| 121 |
)
|
| 122 |
|
| 123 |
tool_call = response.tool_calls[0] if response.tool_calls else None
|
|
|
|
| 127 |
plan_data = tool_call["args"]
|
| 128 |
thinking = plan_data.get("thinking", "")
|
| 129 |
|
| 130 |
+
# Convert to plan steps (hard cap)
|
| 131 |
new_plan = []
|
| 132 |
+
raw_steps = plan_data.get("plan", [])[:settings.max_plan_steps]
|
| 133 |
+
for step in raw_steps:
|
| 134 |
new_plan.append(make_plan_step(
|
| 135 |
step_id=step.get("step_id", f"step_{len(new_plan)+1}"),
|
| 136 |
title=step.get("title", ""),
|
backend/core/config.py
CHANGED
|
@@ -1,38 +1,56 @@
|
|
| 1 |
-
"""backend/core/config.py"""
|
| 2 |
from functools import lru_cache
|
| 3 |
-
from typing import Literal
|
| 4 |
from pydantic_settings import BaseSettings, SettingsConfigDict
|
| 5 |
|
| 6 |
|
| 7 |
class Settings(BaseSettings):
|
| 8 |
model_config = SettingsConfigDict(env_file=".env", extra="ignore", case_sensitive=False)
|
| 9 |
|
| 10 |
-
#
|
| 11 |
google_api_key: str = ""
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
critic_model: str = "gemini-2.5-flash"
|
| 15 |
-
memory_model: str = "gemini-2.5-flash"
|
| 16 |
|
| 17 |
-
#
|
| 18 |
-
redis_url: str = "redis://localhost:6379"
|
| 19 |
-
redis_ttl: int = 86400 # 24h default
|
| 20 |
-
|
| 21 |
-
# DB
|
| 22 |
-
database_url: str = "sqlite+aiosqlite:///./agent_system.db"
|
| 23 |
-
|
| 24 |
-
# App
|
| 25 |
app_env: str = "development"
|
| 26 |
app_host: str = "0.0.0.0"
|
| 27 |
app_port: int = 8000
|
| 28 |
log_level: str = "INFO"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 29 |
|
| 30 |
-
#
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 36 |
|
| 37 |
@property
|
| 38 |
def is_dev(self) -> bool:
|
|
|
|
| 1 |
+
"""backend/core/config.py β All non-secret config lives here. Secrets come from .env."""
|
| 2 |
from functools import lru_cache
|
|
|
|
| 3 |
from pydantic_settings import BaseSettings, SettingsConfigDict
|
| 4 |
|
| 5 |
|
| 6 |
class Settings(BaseSettings):
|
| 7 |
model_config = SettingsConfigDict(env_file=".env", extra="ignore", case_sensitive=False)
|
| 8 |
|
| 9 |
+
# ββ Secrets (from .env) βββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 10 |
google_api_key: str = ""
|
| 11 |
+
groq_api_key: str = ""
|
| 12 |
+
serper_api_key: str = ""
|
|
|
|
|
|
|
| 13 |
|
| 14 |
+
# ββ App βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 15 |
app_env: str = "development"
|
| 16 |
app_host: str = "0.0.0.0"
|
| 17 |
app_port: int = 8000
|
| 18 |
log_level: str = "INFO"
|
| 19 |
+
frontend_url: str = "http://localhost:5500"
|
| 20 |
+
|
| 21 |
+
# ββ Database ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 22 |
+
database_url: str = "sqlite+aiosqlite:///./agent_system.db"
|
| 23 |
+
|
| 24 |
+
# ββ Redis βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 25 |
+
redis_host: str = "localhost"
|
| 26 |
+
redis_port: int = 6379
|
| 27 |
+
redis_db: int = 0
|
| 28 |
+
redis_url: str = "redis://localhost:6379"
|
| 29 |
+
redis_ttl: int = 86400
|
| 30 |
+
|
| 31 |
+
# ββ LLM provider: "gemini" or "groq" βββββββββββββββββββββββββββββββββββββ
|
| 32 |
+
llm_provider: str = "groq"
|
| 33 |
+
|
| 34 |
+
# Gemini models
|
| 35 |
+
planner_model: str = "gemini-2.5-flash"
|
| 36 |
+
executor_model: str = "gemini-2.5-flash"
|
| 37 |
+
critic_model: str = "gemini-2.5-flash"
|
| 38 |
+
memory_model: str = "gemini-2.5-flash"
|
| 39 |
|
| 40 |
+
# Groq models
|
| 41 |
+
groq_planner_model: str = "llama-3.3-70b-versatile"
|
| 42 |
+
groq_executor_model: str = "llama-3.3-70b-versatile"
|
| 43 |
+
groq_critic_model: str = "llama-3.1-8b-instant"
|
| 44 |
+
groq_memory_model: str = "llama-3.1-8b-instant"
|
| 45 |
+
|
| 46 |
+
# ββ Agent behaviour βββββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 47 |
+
max_iterations: int = 3
|
| 48 |
+
max_retries: int = 1
|
| 49 |
+
step_timeout: int = 60
|
| 50 |
+
enable_reflection: bool = True
|
| 51 |
+
enable_memory: bool = True
|
| 52 |
+
max_plan_steps: int = 4
|
| 53 |
+
max_output_tokens: int = 512
|
| 54 |
|
| 55 |
@property
|
| 56 |
def is_dev(self) -> bool:
|
backend/core/llm.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""backend/core/llm.py β LLM factory supporting Gemini and Grok."""
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
def get_llm(role: str, temperature: float = 0.1):
|
| 6 |
+
"""
|
| 7 |
+
Return the right LLM based on LLM_PROVIDER in .env.
|
| 8 |
+
role: "planner" | "executor" | "critic" | "memory"
|
| 9 |
+
"""
|
| 10 |
+
from .config import get_settings
|
| 11 |
+
settings = get_settings()
|
| 12 |
+
|
| 13 |
+
max_tokens = settings.max_output_tokens
|
| 14 |
+
|
| 15 |
+
if settings.llm_provider == "groq":
|
| 16 |
+
from langchain_groq import ChatGroq
|
| 17 |
+
|
| 18 |
+
model_map = {
|
| 19 |
+
"planner": settings.groq_planner_model,
|
| 20 |
+
"executor": settings.groq_executor_model,
|
| 21 |
+
"critic": settings.groq_critic_model,
|
| 22 |
+
"memory": settings.groq_memory_model,
|
| 23 |
+
}
|
| 24 |
+
return ChatGroq(
|
| 25 |
+
model=model_map.get(role, "llama-3.3-70b-versatile"),
|
| 26 |
+
temperature=temperature,
|
| 27 |
+
api_key=settings.groq_api_key,
|
| 28 |
+
max_tokens=max_tokens,
|
| 29 |
+
)
|
| 30 |
+
|
| 31 |
+
else: # gemini (default)
|
| 32 |
+
from langchain_google_genai import ChatGoogleGenerativeAI
|
| 33 |
+
|
| 34 |
+
model_map = {
|
| 35 |
+
"planner": settings.planner_model,
|
| 36 |
+
"executor": settings.executor_model,
|
| 37 |
+
"critic": settings.critic_model,
|
| 38 |
+
"memory": settings.memory_model,
|
| 39 |
+
}
|
| 40 |
+
return ChatGoogleGenerativeAI(
|
| 41 |
+
model=model_map.get(role, settings.planner_model),
|
| 42 |
+
temperature=temperature,
|
| 43 |
+
google_api_key=settings.google_api_key,
|
| 44 |
+
max_output_tokens=max_tokens,
|
| 45 |
+
)
|
backend/tools/registry.py
CHANGED
|
@@ -33,22 +33,45 @@ def err(message: str, details: str = "") -> dict:
|
|
| 33 |
# ββ Web search ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 34 |
|
| 35 |
async def web_search(query: str, max_results: int = 5) -> dict:
|
| 36 |
-
"""Search
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
try:
|
| 38 |
from duckduckgo_search import DDGS
|
| 39 |
loop = asyncio.get_event_loop()
|
| 40 |
|
| 41 |
def _search():
|
| 42 |
with DDGS() as ddgs:
|
| 43 |
-
|
| 44 |
-
return results
|
| 45 |
|
| 46 |
results = await loop.run_in_executor(None, _search)
|
| 47 |
formatted = "\n\n".join(
|
| 48 |
f"**{r.get('title', '')}**\n{r.get('href', '')}\n{r.get('body', '')}"
|
| 49 |
for r in results
|
| 50 |
)
|
| 51 |
-
return ok({"query": query, "results": formatted[:3000], "count": len(results)})
|
| 52 |
except Exception as e:
|
| 53 |
return err(f"Search failed: {e}", str(e))
|
| 54 |
|
|
@@ -124,28 +147,41 @@ def list_files() -> dict:
|
|
| 124 |
|
| 125 |
def run_python(code: str) -> dict:
|
| 126 |
"""
|
| 127 |
-
Execute Python code in a
|
| 128 |
-
|
| 129 |
-
PRODUCTION NOTE: Use Docker sandbox / e2b in real deployment.
|
| 130 |
"""
|
| 131 |
-
import io,
|
| 132 |
-
# Block dangerous
|
| 133 |
-
blocked = ["
|
| 134 |
for b in blocked:
|
| 135 |
-
if f"import {b}" in code
|
| 136 |
return err(f"Blocked: '{b}' not allowed in sandbox")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 137 |
|
| 138 |
stdout_capture = io.StringIO()
|
| 139 |
try:
|
| 140 |
with contextlib.redirect_stdout(stdout_capture):
|
| 141 |
-
exec(code,
|
| 142 |
-
"len": len, "str": str, "int": int,
|
| 143 |
-
"float": float, "list": list, "dict": dict,
|
| 144 |
-
"enumerate": enumerate, "zip": zip,
|
| 145 |
-
"sorted": sorted, "sum": sum, "min": min,
|
| 146 |
-
"max": max, "abs": abs, "round": round}})
|
| 147 |
output = stdout_capture.getvalue()
|
| 148 |
-
return ok({"output": output[:2000], "code": code[:500]})
|
| 149 |
except Exception as e:
|
| 150 |
return err(f"Execution error: {type(e).__name__}: {e}", code[:200])
|
| 151 |
|
|
|
|
| 33 |
# ββ Web search ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 34 |
|
| 35 |
async def web_search(query: str, max_results: int = 5) -> dict:
|
| 36 |
+
"""Search via Serper (Google), falling back to DuckDuckGo if unavailable."""
|
| 37 |
+
from ..core.config import get_settings
|
| 38 |
+
settings = get_settings()
|
| 39 |
+
|
| 40 |
+
# ββ Primary: Serper (Google Search API) ββββββββββββββββββββββββββββββββββ
|
| 41 |
+
if settings.serper_api_key:
|
| 42 |
+
try:
|
| 43 |
+
async with httpx.AsyncClient(timeout=10) as c:
|
| 44 |
+
r = await c.post(
|
| 45 |
+
"https://google.serper.dev/search",
|
| 46 |
+
headers={"X-API-KEY": settings.serper_api_key, "Content-Type": "application/json"},
|
| 47 |
+
json={"q": query, "num": int(max_results)},
|
| 48 |
+
)
|
| 49 |
+
if r.status_code == 200:
|
| 50 |
+
data = r.json()
|
| 51 |
+
items = data.get("organic", [])
|
| 52 |
+
formatted = "\n\n".join(
|
| 53 |
+
f"**{item.get('title', '')}**\n{item.get('link', '')}\n{item.get('snippet', '')}"
|
| 54 |
+
for item in items
|
| 55 |
+
)
|
| 56 |
+
return ok({"query": query, "results": formatted[:3000], "count": len(items), "source": "serper"})
|
| 57 |
+
except Exception:
|
| 58 |
+
pass # fall through to DuckDuckGo
|
| 59 |
+
|
| 60 |
+
# ββ Fallback: DuckDuckGo ββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 61 |
try:
|
| 62 |
from duckduckgo_search import DDGS
|
| 63 |
loop = asyncio.get_event_loop()
|
| 64 |
|
| 65 |
def _search():
|
| 66 |
with DDGS() as ddgs:
|
| 67 |
+
return list(ddgs.text(query, max_results=int(max_results)))
|
|
|
|
| 68 |
|
| 69 |
results = await loop.run_in_executor(None, _search)
|
| 70 |
formatted = "\n\n".join(
|
| 71 |
f"**{r.get('title', '')}**\n{r.get('href', '')}\n{r.get('body', '')}"
|
| 72 |
for r in results
|
| 73 |
)
|
| 74 |
+
return ok({"query": query, "results": formatted[:3000], "count": len(results), "source": "duckduckgo"})
|
| 75 |
except Exception as e:
|
| 76 |
return err(f"Search failed: {e}", str(e))
|
| 77 |
|
|
|
|
| 147 |
|
| 148 |
def run_python(code: str) -> dict:
|
| 149 |
"""
|
| 150 |
+
Execute Python code in a sandbox with common data-science libs available.
|
| 151 |
+
Blocks truly dangerous ops (os, subprocess, socket, open).
|
|
|
|
| 152 |
"""
|
| 153 |
+
import io, contextlib, builtins
|
| 154 |
+
# Block dangerous modules only
|
| 155 |
+
blocked = ["subprocess", "socket", "requests"]
|
| 156 |
for b in blocked:
|
| 157 |
+
if f"import {b}" in code:
|
| 158 |
return err(f"Blocked: '{b}' not allowed in sandbox")
|
| 159 |
+
if "open(" in code:
|
| 160 |
+
return err("Blocked: file 'open()' not allowed in sandbox")
|
| 161 |
+
|
| 162 |
+
# Allow standard builtins + safe data libs
|
| 163 |
+
import math, json
|
| 164 |
+
from datetime import datetime as _dt
|
| 165 |
+
safe_globals = {
|
| 166 |
+
"__builtins__": builtins, # full builtins so import works
|
| 167 |
+
"math": math,
|
| 168 |
+
"json": json,
|
| 169 |
+
"datetime": _dt,
|
| 170 |
+
}
|
| 171 |
+
# Try to inject optional libs if installed
|
| 172 |
+
for lib in ["pandas", "numpy", "matplotlib"]:
|
| 173 |
+
try:
|
| 174 |
+
import importlib
|
| 175 |
+
safe_globals[lib.split(".")[0]] = importlib.import_module(lib)
|
| 176 |
+
except ImportError:
|
| 177 |
+
pass
|
| 178 |
|
| 179 |
stdout_capture = io.StringIO()
|
| 180 |
try:
|
| 181 |
with contextlib.redirect_stdout(stdout_capture):
|
| 182 |
+
exec(code, safe_globals)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 183 |
output = stdout_capture.getvalue()
|
| 184 |
+
return ok({"output": output[:2000] or "(no output)", "code": code[:500]})
|
| 185 |
except Exception as e:
|
| 186 |
return err(f"Execution error: {type(e).__name__}: {e}", code[:200])
|
| 187 |
|
check_api.py
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Check Gemini API key, available models, and rate limits."""
|
| 2 |
+
import asyncio, os, sys
|
| 3 |
+
from dotenv import load_dotenv
|
| 4 |
+
import httpx
|
| 5 |
+
|
| 6 |
+
load_dotenv()
|
| 7 |
+
KEY = os.getenv("GOOGLE_API_KEY", "")
|
| 8 |
+
|
| 9 |
+
BASE = "https://generativelanguage.googleapis.com/v1beta"
|
| 10 |
+
MODELS_TO_TEST = [
|
| 11 |
+
"gemini-2.5-flash",
|
| 12 |
+
"gemini-2.5-pro",
|
| 13 |
+
"gemini-2.0-flash",
|
| 14 |
+
"gemini-1.5-flash",
|
| 15 |
+
"gemini-1.5-pro",
|
| 16 |
+
]
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
async def main():
|
| 20 |
+
if not KEY:
|
| 21 |
+
print("[X] GOOGLE_API_KEY not set in .env")
|
| 22 |
+
sys.exit(1)
|
| 23 |
+
print(f"Key: ...{KEY[-6:]}\n")
|
| 24 |
+
|
| 25 |
+
async with httpx.AsyncClient(timeout=20) as c:
|
| 26 |
+
|
| 27 |
+
# 1) List all available models
|
| 28 |
+
print("=" * 55)
|
| 29 |
+
print("AVAILABLE MODELS")
|
| 30 |
+
print("=" * 55)
|
| 31 |
+
r = await c.get(f"{BASE}/models?key={KEY}")
|
| 32 |
+
if r.status_code != 200:
|
| 33 |
+
print(f"[X] Could not list models: {r.status_code} {r.text[:200]}")
|
| 34 |
+
else:
|
| 35 |
+
models = r.json().get("models", [])
|
| 36 |
+
gen_models = [m for m in models if "generateContent" in m.get("supportedGenerationMethods", [])]
|
| 37 |
+
for m in gen_models:
|
| 38 |
+
name = m["name"].replace("models/", "")
|
| 39 |
+
limit = m.get("description", "")[:60]
|
| 40 |
+
print(f" {name:<35} rpm={m.get('rpmLimit','?'):>6} tpm={m.get('tpmLimit','?'):>10}")
|
| 41 |
+
|
| 42 |
+
# 2) Test each target model with a tiny call
|
| 43 |
+
print("\n" + "=" * 55)
|
| 44 |
+
print("MODEL PING TEST")
|
| 45 |
+
print("=" * 55)
|
| 46 |
+
body = {"contents": [{"parts": [{"text": "Say OK"}]}],
|
| 47 |
+
"generationConfig": {"maxOutputTokens": 5}}
|
| 48 |
+
for model in MODELS_TO_TEST:
|
| 49 |
+
url = f"{BASE}/models/{model}:generateContent?key={KEY}"
|
| 50 |
+
r = await c.post(url, json=body)
|
| 51 |
+
if r.status_code == 200:
|
| 52 |
+
status = "[OK]"
|
| 53 |
+
elif r.status_code == 429:
|
| 54 |
+
status = "[429 rate limit]"
|
| 55 |
+
elif r.status_code == 404:
|
| 56 |
+
status = "[404 not found]"
|
| 57 |
+
elif r.status_code == 403:
|
| 58 |
+
status = "[403 denied]"
|
| 59 |
+
else:
|
| 60 |
+
status = f"[{r.status_code}]"
|
| 61 |
+
print(f" {model:<30} {status}")
|
| 62 |
+
|
| 63 |
+
# 3) Quota info from a real call on the working model
|
| 64 |
+
print("\n" + "=" * 55)
|
| 65 |
+
print("RATE LIMITS (from API metadata)")
|
| 66 |
+
print("=" * 55)
|
| 67 |
+
r = await c.get(f"{BASE}/models/gemini-2.5-flash?key={KEY}")
|
| 68 |
+
if r.status_code == 200:
|
| 69 |
+
m = r.json()
|
| 70 |
+
print(f" Model : {m.get('displayName','')}")
|
| 71 |
+
print(f" Input limit : {m.get('inputTokenLimit','?'):,} tokens")
|
| 72 |
+
print(f" Output limit : {m.get('outputTokenLimit','?'):,} tokens")
|
| 73 |
+
print(f" RPM limit : {m.get('rpmLimit', 'see https://ai.dev/rate-limit')}")
|
| 74 |
+
print(f" TPM limit : {m.get('tpmLimit', 'see https://ai.dev/rate-limit')}")
|
| 75 |
+
print(f"\n Free tier caps: 15 RPM / 1,000,000 TPM / 1,500 RPD")
|
| 76 |
+
print(f" Full limits : https://ai.google.dev/gemini-api/docs/rate-limits")
|
| 77 |
+
|
| 78 |
+
|
| 79 |
+
asyncio.run(main())
|
frontend/app.js
CHANGED
|
@@ -1,5 +1,7 @@
|
|
| 1 |
/* βββ Config ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 2 |
-
|
|
|
|
|
|
|
| 3 |
|
| 4 |
/* βββ State βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 5 |
const state = {
|
|
@@ -397,9 +399,14 @@ async function runBatch(task) {
|
|
| 397 |
}
|
| 398 |
|
| 399 |
/* βββ Show output ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 400 |
function showOutput(text, score) {
|
| 401 |
outputCard.style.display = '';
|
| 402 |
-
outputContent.textContent = text;
|
| 403 |
if (score != null) {
|
| 404 |
const s = Math.round(score);
|
| 405 |
qualityBadge.textContent = `${s}/100`;
|
|
|
|
| 1 |
/* βββ Config ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 2 |
+
// Empty string = same origin (works on HF Spaces and any deployment).
|
| 3 |
+
// Falls back to localhost only when running locally on a different port.
|
| 4 |
+
const API_BASE = window.location.hostname === 'localhost' ? 'http://localhost:8000' : '';
|
| 5 |
|
| 6 |
/* βββ State βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 7 |
const state = {
|
|
|
|
| 399 |
}
|
| 400 |
|
| 401 |
/* βββ Show output ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 402 |
+
function cleanOutput(text) {
|
| 403 |
+
// Strip LaTeX boxing that Llama sometimes outputs
|
| 404 |
+
return (text || '').replace(/\$\\boxed\{([^}]+)\}\$/g, '$1').replace(/\\boxed\{([^}]+)\}/g, '$1');
|
| 405 |
+
}
|
| 406 |
+
|
| 407 |
function showOutput(text, score) {
|
| 408 |
outputCard.style.display = '';
|
| 409 |
+
outputContent.textContent = cleanOutput(text);
|
| 410 |
if (score != null) {
|
| 411 |
const s = Math.round(score);
|
| 412 |
qualityBadge.textContent = `${s}/100`;
|
frontend/index.html
CHANGED
|
@@ -22,7 +22,7 @@
|
|
| 22 |
</svg>
|
| 23 |
<span class="logo-text">Multi-Agent System</span>
|
| 24 |
</div>
|
| 25 |
-
<span class="powered-by">Powered by <span class="gemini-badge">
|
| 26 |
</div>
|
| 27 |
<div class="header-right">
|
| 28 |
<div class="health-indicator" id="healthDot" title="Backend status"></div>
|
|
|
|
| 22 |
</svg>
|
| 23 |
<span class="logo-text">Multi-Agent System</span>
|
| 24 |
</div>
|
| 25 |
+
<span class="powered-by">Powered by <span class="gemini-badge">Groq Β· Llama 3.3 70B</span></span>
|
| 26 |
</div>
|
| 27 |
<div class="header-right">
|
| 28 |
<div class="health-indicator" id="healthDot" title="Backend status"></div>
|
frontend/style.css
CHANGED
|
@@ -139,15 +139,20 @@ body {
|
|
| 139 |
/* βββ Main Layout βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 140 |
.layout {
|
| 141 |
display: grid;
|
| 142 |
-
grid-template-columns:
|
|
|
|
| 143 |
gap: 0;
|
| 144 |
flex: 1;
|
|
|
|
|
|
|
| 145 |
overflow: hidden;
|
| 146 |
}
|
| 147 |
|
| 148 |
.panel {
|
| 149 |
display: flex;
|
| 150 |
flex-direction: column;
|
|
|
|
|
|
|
| 151 |
overflow: hidden;
|
| 152 |
background: var(--bg-panel);
|
| 153 |
}
|
|
@@ -168,6 +173,9 @@ body {
|
|
| 168 |
|
| 169 |
.panel-right {
|
| 170 |
border-left: 1px solid var(--border);
|
|
|
|
|
|
|
|
|
|
| 171 |
overflow-y: auto;
|
| 172 |
}
|
| 173 |
|
|
@@ -327,7 +335,7 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 327 |
.btn-icon-arrow { flex-shrink: 0; }
|
| 328 |
|
| 329 |
/* βββ Left Panel: History βββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 330 |
-
.history-section { padding: 16px; flex: 1; }
|
| 331 |
|
| 332 |
.task-history { display: flex; flex-direction: column; gap: 6px; }
|
| 333 |
|
|
@@ -415,13 +423,13 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 415 |
.status-badge.failed { background: rgba(239,68,68,0.15); color: var(--red); border-color: rgba(239,68,68,0.3); }
|
| 416 |
|
| 417 |
/* βββ Agent Graph βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 418 |
-
.agent-graph { padding:
|
| 419 |
|
| 420 |
.graph-track {
|
| 421 |
display: flex;
|
| 422 |
align-items: center;
|
| 423 |
gap: 0;
|
| 424 |
-
min-width:
|
| 425 |
margin: 0 auto;
|
| 426 |
}
|
| 427 |
|
|
@@ -430,11 +438,12 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 430 |
flex-direction: column;
|
| 431 |
align-items: center;
|
| 432 |
gap: 4px;
|
| 433 |
-
padding:
|
| 434 |
background: var(--bg-input);
|
| 435 |
border: 1.5px solid var(--border);
|
| 436 |
border-radius: var(--radius);
|
| 437 |
-
min-width:
|
|
|
|
| 438 |
transition: all var(--transition);
|
| 439 |
cursor: default;
|
| 440 |
}
|
|
@@ -464,9 +473,13 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 464 |
font-size: 9px;
|
| 465 |
font-weight: 600;
|
| 466 |
text-transform: uppercase;
|
| 467 |
-
letter-spacing: 0.
|
| 468 |
-
padding:
|
| 469 |
border-radius: 99px;
|
|
|
|
|
|
|
|
|
|
|
|
|
| 470 |
}
|
| 471 |
.node-state.idle { background: var(--bg-card); color: var(--text-dim); }
|
| 472 |
.node-state.running { background: rgba(124,58,237,0.2); color: var(--purple-light); animation: blink 1s step-start infinite; }
|
|
@@ -571,7 +584,8 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 571 |
color: var(--text);
|
| 572 |
white-space: pre-wrap;
|
| 573 |
word-break: break-word;
|
| 574 |
-
|
|
|
|
| 575 |
overflow-y: auto;
|
| 576 |
}
|
| 577 |
|
|
@@ -579,10 +593,10 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 579 |
.events-section {
|
| 580 |
padding: 14px;
|
| 581 |
border-bottom: 1px solid var(--border);
|
| 582 |
-
flex: 1;
|
| 583 |
display: flex;
|
| 584 |
flex-direction: column;
|
| 585 |
-
min-height:
|
|
|
|
| 586 |
}
|
| 587 |
|
| 588 |
.btn-ghost-sm {
|
|
@@ -605,7 +619,6 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 605 |
flex-direction: column;
|
| 606 |
gap: 4px;
|
| 607 |
min-height: 0;
|
| 608 |
-
max-height: 300px;
|
| 609 |
}
|
| 610 |
|
| 611 |
.event-item {
|
|
@@ -638,7 +651,7 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 638 |
.event-msg { font-size: 11px; color: var(--text-muted); line-height: 1.4; }
|
| 639 |
|
| 640 |
/* βββ Right Panel: Metrics ββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 641 |
-
.metrics-section { padding: 14px; border-bottom: 1px solid var(--border); }
|
| 642 |
|
| 643 |
.metrics-grid {
|
| 644 |
display: grid;
|
|
@@ -668,7 +681,7 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 668 |
.metric-label { font-size: 9px; color: var(--text-dim); text-transform: uppercase; letter-spacing: 0.06em; }
|
| 669 |
|
| 670 |
/* βββ Right Panel: Tools ββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 671 |
-
.tools-section { padding: 14px; }
|
| 672 |
|
| 673 |
.tool-list { display: flex; flex-direction: column; gap: 3px; }
|
| 674 |
.tool-item {
|
|
@@ -718,12 +731,17 @@ textarea::placeholder { color: var(--text-dim); }
|
|
| 718 |
}
|
| 719 |
|
| 720 |
/* βββ Responsive ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 721 |
-
@media (max-width:
|
| 722 |
-
.layout { grid-template-columns:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 723 |
}
|
| 724 |
-
@media (max-width:
|
| 725 |
.layout { grid-template-columns: 1fr; }
|
| 726 |
-
.panel-left
|
| 727 |
body { overflow: auto; }
|
| 728 |
.panel-center { min-height: 100vh; }
|
| 729 |
}
|
|
|
|
| 139 |
/* βββ Main Layout βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 140 |
.layout {
|
| 141 |
display: grid;
|
| 142 |
+
grid-template-columns: 280px 1fr 260px;
|
| 143 |
+
grid-template-rows: 1fr;
|
| 144 |
gap: 0;
|
| 145 |
flex: 1;
|
| 146 |
+
min-height: 0;
|
| 147 |
+
height: 0; /* forces grid rows to respect flex parent height */
|
| 148 |
overflow: hidden;
|
| 149 |
}
|
| 150 |
|
| 151 |
.panel {
|
| 152 |
display: flex;
|
| 153 |
flex-direction: column;
|
| 154 |
+
min-height: 0;
|
| 155 |
+
height: 100%;
|
| 156 |
overflow: hidden;
|
| 157 |
background: var(--bg-panel);
|
| 158 |
}
|
|
|
|
| 173 |
|
| 174 |
.panel-right {
|
| 175 |
border-left: 1px solid var(--border);
|
| 176 |
+
display: flex;
|
| 177 |
+
flex-direction: column;
|
| 178 |
+
min-height: 0;
|
| 179 |
overflow-y: auto;
|
| 180 |
}
|
| 181 |
|
|
|
|
| 335 |
.btn-icon-arrow { flex-shrink: 0; }
|
| 336 |
|
| 337 |
/* βββ Left Panel: History βββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 338 |
+
.history-section { padding: 16px; flex: 1; overflow-y: auto; min-height: 0; }
|
| 339 |
|
| 340 |
.task-history { display: flex; flex-direction: column; gap: 6px; }
|
| 341 |
|
|
|
|
| 423 |
.status-badge.failed { background: rgba(239,68,68,0.15); color: var(--red); border-color: rgba(239,68,68,0.3); }
|
| 424 |
|
| 425 |
/* βββ Agent Graph βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 426 |
+
.agent-graph { padding: 16px; overflow-x: auto; }
|
| 427 |
|
| 428 |
.graph-track {
|
| 429 |
display: flex;
|
| 430 |
align-items: center;
|
| 431 |
gap: 0;
|
| 432 |
+
min-width: max-content;
|
| 433 |
margin: 0 auto;
|
| 434 |
}
|
| 435 |
|
|
|
|
| 438 |
flex-direction: column;
|
| 439 |
align-items: center;
|
| 440 |
gap: 4px;
|
| 441 |
+
padding: 10px 12px;
|
| 442 |
background: var(--bg-input);
|
| 443 |
border: 1.5px solid var(--border);
|
| 444 |
border-radius: var(--radius);
|
| 445 |
+
min-width: 88px;
|
| 446 |
+
width: 88px;
|
| 447 |
transition: all var(--transition);
|
| 448 |
cursor: default;
|
| 449 |
}
|
|
|
|
| 473 |
font-size: 9px;
|
| 474 |
font-weight: 600;
|
| 475 |
text-transform: uppercase;
|
| 476 |
+
letter-spacing: 0.04em;
|
| 477 |
+
padding: 2px 6px;
|
| 478 |
border-radius: 99px;
|
| 479 |
+
white-space: nowrap;
|
| 480 |
+
max-width: 100%;
|
| 481 |
+
overflow: hidden;
|
| 482 |
+
text-overflow: ellipsis;
|
| 483 |
}
|
| 484 |
.node-state.idle { background: var(--bg-card); color: var(--text-dim); }
|
| 485 |
.node-state.running { background: rgba(124,58,237,0.2); color: var(--purple-light); animation: blink 1s step-start infinite; }
|
|
|
|
| 584 |
color: var(--text);
|
| 585 |
white-space: pre-wrap;
|
| 586 |
word-break: break-word;
|
| 587 |
+
min-height: 80px;
|
| 588 |
+
max-height: 600px;
|
| 589 |
overflow-y: auto;
|
| 590 |
}
|
| 591 |
|
|
|
|
| 593 |
.events-section {
|
| 594 |
padding: 14px;
|
| 595 |
border-bottom: 1px solid var(--border);
|
|
|
|
| 596 |
display: flex;
|
| 597 |
flex-direction: column;
|
| 598 |
+
min-height: 200px;
|
| 599 |
+
max-height: 45vh;
|
| 600 |
}
|
| 601 |
|
| 602 |
.btn-ghost-sm {
|
|
|
|
| 619 |
flex-direction: column;
|
| 620 |
gap: 4px;
|
| 621 |
min-height: 0;
|
|
|
|
| 622 |
}
|
| 623 |
|
| 624 |
.event-item {
|
|
|
|
| 651 |
.event-msg { font-size: 11px; color: var(--text-muted); line-height: 1.4; }
|
| 652 |
|
| 653 |
/* βββ Right Panel: Metrics ββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 654 |
+
.metrics-section { padding: 14px; border-bottom: 1px solid var(--border); flex-shrink: 0; }
|
| 655 |
|
| 656 |
.metrics-grid {
|
| 657 |
display: grid;
|
|
|
|
| 681 |
.metric-label { font-size: 9px; color: var(--text-dim); text-transform: uppercase; letter-spacing: 0.06em; }
|
| 682 |
|
| 683 |
/* βββ Right Panel: Tools ββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 684 |
+
.tools-section { padding: 14px; flex-shrink: 0; }
|
| 685 |
|
| 686 |
.tool-list { display: flex; flex-direction: column; gap: 3px; }
|
| 687 |
.tool-item {
|
|
|
|
| 731 |
}
|
| 732 |
|
| 733 |
/* βββ Responsive ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ */
|
| 734 |
+
@media (max-width: 1200px) {
|
| 735 |
+
.layout { grid-template-columns: 240px 1fr 220px; }
|
| 736 |
+
.agent-node { min-width: 78px; width: 78px; }
|
| 737 |
+
}
|
| 738 |
+
@media (max-width: 960px) {
|
| 739 |
+
.layout { grid-template-columns: 220px 1fr; }
|
| 740 |
+
.panel-right { display: none; }
|
| 741 |
}
|
| 742 |
+
@media (max-width: 680px) {
|
| 743 |
.layout { grid-template-columns: 1fr; }
|
| 744 |
+
.panel-left { display: none; }
|
| 745 |
body { overflow: auto; }
|
| 746 |
.panel-center { min-height: 100vh; }
|
| 747 |
}
|
test.py
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Quick smoke test for the multi-agent system.
|
| 2 |
+
|
| 3 |
+
Checks:
|
| 4 |
+
1. GOOGLE_API_KEY is set and reachable
|
| 5 |
+
2. Core imports work (state, tools, orchestrator)
|
| 6 |
+
3. WorkflowState initialisation is correct
|
| 7 |
+
4. Routing logic behaves as expected
|
| 8 |
+
|
| 9 |
+
Run:
|
| 10 |
+
.venv/Scripts/python test.py
|
| 11 |
+
"""
|
| 12 |
+
from __future__ import annotations
|
| 13 |
+
|
| 14 |
+
import asyncio
|
| 15 |
+
import os
|
| 16 |
+
import sys
|
| 17 |
+
|
| 18 |
+
try:
|
| 19 |
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
| 20 |
+
except Exception:
|
| 21 |
+
pass
|
| 22 |
+
|
| 23 |
+
from dotenv import load_dotenv
|
| 24 |
+
|
| 25 |
+
load_dotenv()
|
| 26 |
+
|
| 27 |
+
SEP = "=" * 60
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def check_env() -> bool:
|
| 31 |
+
key = os.getenv("GOOGLE_API_KEY", "")
|
| 32 |
+
print(SEP)
|
| 33 |
+
print("GOOGLE_API_KEY set :", bool(key), f"(...{key[-4:]})" if key else "")
|
| 34 |
+
print("PLANNER_MODEL :", os.getenv("PLANNER_MODEL", "gemini-2.5-flash"))
|
| 35 |
+
print("DATABASE_URL :", os.getenv("DATABASE_URL", "(not set)"))
|
| 36 |
+
print("REDIS_URL :", os.getenv("REDIS_URL", "(not set)"))
|
| 37 |
+
print(SEP)
|
| 38 |
+
if not key:
|
| 39 |
+
print("[X] No GOOGLE_API_KEY found β set it in .env")
|
| 40 |
+
return False
|
| 41 |
+
return True
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
async def check_gemini(key: str) -> bool:
|
| 45 |
+
import httpx
|
| 46 |
+
|
| 47 |
+
model = os.getenv("PLANNER_MODEL", "gemini-2.5-flash")
|
| 48 |
+
url = (
|
| 49 |
+
f"https://generativelanguage.googleapis.com/v1beta/models/"
|
| 50 |
+
f"{model}:generateContent?key={key}"
|
| 51 |
+
)
|
| 52 |
+
body = {"contents": [{"parts": [{"text": "Reply with exactly: API key works"}]}]}
|
| 53 |
+
async with httpx.AsyncClient(timeout=15) as c:
|
| 54 |
+
r = await c.post(url, json=body)
|
| 55 |
+
|
| 56 |
+
print(f"\n[1] Gemini live call -> HTTP {r.status_code}")
|
| 57 |
+
if r.status_code >= 300:
|
| 58 |
+
print(" error:", r.text[:400])
|
| 59 |
+
print("\n[X] Common causes:")
|
| 60 |
+
print(" 400 API_KEY_INVALID -> wrong/expired key")
|
| 61 |
+
print(" 403 PERMISSION_DENIED -> 'Generative Language API' not enabled,")
|
| 62 |
+
print(" OR Google project access denied β")
|
| 63 |
+
print(" create a new key at https://aistudio.google.com/apikey")
|
| 64 |
+
print(" 404 model not found -> check PLANNER_MODEL in .env")
|
| 65 |
+
print(" 429 -> rate/quota limit, try again shortly")
|
| 66 |
+
return False
|
| 67 |
+
|
| 68 |
+
try:
|
| 69 |
+
reply = r.json()["candidates"][0]["content"]["parts"][0]["text"].strip()
|
| 70 |
+
except Exception:
|
| 71 |
+
reply = r.text[:200]
|
| 72 |
+
print(" model reply:", reply)
|
| 73 |
+
print("[OK] Gemini API key is working.\n")
|
| 74 |
+
return True
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def check_imports() -> bool:
|
| 78 |
+
print("[2] Checking imports ...")
|
| 79 |
+
try:
|
| 80 |
+
from backend.state.graph_state import ( # noqa: F401
|
| 81 |
+
WorkflowState, TaskStatus, StepStatus, AgentRole,
|
| 82 |
+
create_initial_state, make_plan_step, make_agent_event,
|
| 83 |
+
)
|
| 84 |
+
from backend.agents.orchestrator import ( # noqa: F401
|
| 85 |
+
build_workflow, route_after_executor,
|
| 86 |
+
route_after_critic, route_after_planner,
|
| 87 |
+
)
|
| 88 |
+
print(" backend.state OK")
|
| 89 |
+
print(" backend.orchestrator OK")
|
| 90 |
+
except ImportError as e:
|
| 91 |
+
print(f" [X] Import failed: {e}")
|
| 92 |
+
return False
|
| 93 |
+
|
| 94 |
+
try:
|
| 95 |
+
from backend.tools.registry import execute_tool, calculate # noqa: F401
|
| 96 |
+
print(" backend.tools OK")
|
| 97 |
+
except ImportError as e:
|
| 98 |
+
print(f" [X] tools import failed: {e}")
|
| 99 |
+
return False
|
| 100 |
+
|
| 101 |
+
print("[OK] All imports succeeded.\n")
|
| 102 |
+
return True
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
def check_state() -> bool:
|
| 106 |
+
print("[3] WorkflowState smoke test ...")
|
| 107 |
+
from backend.state.graph_state import (
|
| 108 |
+
create_initial_state, TaskStatus, make_plan_step, StepStatus,
|
| 109 |
+
)
|
| 110 |
+
|
| 111 |
+
state = create_initial_state("Write a hello-world script")
|
| 112 |
+
assert state["task"] == "Write a hello-world script"
|
| 113 |
+
assert state["status"] == TaskStatus.PENDING
|
| 114 |
+
assert state["plan"] == []
|
| 115 |
+
assert state["iteration"] == 0
|
| 116 |
+
assert state["total_tokens"] == 0
|
| 117 |
+
assert len(state["events"]) == 1
|
| 118 |
+
|
| 119 |
+
step = make_plan_step("s1", "Write code", "Create hello.py", tool="run_python")
|
| 120 |
+
assert step["status"] == StepStatus.PENDING
|
| 121 |
+
assert step["attempts"] == 0
|
| 122 |
+
|
| 123 |
+
print(" initial state OK")
|
| 124 |
+
print(" make_plan_step OK")
|
| 125 |
+
print("[OK] State checks passed.\n")
|
| 126 |
+
return True
|
| 127 |
+
|
| 128 |
+
|
| 129 |
+
def check_routing() -> bool:
|
| 130 |
+
print("[4] Routing logic smoke test ...")
|
| 131 |
+
from backend.state.graph_state import create_initial_state, TaskStatus, StepStatus
|
| 132 |
+
from backend.agents.orchestrator import (
|
| 133 |
+
route_after_executor, route_after_critic, route_after_planner,
|
| 134 |
+
)
|
| 135 |
+
|
| 136 |
+
def _state(status, plan=None, needs_replanning=False):
|
| 137 |
+
s = create_initial_state("test")
|
| 138 |
+
s["status"] = status
|
| 139 |
+
s["plan"] = plan or []
|
| 140 |
+
s["needs_replanning"] = needs_replanning
|
| 141 |
+
return s
|
| 142 |
+
|
| 143 |
+
# executor β critic when REFLECTING
|
| 144 |
+
assert route_after_executor(_state(TaskStatus.REFLECTING)) == "critic"
|
| 145 |
+
# executor β self when steps pending
|
| 146 |
+
assert route_after_executor(
|
| 147 |
+
_state(TaskStatus.EXECUTING, [{"step_id": "s1", "status": StepStatus.PENDING}])
|
| 148 |
+
) == "executor"
|
| 149 |
+
# executor β end on FAILED
|
| 150 |
+
assert route_after_executor(_state(TaskStatus.FAILED)) == "end"
|
| 151 |
+
# critic β planner on needs_replanning
|
| 152 |
+
assert route_after_critic(_state(TaskStatus.PLANNING, needs_replanning=True)) == "planner"
|
| 153 |
+
# critic β memory_store when approved
|
| 154 |
+
assert route_after_critic(_state(TaskStatus.COMPLETED)) == "memory_store"
|
| 155 |
+
# planner β end with empty plan
|
| 156 |
+
assert route_after_planner(_state(TaskStatus.EXECUTING, plan=[])) == "end"
|
| 157 |
+
# planner β executor with valid plan
|
| 158 |
+
assert route_after_planner(
|
| 159 |
+
_state(TaskStatus.EXECUTING, [{"step_id": "s1", "status": StepStatus.PENDING}])
|
| 160 |
+
) == "executor"
|
| 161 |
+
|
| 162 |
+
print(" route_after_executor OK")
|
| 163 |
+
print(" route_after_critic OK")
|
| 164 |
+
print(" route_after_planner OK")
|
| 165 |
+
print("[OK] Routing checks passed.\n")
|
| 166 |
+
return True
|
| 167 |
+
|
| 168 |
+
|
| 169 |
+
async def main() -> None:
|
| 170 |
+
ok = check_env()
|
| 171 |
+
if not ok:
|
| 172 |
+
sys.exit(1)
|
| 173 |
+
|
| 174 |
+
key = os.getenv("GOOGLE_API_KEY", "")
|
| 175 |
+
gemini_ok = await check_gemini(key)
|
| 176 |
+
|
| 177 |
+
imports_ok = check_imports()
|
| 178 |
+
state_ok = check_state() if imports_ok else False
|
| 179 |
+
routing_ok = check_routing() if imports_ok else False
|
| 180 |
+
|
| 181 |
+
print(SEP)
|
| 182 |
+
print("Summary")
|
| 183 |
+
print(SEP)
|
| 184 |
+
print(" Gemini API :", "[OK]" if gemini_ok else "[FAIL]")
|
| 185 |
+
print(" Imports :", "[OK]" if imports_ok else "[FAIL]")
|
| 186 |
+
print(" State :", "[OK]" if state_ok else "[FAIL]")
|
| 187 |
+
print(" Routing :", "[OK]" if routing_ok else "[FAIL]")
|
| 188 |
+
print(SEP)
|
| 189 |
+
|
| 190 |
+
system_ok = imports_ok and state_ok and routing_ok
|
| 191 |
+
if system_ok and gemini_ok:
|
| 192 |
+
print("\nAll checks passed. The system is ready.")
|
| 193 |
+
elif system_ok:
|
| 194 |
+
print("\nSystem checks passed. Fix the Gemini API key to enable LLM calls.")
|
| 195 |
+
else:
|
| 196 |
+
print("\nSystem checks failed. See above for details.")
|
| 197 |
+
sys.exit(1)
|
| 198 |
+
|
| 199 |
+
|
| 200 |
+
if __name__ == "__main__":
|
| 201 |
+
asyncio.run(main())
|