-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathagent.py
More file actions
165 lines (131 loc) · 5.95 KB
/
Copy pathagent.py
File metadata and controls
165 lines (131 loc) · 5.95 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
"""
CodeMentor Pro - Agent & Orchestration (LangGraph Upgraded)
Implements a state-of-the-art LangGraph ReAct agent bound to DuckDuckGo
to search unfamiliar libraries and return verified documentation.
"""
import re
from ddgs import DDGS
from langgraph.prebuilt import create_react_agent
from langchain_community.tools import DuckDuckGoSearchRun
from langchain_groq import ChatGroq
from config import GROQ_API_KEY1, GROQ_MODEL
from prompts import (
LEVEL_PROMPTS, JSON_OUTPUT_INSTRUCTIONS, PLAN_PROMPT,
CLEANUP_PROMPT, OPTIMIZATION_PROMPT, RESOURCE_QUERY_HINT,
TOPIC_EXTRACTION_PROMPT,
)
from llm import call_groq_json
_search_tool = DuckDuckGoSearchRun()
_search_tool.name = "duckduckgo_search" # Required for LangGraph routing
# Common stdlib modules we don't bother looking up
_KNOWN_STDLIB = {
"os", "sys", "json", "re", "math", "time", "random", "collections",
"itertools", "functools", "typing", "datetime", "string", "copy",
}
def _get_llm(temperature=0.2):
if not GROQ_API_KEY1:
raise RuntimeError("GROQ_API_KEY not set.")
return ChatGroq(model=GROQ_MODEL, temperature=temperature, api_key=GROQ_API_KEY1)
def _build_doc_agent():
"""Builds a modern LangGraph ReAct agent."""
llm = _get_llm()
tools = [_search_tool]
# LangGraph handles the system prompting and ReAct loop automatically
return create_react_agent(llm, tools)
def _detect_candidate_libraries(code: str) -> list:
"""Heuristic: pull import statements as agent search candidates."""
imports = re.findall(r"^\s*(?:import|from)\s+([\w\.]+)", code, re.MULTILINE)
top_level = {imp.split(".")[0] for imp in imports}
return sorted(top_level - _KNOWN_STDLIB)
def _youtube_thumbnail(url: str):
match = re.search(r"(?:v=|youtu\.be/)([\w-]{11})", url)
return f"https://img.youtube.com/vi/{match.group(1)}/hqdefault.jpg" if match else None
def _structured_citations(query: str, max_results: int = 3) -> list:
"""Direct DDGS search — real titles + links, so citations are always structured."""
try:
with DDGS() as ddgs:
raw = list(ddgs.text(query, max_results=max_results))
except Exception:
return []
return [
{
"title": r.get("title", "Untitled"),
"url": r.get("href", ""),
"snippet": (r.get("body", "") or "")[:160],
"thumbnail": _youtube_thumbnail(r.get("href", "")),
}
for r in raw
]
def search_documentation(code: str) -> list:
candidates = _detect_candidate_libraries(code)
if not candidates:
return []
try:
agent = _build_doc_agent()
except RuntimeError as e:
return [{"library": lib, "summary": f"(doc search unavailable: {e})", "citations": []} for lib in candidates]
findings = []
for lib in candidates:
try:
query = (
f"Search for official documentation on the '{lib}' Python library/module "
f"and summarize in 2-3 sentences what it's for and the most relevant "
f"function/API that would show up in a typical code snippet."
)
response = agent.invoke({"messages": [("user", query)]})
summary = response["messages"][-1].content
except Exception as e:
summary = f"(doc search failed: {e})"
citations = _structured_citations(f"{lib} python library documentation", max_results=3)
findings.append({"library": lib, "summary": summary, "citations": citations})
return findings
def extract_code_topic(code: str, language: str) -> str:
"""
Identifies the core algorithm/data-structure/concept the code implements
(e.g. "Dijkstra's Algorithm", "Binary Search"), so resource search can
target that exact topic instead of just the language name or raw import
names — this is what makes 'Further Resources' show topic-specific
videos/articles rather than generic language tutorials.
"""
try:
result = call_groq_json(TOPIC_EXTRACTION_PROMPT, f"Language: {language}\n\nCode:\n{code}")
topic = (result or {}).get("topic", "").strip()
return topic or f"{language} programming"
except Exception:
return f"{language} programming"
def find_resources(language: str, topic: str) -> dict:
"""
Finds learning resources for the CORE TOPIC extracted from the code
(e.g. "Dijkstra's Algorithm"), not just the language name. A dedicated
video-biased query runs first — a walkthrough video of a named
algorithm/concept is usually the most useful resource — then a broader
query fills in with articles/docs, skipping anything already found.
"""
video_query = f"{topic} {language} explained tutorial site:youtube.com"
article_query = f"{topic} {language} tutorial {RESOURCE_QUERY_HINT}"
video_citations = _structured_citations(video_query, max_results=3)
seen_urls = {c["url"] for c in video_citations}
article_citations = [
c for c in _structured_citations(article_query, max_results=4)
if c["url"] not in seen_urls
]
citations = video_citations + article_citations
return {
"query": video_query,
"topic": topic,
"citations": citations,
"ok": bool(citations),
}
def generate_breakdown(code: str, language: str, level: str) -> dict:
system_prompt = LEVEL_PROMPTS[level] + "\n\n" + JSON_OUTPUT_INSTRUCTIONS
user_content = f"Language: {language}\n\nCode:\n{code}"
return call_groq_json(system_prompt, user_content)
def generate_plan(code: str, language: str) -> dict:
user_content = f"Language: {language}\n\nCode:\n{code}"
return call_groq_json(PLAN_PROMPT, user_content)
def cleanup_code(code: str, language: str) -> dict:
user_content = f"Language: {language}\n\nCode:\n{code}"
return call_groq_json(CLEANUP_PROMPT, user_content)
def suggest_optimal_solution(code: str, language: str) -> dict:
user_content = f"Language: {language}\n\nCode:\n{code}"
return call_groq_json(OPTIMIZATION_PROMPT, user_content)