commit 08d9578cb47bb40ef8909fc5dad44b18352d3ce6 Author: Danil Parunin 5f1b81b8-4f5d-11e8-9c2d-fa7ae01bbebc Date: Mon Jun 15 12:08:49 2026 +0000 add: main.py diff --git a/main.py b/main.py new file mode 100644 index 0000000..b1798a2 --- /dev/null +++ b/main.py @@ -0,0 +1,203 @@ +"""Code Review Agent with LangGraph and DeepAgents + +This script implements the assignment described in the prompt. It uses +* LangGraph to model the review cycle (draft → reflect → rewrite → reflect …) +* DeepAgents to expose the whole workflow as a single LLM‑driven agent. +* OpenRouter via langchain‑openai for all LLM calls. + +Run the demo with: + +```bash +python main.py +``` + +The demo reviews a simple `sort_numbers` function and prints the draft review, +the critic’s scores, and any rewritten sections. +""" + +import os +import asyncio +from typing import TypedDict, Annotated, Dict + +from langchain_openai import ChatOpenAI +from langchain_core.messages import HumanMessage +from langchain.tools import tool +from langchain.output_parsers import PydanticOutputParser +from pydantic import BaseModel, Field + +from langgraph.graph import StateGraph, START, END +from langgraph.graph.message import add_messages + +from deepagents import create_deep_agent +from deepagents.backends import FilesystemBackend, LocalShellBackend, CompositeBackend + +# --------------------------------------------------------------------------- +# 1. LLM configuration (OpenRouter) +# --------------------------------------------------------------------------- +llm = ChatOpenAI( + model="openai/gpt-oss-20b:free", + base_url="https://openrouter.ai/api/v1", + api_key=os.getenv("OPENAI_API_KEY"), + temperature=0.0, +) + +# --------------------------------------------------------------------------- +# 2. State definition +# --------------------------------------------------------------------------- +class CodeReviewState(TypedDict): + code: str + draft_review: str + criteria_scores: Dict[str, int] # {"pep8": 0-10, "type_hints": ..., "edge_cases": ..., "naming": ...} + weakest_criterion: str + verdict: str # "ok" | "needs_revision" + round: int + max_rounds: int + +# --------------------------------------------------------------------------- +# 3. Structured output for the critic (reflect node) +# --------------------------------------------------------------------------- +class CriticOutput(BaseModel): + pep8: int = Field(..., ge=0, le=10, description="Score for PEP8 compliance") + type_hints: int = Field(..., ge=0, le=10, description="Score for type hints usage") + edge_cases: int = Field(..., ge=0, le=10, description="Score for handling edge cases") + naming: int = Field(..., ge=0, le=10, description="Score for naming conventions") + weakest_criterion: str = Field(..., description="Criterion with the lowest score") + verdict: str = Field(..., description="'ok' or 'needs_revision'") + +critic_parser = PydanticOutputParser(pydantic_object=CriticOutput) + +# --------------------------------------------------------------------------- +# 4. Graph nodes +# --------------------------------------------------------------------------- +async def draft_review(state: CodeReviewState) -> CodeReviewState: + prompt = ( + "You are a senior Python reviewer.\n" + "Given the following function, write a concise code review (3–6 points).\n" + "Focus on style, correctness, and best practices.\n" + "Return the review as plain text.\n" + f"Function:\n{state['code']}" + ) + review = await llm.ainvoke([HumanMessage(content=prompt)]) + state['draft_review'] = review.content.strip() + return state + +async def reflect(state: CodeReviewState) -> CodeReviewState: + prompt = ( + "You are a code quality critic.\n" + "Evaluate the following review against four criteria: PEP8, type hints, edge cases, naming.\n" + "Assign a score 0–10 for each criterion.\n" + "Identify the weakest criterion and decide if the review is "ok" or "needs_revision".\n" + "Return the results in JSON matching the following schema:\n" + f"{critic_parser.get_format_instructions()}\n" + f"Review:\n{state['draft_review']}" + ) + result = await llm.ainvoke([HumanMessage(content=prompt)]) + parsed = critic_parser.parse(result.content) + state['criteria_scores'] = { + "pep8": parsed.pep8, + "type_hints": parsed.type_hints, + "edge_cases": parsed.edge_cases, + "naming": parsed.naming, + } + state['weakest_criterion'] = parsed.weakest_criterion + state['verdict'] = parsed.verdict + return state + +async def rewrite(state: CodeReviewState) -> CodeReviewState: + # Rewrite only the section of the review that addresses the weakest criterion + prompt = ( + "You are a code reviewer.\n" + "The current review is: \n" + f"{state['draft_review']}\n" + "The weakest criterion is: " + state['weakest_criterion'] + ".\n" + "Rewrite the review to strengthen this part, keeping the overall tone.\n" + "Return only the updated review text." + ) + updated = await llm.ainvoke([HumanMessage(content=prompt)]) + state['draft_review'] = updated.content.strip() + state['round'] += 1 + return state + +# --------------------------------------------------------------------------- +# 5. Build the LangGraph +# --------------------------------------------------------------------------- +def build_graph() -> StateGraph[CodeReviewState]: + graph = StateGraph(CodeReviewState) + graph.add_node("draft_review", draft_review) + graph.add_node("reflect", reflect) + graph.add_node("rewrite", rewrite) + + graph.set_entry_point("draft_review") + graph.add_edge("draft_review", "reflect") + graph.add_conditional_edges( + "reflect", + lambda state: "rewrite" if state["verdict"] == "needs_revision" and state["round"] < state["max_rounds"] else "END", + ) + graph.add_edge("rewrite", "reflect") + return graph + +# --------------------------------------------------------------------------- +# 6. Tool that runs the graph +# --------------------------------------------------------------------------- +@tool +def run_review(code: str) -> str: + """Run the full review cycle on the provided Python code.""" + # Initial state + state: CodeReviewState = { + "code": code, + "draft_review": "", + "criteria_scores": {}, + "weakest_criterion": "", + "verdict": "", + "round": 1, + "max_rounds": 2, + } + graph = build_graph() + # Execute graph synchronously + final_state = graph.invoke(state) + # Prepare a readable output + output = [ + "--- Draft Review ---", + final_state["draft_review"], + "\n--- Critic Scores ---", + f"PEP8: {final_state['criteria_scores'].get('pep8', 'N/A')}\n" + f"Type Hints: {final_state['criteria_scores'].get('type_hints', 'N/A')}\n" + f"Edge Cases: {final_state['criteria_scores'].get('edge_cases', 'N/A')}\n" + f"Naming: {final_state['criteria_scores'].get('naming', 'N/A')}\n", + f"Verdict: {final_state['verdict']} (round {final_state['round']})", + ] + return "\n".join(output) + +# --------------------------------------------------------------------------- +# 7. DeepAgents setup +# --------------------------------------------------------------------------- +backend = CompositeBackend([ + LocalShellBackend(workspace_dir="./workspace"), + FilesystemBackend(), +]) + +agent = create_deep_agent( + model=llm, + tools=[run_review], + backend=backend, + system_prompt="You are a helpful code review assistant.", +) + +# --------------------------------------------------------------------------- +# 8. Demo CLI +# --------------------------------------------------------------------------- +async def main(): + example_code = """ + def sort_numbers(arr): + return sorted(arr) + """ + result = await agent.ainvoke( + {"messages": [HumanMessage(content="Please review the following function: + +"" + example_code + "")]}, + {"configurable": {"thread_id": "demo-session"}}, + ) + print(result["messages"][-1].content) + +if __name__ == "__main__": + asyncio.run(main())