diff --git a/main.py b/main.py index b1798a2..f91278b 100644 --- a/main.py +++ b/main.py @@ -1,20 +1,3 @@ -"""Code Review Agent with LangGraph and DeepAgents - -This script implements the assignment described in the prompt. It uses -* LangGraph to model the review cycle (draft → reflect → rewrite → reflect …) -* DeepAgents to expose the whole workflow as a single LLM‑driven agent. -* OpenRouter via langchain‑openai for all LLM calls. - -Run the demo with: - -```bash -python main.py -``` - -The demo reviews a simple `sort_numbers` function and prints the draft review, -the critic’s scores, and any rewritten sections. -""" - import os import asyncio from typing import TypedDict, Annotated, Dict @@ -22,18 +5,14 @@ from typing import TypedDict, Annotated, Dict from langchain_openai import ChatOpenAI from langchain_core.messages import HumanMessage from langchain.tools import tool -from langchain.output_parsers import PydanticOutputParser -from pydantic import BaseModel, Field - -from langgraph.graph import StateGraph, START, END -from langgraph.graph.message import add_messages - from deepagents import create_deep_agent from deepagents.backends import FilesystemBackend, LocalShellBackend, CompositeBackend +from langgraph.graph import StateGraph, START, END +from langgraph.graph.message import add_messages +from pydantic import BaseModel, Field +from langchain_core.output_parsers import PydanticOutputParser -# --------------------------------------------------------------------------- -# 1. LLM configuration (OpenRouter) -# --------------------------------------------------------------------------- +# ---------- LLM ---------- llm = ChatOpenAI( model="openai/gpt-oss-20b:free", base_url="https://openrouter.ai/api/v1", @@ -41,163 +20,163 @@ llm = ChatOpenAI( temperature=0.0, ) -# --------------------------------------------------------------------------- -# 2. State definition -# --------------------------------------------------------------------------- -class CodeReviewState(TypedDict): - code: str - draft_review: str - criteria_scores: Dict[str, int] # {"pep8": 0-10, "type_hints": ..., "edge_cases": ..., "naming": ...} - weakest_criterion: str - verdict: str # "ok" | "needs_revision" - round: int - max_rounds: int - -# --------------------------------------------------------------------------- -# 3. Structured output for the critic (reflect node) -# --------------------------------------------------------------------------- -class CriticOutput(BaseModel): - pep8: int = Field(..., ge=0, le=10, description="Score for PEP8 compliance") - type_hints: int = Field(..., ge=0, le=10, description="Score for type hints usage") - edge_cases: int = Field(..., ge=0, le=10, description="Score for handling edge cases") - naming: int = Field(..., ge=0, le=10, description="Score for naming conventions") - weakest_criterion: str = Field(..., description="Criterion with the lowest score") - verdict: str = Field(..., description="'ok' or 'needs_revision'") - -critic_parser = PydanticOutputParser(pydantic_object=CriticOutput) - -# --------------------------------------------------------------------------- -# 4. Graph nodes -# --------------------------------------------------------------------------- -async def draft_review(state: CodeReviewState) -> CodeReviewState: - prompt = ( - "You are a senior Python reviewer.\n" - "Given the following function, write a concise code review (3–6 points).\n" - "Focus on style, correctness, and best practices.\n" - "Return the review as plain text.\n" - f"Function:\n{state['code']}" - ) - review = await llm.ainvoke([HumanMessage(content=prompt)]) - state['draft_review'] = review.content.strip() - return state - -async def reflect(state: CodeReviewState) -> CodeReviewState: - prompt = ( - "You are a code quality critic.\n" - "Evaluate the following review against four criteria: PEP8, type hints, edge cases, naming.\n" - "Assign a score 0–10 for each criterion.\n" - "Identify the weakest criterion and decide if the review is "ok" or "needs_revision".\n" - "Return the results in JSON matching the following schema:\n" - f"{critic_parser.get_format_instructions()}\n" - f"Review:\n{state['draft_review']}" - ) - result = await llm.ainvoke([HumanMessage(content=prompt)]) - parsed = critic_parser.parse(result.content) - state['criteria_scores'] = { - "pep8": parsed.pep8, - "type_hints": parsed.type_hints, - "edge_cases": parsed.edge_cases, - "naming": parsed.naming, - } - state['weakest_criterion'] = parsed.weakest_criterion - state['verdict'] = parsed.verdict - return state - -async def rewrite(state: CodeReviewState) -> CodeReviewState: - # Rewrite only the section of the review that addresses the weakest criterion - prompt = ( - "You are a code reviewer.\n" - "The current review is: \n" - f"{state['draft_review']}\n" - "The weakest criterion is: " + state['weakest_criterion'] + ".\n" - "Rewrite the review to strengthen this part, keeping the overall tone.\n" - "Return only the updated review text." - ) - updated = await llm.ainvoke([HumanMessage(content=prompt)]) - state['draft_review'] = updated.content.strip() - state['round'] += 1 - return state - -# --------------------------------------------------------------------------- -# 5. Build the LangGraph -# --------------------------------------------------------------------------- -def build_graph() -> StateGraph[CodeReviewState]: - graph = StateGraph(CodeReviewState) - graph.add_node("draft_review", draft_review) - graph.add_node("reflect", reflect) - graph.add_node("rewrite", rewrite) - - graph.set_entry_point("draft_review") - graph.add_edge("draft_review", "reflect") - graph.add_conditional_edges( - "reflect", - lambda state: "rewrite" if state["verdict"] == "needs_revision" and state["round"] < state["max_rounds"] else "END", - ) - graph.add_edge("rewrite", "reflect") - return graph - -# --------------------------------------------------------------------------- -# 6. Tool that runs the graph -# --------------------------------------------------------------------------- -@tool -def run_review(code: str) -> str: - """Run the full review cycle on the provided Python code.""" - # Initial state - state: CodeReviewState = { - "code": code, - "draft_review": "", - "criteria_scores": {}, - "weakest_criterion": "", - "verdict": "", - "round": 1, - "max_rounds": 2, - } - graph = build_graph() - # Execute graph synchronously - final_state = graph.invoke(state) - # Prepare a readable output - output = [ - "--- Draft Review ---", - final_state["draft_review"], - "\n--- Critic Scores ---", - f"PEP8: {final_state['criteria_scores'].get('pep8', 'N/A')}\n" - f"Type Hints: {final_state['criteria_scores'].get('type_hints', 'N/A')}\n" - f"Edge Cases: {final_state['criteria_scores'].get('edge_cases', 'N/A')}\n" - f"Naming: {final_state['criteria_scores'].get('naming', 'N/A')}\n", - f"Verdict: {final_state['verdict']} (round {final_state['round']})", - ] - return "\n".join(output) - -# --------------------------------------------------------------------------- -# 7. DeepAgents setup -# --------------------------------------------------------------------------- +# ---------- Backend ---------- backend = CompositeBackend([ LocalShellBackend(workspace_dir="./workspace"), FilesystemBackend(), ]) +# ---------- State ---------- +class CodeReviewState(TypedDict): + code: str + draft_review: str + criteria_scores: Dict[str, int] + weakest_criterion: str + verdict: str + round: int + max_rounds: int + +# ---------- Pydantic models for structured output ---------- +class ReviewScores(BaseModel): + pep8: int = Field(..., ge=0, le=10) + type_hints: int = Field(..., ge=0, le=10) + edge_cases: int = Field(..., ge=0, le=10) + naming: int = Field(..., ge=0, le=10) + weakest_criterion: str + verdict: str + +class ReviewRewrite(BaseModel): + draft_review: str + criteria_scores: Dict[str, int] + weakest_criterion: str + verdict: str + round: int + max_rounds: int + +# ---------- Output parsers ---------- +review_parser = PydanticOutputParser(pydantic_object=ReviewScores) +rewrite_parser = PydanticOutputParser(pydantic_object=ReviewRewrite) + +# ---------- Nodes ---------- + +def draft_review_node(state: CodeReviewState) -> CodeReviewState: + code = state["code"] + prompt = f""" +You are a senior Python reviewer. Provide a concise code review for the following function. Output exactly 3-6 bullet points, each starting with a dash. Do not include any additional text. + +{code} +""" + response = llm.invoke([HumanMessage(content=prompt)]) + state["draft_review"] = response.content.strip() + return state + +# DESIGN DECISION: reflect node returns structured JSON with scores and verdict +# NECESSITY: required by assignment to have structured output for automated parsing +# OPTIMALITY: eliminates ambiguity and parsing errors compared to free text +# ALTERNATIVES CONSIDERED: free text parsing, regex extraction – rejected due to unreliability + +def reflect_node(state: CodeReviewState) -> CodeReviewState: + prompt = f""" +You are an automated code quality critic. Evaluate the following draft review against these criteria: +- PEP8 compliance +- Presence of type hints +- Handling of edge cases +- Naming conventions + +Return a JSON object with integer scores 0-10 for each criterion, the name of the weakest criterion, and a verdict "ok" or "needs_revision". + +Draft review: +{state["draft_review"]} +""" + response = llm.invoke([HumanMessage(content=prompt)]) + parsed = review_parser.parse(response.content) + state["criteria_scores"] = { + "pep8": parsed.pep8, + "type_hints": parsed.type_hints, + "edge_cases": parsed.edge_cases, + "naming": parsed.naming, + } + state["weakest_criterion"] = parsed.weakest_criterion + state["verdict"] = parsed.verdict + return state + +# DESIGN DECISION: rewrite node focuses only on weakest criterion +# NECESSITY: assignment specifies targeted rewrite +# OPTIMALITY: keeps changes minimal and focused, avoids over‑engineering +# ALTERNATIVES CONSIDERED: full rewrite of review – rejected for unnecessary complexity + +def rewrite_node(state: CodeReviewState) -> CodeReviewState: + prompt = f""" +You are a code reviewer. The previous draft review was: +{state["draft_review"]} + +The weakest criterion is {state["weakest_criterion"]}. Rewrite only the part of the review that addresses this criterion, improving it. Keep the rest of the review unchanged. Output the updated draft review and updated scores (same format as in reflect). Also increment the round counter. +""" + response = llm.invoke([HumanMessage(content=prompt)]) + parsed = rewrite_parser.parse(response.content) + state["draft_review"] = parsed.draft_review + state["criteria_scores"] = parsed.criteria_scores + state["weakest_criterion"] = parsed.weakest_criterion + state["verdict"] = parsed.verdict + state["round"] = parsed.round + state["max_rounds"] = parsed.max_rounds + return state + +# ---------- Graph ---------- + +graph = StateGraph(CodeReviewState) + +graph.add_node("draft_review", draft_review_node) +graph.add_node("reflect", reflect_node) +graph.add_node("rewrite", rewrite_node) + +# Entry point +graph.add_edge(START, "draft_review") +graph.add_edge("draft_review", "reflect") +# Conditional edges after reflect + +def decide_next(state: CodeReviewState): + if state["verdict"] == "ok": + return END + if state["round"] < state["max_rounds"]: + return "rewrite" + return END + +graph.add_conditional_edges("reflect", decide_next, {"rewrite": "rewrite", END: END}) +# After rewrite go back to reflect +graph.add_edge("rewrite", "reflect") + +app = graph.compile() + +# ---------- DeepAgent wrapper ---------- agent = create_deep_agent( model=llm, - tools=[run_review], + tools=[], backend=backend, - system_prompt="You are a helpful code review assistant.", + system_prompt="You are a code review assistant.", ) -# --------------------------------------------------------------------------- -# 8. Demo CLI -# --------------------------------------------------------------------------- +# ---------- CLI Demo ---------- async def main(): - example_code = """ + # Example function to review + code = """ def sort_numbers(arr): return sorted(arr) """ - result = await agent.ainvoke( - {"messages": [HumanMessage(content="Please review the following function: - -"" + example_code + "")]}, - {"configurable": {"thread_id": "demo-session"}}, - ) - print(result["messages"][-1].content) + initial_state: CodeReviewState = { + "code": code.strip(), + "draft_review": "", # will be filled + "criteria_scores": {}, + "weakest_criterion": "", + "verdict": "", + "round": 0, + "max_rounds": 2, + } + result = await app.ainvoke(initial_state) + print("\n--- Final Review ---") + print(result["draft_review"]) + print("\nScores:", result["criteria_scores"]) + print("Verdict:", result["verdict"]) if __name__ == "__main__": asyncio.run(main())