parser.py created
This commit is contained in:
@@ -0,0 +1,108 @@
|
||||
"""
|
||||
Raw text → flat card parser.
|
||||
|
||||
This module demonstrates how to convert a free‑form assignment description into a structured
|
||||
Pydantic model using LangChain’s PromptTemplate, LLM and PydanticOutputParser.
|
||||
|
||||
The public function ``parse_assignment(text: str) -> AssignmentCard`` returns an instance of
|
||||
the :class:`AssignmentCard` dataclass. The implementation is intentionally minimal but fully
|
||||
type‑checked and ready for unit testing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Dict
|
||||
|
||||
# LangChain imports – the core library provides PromptTemplate and LLM wrappers
|
||||
from langchain_core.prompts import PromptTemplate
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain.output_parsers import PydanticOutputParser
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Define the output schema with Pydantic
|
||||
# ---------------------------------------------------------------------------
|
||||
class AssignmentCard(BaseModel):
|
||||
"""Structured representation of an assignment description.
|
||||
|
||||
Attributes
|
||||
----------
|
||||
title : str
|
||||
Short title of the task.
|
||||
subject : str
|
||||
Subject or topic covered by the assignment.
|
||||
deadline_hint : str | None
|
||||
Human‑readable hint about the due date (e.g. "by Friday").
|
||||
deliverable_type : str
|
||||
What should be submitted – e.g. "report", "code".
|
||||
grading_hints : List[str]
|
||||
Optional list of hints that influence grading.
|
||||
"""
|
||||
|
||||
title: str = Field(..., description="Short title of the task")
|
||||
subject: str = Field(..., description="Subject or topic covered by the assignment")
|
||||
deadline_hint: str | None = Field(None, description="Human‑readable hint about due date")
|
||||
deliverable_type: str = Field(..., description="What should be submitted – e.g. report, code")
|
||||
grading_hints: List[str] = Field(default_factory=list, description="Hints that influence grading")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Prompt template – instruct the LLM to output JSON matching the schema
|
||||
# ---------------------------------------------------------------------------
|
||||
PROMPT_TEMPLATE = (
|
||||
"You are an assistant that extracts structured information from a free‑form assignment description.
|
||||
Return a JSON object with the following fields exactly as defined in the AssignmentCard model:
|
||||
{{schema}}
|
||||
The input text is: "{{text}}"
|
||||
""")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Parser that validates the LLM output against the Pydantic schema
|
||||
# ---------------------------------------------------------------------------
|
||||
parser = PydanticOutputParser(pydantic_object=AssignmentCard)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. The main function – orchestrates prompt → LLM → parser
|
||||
# ---------------------------------------------------------------------------
|
||||
def parse_assignment(text: str, *, llm_model: str = "gpt-3.5-turbo") -> AssignmentCard:
|
||||
"""Parse a raw assignment description into an :class:`AssignmentCard`.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
text : str
|
||||
Free‑form assignment description.
|
||||
llm_model : str, optional
|
||||
Name of the OpenAI model to use. Defaults to ``gpt-3.5-turbo``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
AssignmentCard
|
||||
Validated dataclass instance.
|
||||
"""
|
||||
# Build prompt with schema description
|
||||
template = PromptTemplate(
|
||||
input_variables=["text", "schema"],
|
||||
template=PROMPT_TEMPLATE,
|
||||
)
|
||||
prompt = template.format(text=text, schema=parser.get_format_instructions())
|
||||
|
||||
# Call the LLM – we use ChatOpenAI from langchain_openai for simplicity
|
||||
llm = ChatOpenAI(model_name=llm_model, temperature=0)
|
||||
raw_output = llm.invoke(prompt).content
|
||||
|
||||
# Parse and validate
|
||||
return parser.parse(raw_output)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. Demo – run when executed as a script
|
||||
# ---------------------------------------------------------------------------
|
||||
if __name__ == "__main__":
|
||||
import os
|
||||
if not os.getenv("OPENAI_API_KEY"):
|
||||
raise RuntimeError("Set OPENAI_API_KEY environment variable.")
|
||||
|
||||
sample = (
|
||||
"Сдайте к пятнице мини‑отчёт по LangChain. В отчёте должно быть описание модели, пример кода и выводы."
|
||||
)
|
||||
card = parse_assignment(sample)
|
||||
print("Parsed assignment:", card.json(indent=2))
|
||||
Reference in New Issue
Block a user