LLM Integration Patterns
Established architectural patterns and best practices for integrating Large Language Models into production applications, with emphasis on reliability, structured output, and robust error handling. Demonstrated in practice through the lesphinx project implementation and various production systems.
Structured Output Management
Pydantic Schema Validation
Use strongly-typed schemas to ensure LLM outputs conform to expected structures:
from pydantic import BaseModel, Field
from typing import Literal, Optional
class SphinxAction(BaseModel):
action_type: Literal["question", "guess"] = Field(
description="Whether to ask a question or make a guess"
)
content: str = Field(
min_length=1,
max_length=500,
description="The question to ask or character to guess"
)
confidence: float = Field(
ge=0.0,
le=1.0,
description="Confidence level from 0.0 to 1.0"
)
reasoning: Optional[str] = Field(
default=None,
description="Internal reasoning (not shown to user)"
)
class LLMClient:
async def get_sphinx_action(self, conversation_history: list) -> SphinxAction:
response = await self.client.chat(
model="mistral-large-latest",
messages=conversation_history,
response_format={"type": "json_object"}
)
return SphinxAction.model_validate_json(response.content)
Conversation History Management
Maintain context while preventing token bloat:
class ConversationManager:
def __init__(self, max_messages: int = 20):
self.max_messages = max_messages
self.system_prompt = "You are the Sphinx, master of riddles..."
def build_conversation_messages(self, session_data: dict) -> list:
"""Build conversation with proper JSON serialization."""
messages = [{"role": "system", "content": self.system_prompt}]
# Add conversation history (avoiding JSON injection)
for turn in session_data.get("turns", []):
messages.append({
"role": "user",
"content": turn["user_answer"]
})
# Safe JSON serialization instead of f-strings
assistant_content = json.dumps({
"action_type": turn["sphinx_action_type"],
"content": turn["sphinx_content"],
"confidence": turn["sphinx_confidence"]
})
messages.append({
"role": "assistant",
"content": assistant_content
})
# Trim to max length, preserving system message
if len(messages) > self.max_messages:
keep_count = self.max_messages - 1 # Reserve space for system
messages = messages[:1] + messages[-keep_count:]
return messages
Reliability Patterns
Exponential Backoff Retry Logic
Handle temporary API failures gracefully:
import asyncio
import logging
from typing import Optional
class LLMClient:
def __init__(self, max_retries: int = 3, base_delay: float = 0.5):
self.max_retries = max_retries
self.base_delay = base_delay
self.logger = logging.getLogger(__name__)
async def chat_with_retry(self, **kwargs) -> Optional[dict]:
"""Execute LLM call with exponential backoff retry."""
last_exception = None
for attempt in range(self.max_retries):
try:
self.logger.info(f"LLM attempt {attempt + 1}/{self.max_retries}")
response = await self.client.chat(**kwargs)
self.logger.info(f"LLM success on attempt {attempt + 1}")
return response
except Exception as e:
last_exception = e
self.logger.warning(
f"LLM attempt {attempt + 1} failed: {str(e)[:100]}"
)
if attempt < self.max_retries - 1:
# Exponential backoff: 0.5s, 1.0s, 2.0s
delay = self.base_delay * (2 ** attempt)
await asyncio.sleep(delay)
self.logger.error(f"All LLM attempts failed. Last error: {last_exception}")
return None
Fallback System Implementation
Provide graceful degradation when LLM fails:
class GameEngine:
def __init__(self):
self.fallback_questions = [
"Tell me more about this character.",
"What else can you share about them?",
"Any other details that might help?",
"Can you describe their appearance?",
"What about their personality?"
]
self.fallback_index = 0
def fallback_question(self, language: str = "en") -> str:
"""Provide fallback question when LLM fails."""
questions = (
self.fallback_questions if language == "en"
else self.fallback_questions_fr
)
question = questions[self.fallback_index % len(questions)]
self.fallback_index += 1
return question
async def process_turn(self, session, user_answer: str):
# Try LLM first
llm_response = await self.llm_client.get_sphinx_action(
self.conversation_manager.build_messages(session)
)
if llm_response:
return self.create_llm_turn(session, llm_response, user_answer)
else:
# Fallback to predefined question
return self.create_fallback_turn(session, user_answer)
State Management Patterns
Deterministic Flow Control
Ensure critical game logic doesn't depend on LLM decisions:
class GameEngine:
def process_guess_confirmation(self, session: dict, correct: bool) -> Optional['Turn']:
"""Handle guess confirmations deterministically."""
if correct:
# Immediate victory - don't consult LLM
session["state"] = GameState.ENDED
session["result"] = "win"
victory_text = (
"🎉 Magnificent! I have successfully guessed your character!"
if session["language"] == "en" else
"🎉 Magnifique ! J'ai deviné votre personnage !"
)
return Turn(
sphinx_action_type="victory",
sphinx_content=victory_text,
sphinx_confidence=1.0,
user_answer="yes"
)
else:
# Wrong guess - continue or end based on remaining guesses
session["guesses_made"] += 1
if session["guesses_made"] >= self.MAX_GUESSES:
session["state"] = GameState.ENDED
session["result"] = "lose"
return self.create_defeat_turn(session)
else:
# Continue game - let route call LLM for next question
return None # Signal to route to continue with LLM
Context-Aware Processing
Adapt LLM behavior based on game state:
class ContextAwareLLM:
def build_system_prompt(self, session_data: dict) -> str:
base_prompt = "You are the Sphinx, master of riddles and ancient wisdom."
# Adapt based on game progress
turn_count = len(session_data.get("turns", []))
guesses_made = session_data.get("guesses_made", 0)
if turn_count < 3:
context = "Focus on broad, general questions to narrow down possibilities."
elif turn_count < 8:
context = "Ask more specific questions based on what you've learned."
elif guesses_made == 0:
context = "You should start making educated guesses now."
else:
context = "Be more careful with your remaining guesses."
return f"{base_prompt}\n\nCurrent strategy: {context}"
Error Handling and Monitoring
Structured Error Handling
Provide clear error boundaries and recovery:
class LLMIntegrationError(Exception):
"""Base exception for LLM integration issues."""
pass
class LLMTimeoutError(LLMIntegrationError):
"""LLM request timed out."""
pass
class LLMValidationError(LLMIntegrationError):
"""LLM output failed validation."""
pass
class SafeLLMClient:
async def safe_chat(self, **kwargs) -> tuple[Optional[dict], Optional[str]]:
"""Return (response, error_message) tuple."""
try:
response = await self.chat_with_retry(**kwargs)
if not response:
return None, "Service temporarily unavailable"
# Validate response structure
validated = SphinxAction.model_validate_json(response.content)
return validated.model_dump(), None
except ValidationError as e:
self.logger.error(f"LLM output validation failed: {e}")
return None, "Response format error"
except asyncio.TimeoutError:
self.logger.error("LLM request timed out")
return None, "Request timed out"
except Exception as e:
self.logger.error(f"Unexpected LLM error: {e}")
return None, "Service error"
Performance Monitoring
Track LLM integration health:
class LLMMetrics:
def __init__(self):
self.success_count = 0
self.failure_count = 0
self.total_tokens = 0
self.response_times = []
def record_success(self, tokens_used: int, response_time: float):
self.success_count += 1
self.total_tokens += tokens_used
self.response_times.append(response_time)
def record_failure(self, error_type: str):
self.failure_count += 1
self.logger.warning(f"LLM failure: {error_type}")
@property
def success_rate(self) -> float:
total = self.success_count + self.failure_count
return self.success_count / total if total > 0 else 0.0
@property
def avg_response_time(self) -> float:
return sum(self.response_times) / len(self.response_times) if self.response_times else 0.0
Security Considerations
JSON Injection Prevention
Never construct JSON with string interpolation:
# ❌ DANGEROUS - JSON injection vulnerability
def build_messages_unsafe(turns):
messages = []
for turn in turns:
content = f'{{"action": "{turn.action}", "text": "{turn.text}"}}'
messages.append({"role": "assistant", "content": content})
return messages
# ✅ SAFE - Use json.dumps for serialization
def build_messages_safe(turns):
messages = []
for turn in turns:
content = json.dumps({
"action": turn.action,
"text": turn.text
})
messages.append({"role": "assistant", "content": content})
return messages
Input Sanitization
Validate and sanitize all inputs before sending to LLM:
class InputSanitizer:
def __init__(self, max_length: int = 1000):
self