fix: connect Peg to local AI runtime
MetaBloom CI / app (push) Canceled after 0s
MetaBloom CI / api (push) Canceled after 0s

This commit is contained in:
AI Company
2026-07-21 22:06:38 +00:00
parent 02e24cbc92
commit 0c2430ad13
7 changed files with 50 additions and 8 deletions
+4 -3
View File
@@ -4,7 +4,8 @@ ENVIRONMENT=production
ALLOWED_ORIGINS=https://app.example.com ALLOWED_ORIGINS=https://app.example.com
POSTGRES_PASSWORD=change-me POSTGRES_PASSWORD=change-me
# API OpenAI-compatible : Ollama local par défaut, ou fournisseur autorisé. # API OpenAI-compatible : Ollama local par défaut, ou fournisseur autorisé.
AI_BASE_URL=http://localhost:11434/v1 # Docker sur le réseau ai-company : utiliser le nom DNS du service, pas localhost.
AI_MODEL=llama3.2:3b AI_BASE_URL=http://ollama:11434/v1
AI_MODEL=qwen2.5:1.5b
AI_API_KEY= AI_API_KEY=
AI_TIMEOUT_SECONDS=20 AI_TIMEOUT_SECONDS=90
+3
View File
@@ -33,6 +33,7 @@ Le port n’écoute que sur localhost ; placer un reverse proxy TLS devant en pr
- `GET /account/export`, `DELETE /account` - `GET /account/export`, `DELETE /account`
- `POST /coach/chat` — conversation authentifiée avec Peg - `POST /coach/chat` — conversation authentifiée avec Peg
- `GET /health` - `GET /health`
- `GET /health/ai` — joignabilité du moteur IA et nom du modèle, sans exposer de secret
Envoyer `Authorization: Bearer <token>`. La création dentrées exige `consent_health_data=true`. Les limites de saisie préviennent les erreurs grossières mais ne constituent pas une validation médicale. Envoyer `Authorization: Bearer <token>`. La création dentrées exige `consent_health_data=true`. Les limites de saisie préviennent les erreurs grossières mais ne constituent pas une validation médicale.
@@ -42,6 +43,8 @@ Envoyer `Authorization: Bearer <token>`. La création dentrées exige `consen
Peg utilise une API compatible OpenAI (`/v1/chat/completions`). La configuration par défaut vise Ollama local et ne nécessite aucune clé. Définir `AI_BASE_URL`, `AI_MODEL` et, si nécessaire, `AI_API_KEY`. Aucun secret n'est commité. Un refus déterministe couvre diagnostic, prescription et dosage ; les urgences renvoient vers le 15/112. Si le modèle est indisponible ou produit une réponse suspecte, lAPI renvoie un conseil général sûr avec `source=fallback`. Peg utilise une API compatible OpenAI (`/v1/chat/completions`). La configuration par défaut vise Ollama local et ne nécessite aucune clé. Définir `AI_BASE_URL`, `AI_MODEL` et, si nécessaire, `AI_API_KEY`. Aucun secret n'est commité. Un refus déterministe couvre diagnostic, prescription et dosage ; les urgences renvoient vers le 15/112. Si le modèle est indisponible ou produit une réponse suspecte, lAPI renvoie un conseil général sûr avec `source=fallback`.
En Docker, `localhost` désigne le conteneur API et non Ollama. Le Compose dexemple rattache donc lAPI au réseau externe `ai-company` et utilise `http://ollama:11434/v1`. Le service Ollama doit porter le nom DNS `ollama`, le modèle configuré doit être téléchargé, et `GET /health/ai` doit retourner `reachable: true`.
Exemple de corps : `{"message":"Comment reprendre doucement la marche ?","history":[]}`. Lhistorique est limité à 12 messages et nest pas persisté par cet endpoint. Exemple de corps : `{"message":"Comment reprendre doucement la marche ?","history":[]}`. Lhistorique est limité à 12 messages et nest pas persisté par cet endpoint.
## Tests ## Tests
+15 -1
View File
@@ -48,7 +48,8 @@ async def chat_with_peg(data: CoachChatIn, cfg: Settings) -> CoachChatOut:
response = await client.post( response = await client.post(
f"{cfg.ai_base_url.rstrip('/')}/chat/completions", f"{cfg.ai_base_url.rstrip('/')}/chat/completions",
headers=headers, headers=headers,
json={"model": cfg.ai_model, "messages": messages, "temperature": 0.4, "max_tokens": 350}, # Keep answers useful but short so a local CPU model responds reliably.
json={"model": cfg.ai_model, "messages": messages, "temperature": 0.4, "max_tokens": 160},
) )
response.raise_for_status() response.raise_for_status()
content = response.json()["choices"][0]["message"]["content"].strip() content = response.json()["choices"][0]["message"]["content"].strip()
@@ -60,3 +61,16 @@ async def chat_with_peg(data: CoachChatIn, cfg: Settings) -> CoachChatOut:
return CoachChatOut(message=content, source="model") return CoachChatOut(message=content, source="model")
except (httpx.HTTPError, KeyError, IndexError, TypeError, ValueError): except (httpx.HTTPError, KeyError, IndexError, TypeError, ValueError):
return CoachChatOut(message="Je suis momentanément indisponible. En attendant, choisis une petite action sans risque : boire un verre deau, marcher quelques minutes si tu te sens bien, ou noter ton prochain repas. Pour une question médicale, contacte un professionnel de santé.", source="fallback") return CoachChatOut(message="Je suis momentanément indisponible. En attendant, choisis une petite action sans risque : boire un verre deau, marcher quelques minutes si tu te sens bien, ou noter ton prochain repas. Pour une question médicale, contacte un professionnel de santé.", source="fallback")
async def peg_provider_health(cfg: Settings) -> dict[str, str | bool]:
"""Check the configured AI provider without leaking credentials or prompts."""
base_url = cfg.ai_base_url.rstrip("/")
headers = {"Authorization": f"Bearer {cfg.ai_api_key}"} if cfg.ai_api_key else {}
try:
async with httpx.AsyncClient(timeout=min(cfg.ai_timeout_seconds, 5.0)) as client:
response = await client.get(f"{base_url}/models", headers=headers)
response.raise_for_status()
return {"status": "ok", "reachable": True, "model": cfg.ai_model}
except httpx.HTTPError:
return {"status": "unavailable", "reachable": False, "model": cfg.ai_model}
+3 -2
View File
@@ -11,8 +11,9 @@ class Settings(BaseSettings):
environment: str = "development" environment: str = "development"
ai_base_url: str = "http://localhost:11434/v1" ai_base_url: str = "http://localhost:11434/v1"
ai_api_key: str = "" ai_api_key: str = ""
ai_model: str = "llama3.2:3b" ai_model: str = "qwen2.5:1.5b"
ai_timeout_seconds: float = 20.0 # Local models may need extra time for their first response while loading.
ai_timeout_seconds: float = 90.0
model_config = SettingsConfigDict(env_file=".env", extra="ignore") model_config = SettingsConfigDict(env_file=".env", extra="ignore")
@property @property
+6 -1
View File
@@ -10,7 +10,7 @@ from .config import get_settings
from .db import Base, engine, get_db from .db import Base, engine, get_db
from .models import Entry, Profile, User from .models import Entry, Profile, User
from .schemas import CoachChatIn, CoachChatOut, EntryIn, EntryOut, ProfileIn, ProfileOut, Register, Token from .schemas import CoachChatIn, CoachChatOut, EntryIn, EntryOut, ProfileIn, ProfileOut, Register, Token
from .coach import chat_with_peg from .coach import chat_with_peg, peg_provider_health
from .security import current_user, hash_password, make_token, verify_password from .security import current_user, hash_password, make_token, verify_password
cfg = get_settings() cfg = get_settings()
@@ -31,6 +31,11 @@ def health():
return {"status": "ok", "medical_notice": "MetaBloom does not diagnose or replace medical care."} return {"status": "ok", "medical_notice": "MetaBloom does not diagnose or replace medical care."}
@app.get("/health/ai")
async def ai_health():
return await peg_provider_health(cfg)
@app.post("/coach/chat", response_model=CoachChatOut) @app.post("/coach/chat", response_model=CoachChatOut)
async def coach_chat(data: CoachChatIn, user: User = Depends(current_user)): async def coach_chat(data: CoachChatIn, user: User = Depends(current_user)):
return await chat_with_peg(data, cfg) return await chat_with_peg(data, cfg)
+8 -1
View File
@@ -6,10 +6,15 @@ services:
JWT_SECRET: ${JWT_SECRET} JWT_SECRET: ${JWT_SECRET}
ENVIRONMENT: production ENVIRONMENT: production
ALLOWED_ORIGINS: ${ALLOWED_ORIGINS:-http://localhost:8081} ALLOWED_ORIGINS: ${ALLOWED_ORIGINS:-http://localhost:8081}
AI_BASE_URL: ${AI_BASE_URL:-http://ollama:11434/v1}
AI_MODEL: ${AI_MODEL:-qwen2.5:1.5b}
AI_API_KEY: ${AI_API_KEY:-}
AI_TIMEOUT_SECONDS: ${AI_TIMEOUT_SECONDS:-90}
depends_on: depends_on:
db: db:
condition: service_healthy condition: service_healthy
ports: ["127.0.0.1:8000:8000"] ports: ["127.0.0.1:8000:8000"]
networks: [default, ai-company]
db: db:
image: postgres:16-alpine image: postgres:16-alpine
environment: environment:
@@ -24,4 +29,6 @@ services:
retries: 10 retries: 10
volumes: volumes:
postgres_data: postgres_data:
networks:
ai-company:
external: true
+11
View File
@@ -88,3 +88,14 @@ def test_coach_fallback_when_provider_is_down(monkeypatch):
response = client.post("/coach/chat", headers=auth(client), json={"message": "Donne-moi une idée simple"}) response = client.post("/coach/chat", headers=auth(client), json={"message": "Donne-moi une idée simple"})
assert response.status_code == 200 assert response.status_code == 200
assert response.json()["source"] == "fallback" assert response.json()["source"] == "fallback"
def test_ai_health_reports_provider_status(monkeypatch):
async def fake_get(self, url, **kwargs):
request = httpx.Request("GET", url)
return httpx.Response(200, request=request, json={"data": []})
monkeypatch.setattr(httpx.AsyncClient, "get", fake_get)
with TestClient(app) as client:
response = client.get("/health/ai")
assert response.status_code == 200
assert response.json()["reachable"] is True