Ajout d'une interface de chat #2
henoc Gnaoule · il y a 2 h
9 fichiers concernés
cv_service/.dockerignore Ajouté
+3 −0
| … | ||
| 1 | +.env | |
| 2 | +*.md | |
| 3 | + | |
cv_service/.env.example Ajouté
+8 −0
| … | ||
| 1 | +# Copiez ce fichier en « .env » et remplissez-le. Ne le partagez jamais. | |
| 2 | +# Clé secrète : générez-la avec openssl rand -hex 32 (la même valeur va dans cv_service_config.php sur LWS) | |
| 3 | +CV_API_KEY=remplacez-par-une-cle-de-64-caracteres | |
| 4 | +# Sous-domaine qui pointe vers l'adresse IP du VPS (enregistrement DNS de type A) | |
| 5 | +CV_DOMAINE=api.doobito.com | |
| 6 | +CV_OCR_LANG=fra+eng | |
| 7 | +CV_SIMULTANES=3 | |
| 8 | + | |
cv_service/app.py Ajouté
+80 −0
| … | ||
| 1 | +# -*- coding: utf-8 -*- | |
| 2 | +""" | |
| 3 | +app.py — service d'extraction de texte pour les CV de Doobito (tourne sur le VPS). | |
| 4 | + | |
| 5 | + POST /extraire (multipart, champ « fichier », en-tête X-API-Key) -> {"ok": true, "texte": "...", "methode": "..."} | |
| 6 | + GET /sante -> {"ok": true} | |
| 7 | + | |
| 8 | +Le service ne conserve rien : le fichier reçu est écrit dans un dossier temporaire, lu par extracteur.py dans un | |
| 9 | +processus séparé (temps et mémoire bornés), puis supprimé. Rien n'est journalisé du contenu des CV. | |
| 10 | +""" | |
| 11 | +import hmac | |
| 12 | +import json | |
| 13 | +import os | |
| 14 | +import subprocess | |
| 15 | +import sys | |
| 16 | +import tempfile | |
| 17 | +import threading | |
| 18 | + | |
| 19 | +from fastapi import FastAPI, File, Header, HTTPException, UploadFile | |
| 20 | +from fastapi.responses import JSONResponse | |
| 21 | + | |
| 22 | +CLE_API = os.environ.get("CV_API_KEY", "") | |
| 23 | +if len(CLE_API) < 32: | |
| 24 | + raise RuntimeError("CV_API_KEY manquante ou trop courte (32 caractères minimum).") | |
| 25 | + | |
| 26 | +TAILLE_MAX = 3 * 1024 * 1024 | |
| 27 | +DELAI_SANS_OCR = 25 | |
| 28 | +DELAI_AVEC_OCR = 70 | |
| 29 | +SIMULTANES = threading.BoundedSemaphore(int(os.environ.get("CV_SIMULTANES", "3"))) | |
| 30 | +DOSSIER = os.path.dirname(os.path.abspath(__file__)) | |
| 31 | + | |
| 32 | +app = FastAPI(docs_url=None, redoc_url=None, openapi_url=None) | |
| 33 | + | |
| 34 | + | |
| 35 | +def verifier_cle(cle): | |
| 36 | + if not cle or not hmac.compare_digest(cle.encode("utf-8"), CLE_API.encode("utf-8")): | |
| 37 | + raise HTTPException(status_code=401, detail="Clé invalide.") | |
| 38 | + | |
| 39 | + | |
| 40 | +@app.get("/sante") | |
| 41 | +def sante(): | |
| 42 | + return {"ok": True} | |
| 43 | + | |
| 44 | + | |
| 45 | +@app.post("/extraire") | |
| 46 | +def extraire(fichier: UploadFile = File(...), x_api_key: str = Header(default="")): | |
| 47 | + verifier_cle(x_api_key) | |
| 48 | + | |
| 49 | + contenu = fichier.file.read(TAILLE_MAX + 1) | |
| 50 | + if len(contenu) > TAILLE_MAX: | |
| 51 | + raise HTTPException(status_code=413, detail="Fichier trop volumineux.") | |
| 52 | + if not contenu: | |
| 53 | + raise HTTPException(status_code=400, detail="Fichier vide.") | |
| 54 | + | |
| 55 | + if not SIMULTANES.acquire(blocking=False): | |
| 56 | + raise HTTPException(status_code=503, detail="Service occupé, réessayez.") | |
| 57 | + try: | |
| 58 | + with tempfile.TemporaryDirectory(prefix="cv_") as dossier: | |
| 59 | + chemin = os.path.join(dossier, "cv.bin") # nom fixe : le nom envoyé par le client n'est jamais utilisé | |
| 60 | + with open(chemin, "wb") as f: | |
| 61 | + f.write(contenu) | |
| 62 | + | |
| 63 | + reponse = None | |
| 64 | + for avec_ocr in (False, True): | |
| 65 | + commande = [sys.executable, "-I", os.path.join(DOSSIER, "extracteur.py"), chemin] + (["--ocr"] if avec_ocr else []) | |
| 66 | + try: | |
| 67 | + resultat = subprocess.run(commande, capture_output=True, timeout=DELAI_AVEC_OCR if avec_ocr else DELAI_SANS_OCR) | |
| 68 | + except subprocess.TimeoutExpired: | |
| 69 | + raise HTTPException(status_code=504, detail="Lecture trop longue.") | |
| 70 | + if resultat.returncode != 0: | |
| 71 | + raise HTTPException(status_code=422, detail="Fichier illisible.") | |
| 72 | + reponse = json.loads(resultat.stdout.decode("utf-8")) | |
| 73 | + # OCR seulement si la lecture normale n'a rien donné d'exploitable (CV scanné). | |
| 74 | + lettres = sum(1 for c in reponse["texte"] if c.isalpha()) | |
| 75 | + if lettres >= 60 or avec_ocr or reponse["methode"] not in ("pypdf", "pdfminer", "aucune"): | |
| 76 | + break | |
| 77 | + return JSONResponse({"ok": True, "texte": reponse["texte"], "methode": reponse["methode"]}) | |
| 78 | + finally: | |
| 79 | + SIMULTANES.release() | |
| 80 | + | |
cv_service/Caddyfile Ajouté
+8 −0
| … | ||
| 1 | +# HTTPS automatique (certificat Let's Encrypt) pour le sous-domaine du service. | |
| 2 | +{$CV_DOMAINE} { | |
| 3 | + request_body { | |
| 4 | + max_size 4MB | |
| 5 | + } | |
| 6 | + reverse_proxy cv-service:8000 | |
| 7 | +} | |
| 8 | + | |
cv_service/docker-compose.yml Ajouté
+33 −0
| … | ||
| 1 | +services: | |
| 2 | + cv-service: | |
| 3 | + build: . | |
| 4 | + env_file: .env | |
| 5 | + restart: unless-stopped | |
| 6 | + read_only: true | |
| 7 | + tmpfs: | |
| 8 | + - /tmp | |
| 9 | + mem_limit: 2g | |
| 10 | + cpus: 2 | |
| 11 | + security_opt: | |
| 12 | + - no-new-privileges:true | |
| 13 | + # Aucun port publié : seul Caddy (HTTPS) peut joindre le service. | |
| 14 | + | |
| 15 | + caddy: | |
| 16 | + image: caddy:2 | |
| 17 | + restart: unless-stopped | |
| 18 | + depends_on: | |
| 19 | + - cv-service | |
| 20 | + ports: | |
| 21 | + - "80:80" | |
| 22 | + - "443:443" | |
| 23 | + environment: | |
| 24 | + - CV_DOMAINE=${CV_DOMAINE} | |
| 25 | + volumes: | |
| 26 | + - ./Caddyfile:/etc/caddy/Caddyfile:ro | |
| 27 | + - caddy_data:/data | |
| 28 | + - caddy_config:/config | |
| 29 | + | |
| 30 | +volumes: | |
| 31 | + caddy_data: | |
| 32 | + caddy_config: | |
| 33 | + | |
cv_service/Dockerfile Ajouté
+18 −0
| … | ||
| 1 | +FROM python:3.12-slim | |
| 2 | + | |
| 3 | +# tesseract + poppler : reconnaissance de texte (OCR) pour les CV scannés | |
| 4 | +RUN apt-get update \ | |
| 5 | + && apt-get install -y --no-install-recommends tesseract-ocr tesseract-ocr-fra tesseract-ocr-eng poppler-utils \ | |
| 6 | + && rm -rf /var/lib/apt/lists/* | |
| 7 | + | |
| 8 | +WORKDIR /app | |
| 9 | +COPY requirements.txt . | |
| 10 | +RUN pip install --no-cache-dir -r requirements.txt | |
| 11 | +COPY app.py extracteur.py ./ | |
| 12 | + | |
| 13 | +RUN useradd --system --uid 10001 --no-create-home appuser | |
| 14 | +USER appuser | |
| 15 | + | |
| 16 | +EXPOSE 8000 | |
| 17 | +CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "2"] | |
| 18 | + | |
cv_service/extracteur.py Ajouté
+143 −0
| … | ||
| 1 | +#!/usr/bin/env python3 | |
| 2 | +# -*- coding: utf-8 -*- | |
| 3 | +""" | |
| 4 | +extracteur.py — lit le texte d'un CV (PDF, DOCX ou texte) et l'écrit en JSON sur la sortie standard. | |
| 5 | + | |
| 6 | + python3 extracteur.py /chemin/fichier [--ocr] | |
| 7 | + | |
| 8 | +Lancé par app.py dans un processus séparé : si un fichier piégé fait boucler ou saturer une bibliothèque, | |
| 9 | +seul ce processus est arrêté (limites de mémoire et de temps ci-dessous), pas le service. | |
| 10 | +Sortie : {"texte": "...", "methode": "pypdf|pdfminer|ocr|docx|texte|aucune", "pages": N} | |
| 11 | +""" | |
| 12 | +import json | |
| 13 | +import os | |
| 14 | +import re | |
| 15 | +import sys | |
| 16 | +import zipfile | |
| 17 | + | |
| 18 | +PAGES_MAX = 12 | |
| 19 | +OCR_PAGES_MAX = 3 | |
| 20 | +TEXTE_MAX = 200_000 | |
| 21 | +LETTRES_MIN = 60 | |
| 22 | +DOCX_MAX_OCTETS = 8 * 1024 * 1024 | |
| 23 | + | |
| 24 | + | |
| 25 | +def limiter_ressources(): | |
| 26 | + try: | |
| 27 | + import resource | |
| 28 | + resource.setrlimit(resource.RLIMIT_AS, (2 * 1024 ** 3, 2 * 1024 ** 3)) | |
| 29 | + resource.setrlimit(resource.RLIMIT_CPU, (50, 50)) | |
| 30 | + except Exception: | |
| 31 | + pass | |
| 32 | + | |
| 33 | + | |
| 34 | +def nb_lettres(texte): | |
| 35 | + return sum(1 for c in texte if c.isalpha()) | |
| 36 | + | |
| 37 | + | |
| 38 | +def assez_de_texte(texte): | |
| 39 | + """Du vrai texte, pas une bouillie de symboles : assez de lettres et elles dominent.""" | |
| 40 | + signes = sum(1 for c in texte if not c.isspace()) | |
| 41 | + lettres = nb_lettres(texte) | |
| 42 | + return lettres >= LETTRES_MIN and signes > 0 and lettres / signes >= 0.6 | |
| 43 | + | |
| 44 | + | |
| 45 | +def lire_pypdf(chemin): | |
| 46 | + from pypdf import PdfReader | |
| 47 | + lecteur = PdfReader(chemin, strict=False) | |
| 48 | + if lecteur.is_encrypted: | |
| 49 | + try: | |
| 50 | + lecteur.decrypt("") | |
| 51 | + except Exception: | |
| 52 | + return "", 0 | |
| 53 | + textes = [] | |
| 54 | + pages = lecteur.pages[:PAGES_MAX] | |
| 55 | + for page in pages: | |
| 56 | + try: | |
| 57 | + textes.append(page.extract_text() or "") | |
| 58 | + except Exception: | |
| 59 | + textes.append("") | |
| 60 | + return "\n".join(textes), len(pages) | |
| 61 | + | |
| 62 | + | |
| 63 | +def lire_pdfminer(chemin): | |
| 64 | + from pdfminer.high_level import extract_text | |
| 65 | + return extract_text(chemin, maxpages=PAGES_MAX) or "" | |
| 66 | + | |
| 67 | + | |
| 68 | +def lire_ocr(chemin): | |
| 69 | + """CV scanné (image) : reconnaissance de texte sur les premières pages. Demande tesseract + poppler.""" | |
| 70 | + from pdf2image import convert_from_path | |
| 71 | + import pytesseract | |
| 72 | + images = convert_from_path(chemin, dpi=200, first_page=1, last_page=OCR_PAGES_MAX) | |
| 73 | + return "\n".join(pytesseract.image_to_string(image, lang=os.environ.get("CV_OCR_LANG", "fra+eng")) for image in images) | |
| 74 | + | |
| 75 | + | |
| 76 | +def lire_docx(chemin): | |
| 77 | + with zipfile.ZipFile(chemin) as z: | |
| 78 | + info = z.getinfo("word/document.xml") | |
| 79 | + if info.file_size > DOCX_MAX_OCTETS: | |
| 80 | + return "" | |
| 81 | + xml = z.read("word/document.xml").decode("utf-8", errors="replace") | |
| 82 | + xml = re.sub(r"</w:p>", "\n", xml) | |
| 83 | + xml = re.sub(r"<w:(?:tab|br)\s*/>", " ", xml) | |
| 84 | + xml = re.sub(r"<[^>]+>", "", xml) | |
| 85 | + return (xml.replace("&", "&").replace("<", "<").replace(">", ">") | |
| 86 | + .replace(""", '"').replace("'", "'")) | |
| 87 | + | |
| 88 | + | |
| 89 | +def main(): | |
| 90 | + if len(sys.argv) < 2: | |
| 91 | + sys.stderr.write("usage : extracteur.py fichier [--ocr]\n") | |
| 92 | + return 2 | |
| 93 | + limiter_ressources() | |
| 94 | + chemin = sys.argv[1] | |
| 95 | + ocr = "--ocr" in sys.argv[2:] | |
| 96 | + with open(chemin, "rb") as f: | |
| 97 | + debut = f.read(8) | |
| 98 | + | |
| 99 | + texte, methode, pages = "", "aucune", 0 | |
| 100 | + if debut.startswith(b"%PDF"): | |
| 101 | + for nom, fonction in (("pypdf", None), ("pdfminer", lire_pdfminer)): | |
| 102 | + try: | |
| 103 | + if nom == "pypdf": | |
| 104 | + t, pages = lire_pypdf(chemin) | |
| 105 | + else: | |
| 106 | + t = fonction(chemin) | |
| 107 | + except ImportError: | |
| 108 | + continue | |
| 109 | + except Exception as erreur: | |
| 110 | + sys.stderr.write("%s : %s\n" % (nom, erreur)) | |
| 111 | + continue | |
| 112 | + if len(t) > len(texte): | |
| 113 | + texte, methode = t, nom | |
| 114 | + if assez_de_texte(t): | |
| 115 | + texte, methode = t, nom | |
| 116 | + break | |
| 117 | + if not assez_de_texte(texte) and ocr: | |
| 118 | + try: | |
| 119 | + t = lire_ocr(chemin) | |
| 120 | + if nb_lettres(t) > nb_lettres(texte): | |
| 121 | + texte, methode = t, "ocr" | |
| 122 | + except Exception as erreur: | |
| 123 | + sys.stderr.write("ocr : %s\n" % erreur) | |
| 124 | + elif debut.startswith(b"PK"): | |
| 125 | + try: | |
| 126 | + texte, methode = lire_docx(chemin), "docx" | |
| 127 | + except Exception as erreur: | |
| 128 | + sys.stderr.write("docx : %s\n" % erreur) | |
| 129 | + else: | |
| 130 | + brut = open(chemin, "rb").read(TEXTE_MAX * 4) | |
| 131 | + try: | |
| 132 | + texte = brut.decode("utf-8") | |
| 133 | + except UnicodeDecodeError: | |
| 134 | + texte = brut.decode("cp1252", errors="replace") | |
| 135 | + methode = "texte" | |
| 136 | + | |
| 137 | + json.dump({"texte": texte[:TEXTE_MAX], "methode": methode, "pages": pages}, sys.stdout, ensure_ascii=False) | |
| 138 | + return 0 | |
| 139 | + | |
| 140 | + | |
| 141 | +if __name__ == "__main__": | |
| 142 | + sys.exit(main()) | |
| 143 | + | |
cv_service/README.md Ajouté
+28 −0
| … | ||
| 1 | +# Service d'extraction de CV (VPS) | |
| 2 | + | |
| 3 | +Reçoit un CV (PDF, Word, texte) de Doobito, renvoie son texte. Lit les PDF de Word/Google Docs/Canva avec pypdf, | |
| 4 | +et applique la reconnaissance de texte (OCR) aux PDF scannés. Rien n'est conservé. | |
| 5 | + | |
| 6 | +## Installation sur le VPS (une seule fois) | |
| 7 | + | |
| 8 | +1. DNS : créez un enregistrement **A** `api.doobito.com` -> adresse IP du VPS (chez le gestionnaire de votre domaine). | |
| 9 | +2. Copiez le dossier `cv_service` sur le VPS (WinSCP, ou `scp -r cv_service utilisateur@IP:~/`). | |
| 10 | +3. Sur le VPS : | |
| 11 | + | |
| 12 | + cd ~/cv_service | |
| 13 | + cp .env.example .env | |
| 14 | + openssl rand -hex 32 # copiez le résultat | |
| 15 | + nano .env # collez-le dans CV_API_KEY, vérifiez CV_DOMAINE | |
| 16 | + sudo ufw allow 80/tcp && sudo ufw allow 443/tcp # si le pare-feu ufw est actif | |
| 17 | + docker compose up -d --build | |
| 18 | + | |
| 19 | +4. Test (depuis n'importe où) : ouvrez `https://api.doobito.com/sante` -> `{"ok":true}`. | |
| 20 | +5. Sur LWS, dans le dossier de `cv_extraction.php`, éditez `cv_service_config.php` : mettez la même clé et l'adresse. | |
| 21 | + | |
| 22 | +## Exploitation | |
| 23 | + | |
| 24 | +- Journaux : `docker compose logs -f cv-service` | |
| 25 | +- Mise à jour : copiez les nouveaux fichiers puis `docker compose up -d --build` | |
| 26 | +- Arrêt : `docker compose down` | |
| 27 | +- Si le service est arrêté, Doobito continue de fonctionner : PHP retombe sur son propre lecteur de PDF. | |
| 28 | + | |
cv_service/requirements.txt Ajouté
+8 −0
| … | ||
| 1 | +fastapi>=0.110 | |
| 2 | +uvicorn>=0.29 | |
| 3 | +python-multipart>=0.0.9 | |
| 4 | +pypdf>=4.0 | |
| 5 | +pdfminer.six>=20231228 | |
| 6 | +pdf2image>=1.17 | |
| 7 | +pytesseract>=0.3.10 | |
| 8 | + | |