Ajout du matching entre nouvelle correspondance et correspondance IGPDE

This commit is contained in:
2026-04-15 17:28:20 +02:00
parent 641885ed85
commit 84be398b82
5 changed files with 262 additions and 228 deletions
+101
View File
@@ -0,0 +1,101 @@
Nouvelle nomenclature,Nomenclature IGPDE
"700 Management, Leadership et Compétences Managériales",
701 - Management d'équipe,Management
703 - Communication managériale,Communication > Communication managériale
704 - Gestion des conflits et des situations difficiles,Médiation
705 - Accompagnement du changement,
706 - Intelligence collective et émotionnelle,
707 - Évaluation et entretien professionnel,
,
"500 Ressources Humaines, Carrière, Prépa concours",
501 Recrutement,
502 - Gestion de carrière et mobilité,Ressources humaines > Mobilité et transitions professionnelles
"503 - Diversité, inclusion et laïcité, lutte contre les discriminations","Ressources humaines > Diversité, laïcité et égalité professionnelle"
504 - Dialogue social et relations professionnelles,
505 - Accueil et intégration,
506 - RH à lInsee : outils et métiers,Ressources humaines
507 - Préparation aux concours et examens,
508 - Jury,
509 Formation initiale,
509 Formation initiale,
,
100 Méthodologie et production statistique,
101 - Enquêtes entreprises,
"102 - Enquêtes ménages (yc PCS, ISCO)",
103 - Enquêtes ménages formations spécifiques enquêteurs,
104 Répertoire Sirene,
105 Recensement de la population,
"106 Conjoncture, études économiques",
107 Comptabilité nationale,
108 - Autres outils et logiciels d'enquête,
109 - Méthodologie denquête ,
110 Communication avec les enquêtés,
,
"200 Statistique, Analyse de Données et Data Science",
201 - Statistique descriptive et analyse,
"202 - Programmation statistique (R, Python pour les statisticiens)","Big data, IA, DevOps > Langages de programmation"
203 - Visualisation de données (QGIS et autres),
204 - Séries temporelles et prévision,
,
"300 Informatique, Numérique et Systèmes d'Information",
301 - Développement et programmation informatique,Informatique > Langages et Développement
301 - Développement et programmation informatique,"Big data, IA, DevOps"
302 - Bases de données et SQL,Informatique > Bases de données
303 - Outils bureautiques,Bureautique
304 - Sécurité informatique,Informatique > Environnement informatique et sécurité
305 - Accessibilité numérique,
306 - Infrastructure et administration système,Informatique > Modélisation de projets SI
306 - Infrastructure et administration système,Informatique > Réseaux
306 - Infrastructure et administration système,Informatique > Systèmes dexploitation Poste de travail
306 - Infrastructure et administration système,Informatique > Systèmes dexploitation Serveurs
307 - Agilité et gestion de projet,Informatique > Gestion de projet SI
308 - Eco-conception et Green IT,
309 Intelligence artificielle,"Big data, IA, DevOps > Intelligence Artificielle"
310 Autres formations au numérique,"Numérique, innovation, créativité"
,
"600 Santé, Sécurité et Prévention des Risques","Ressources humaines > Qualité de vie, santé au travail et sécurité au travail"
"600 Santé, Sécurité et Prévention des Risques",
,
"400 Connaissance de l'Insee, des Régions, de l’État, des entreprises",
401 - Présentation de l'Insee et de ses missions,
402 - Connaissance des régions et des territoires,
"403 Action publique, connaissance de ladministration",Action publique Environnement administratif
403 Europe et international,Europe International
404 Economie dentreprise,Connaître et accompagner les entreprises
,"Comptabilité, gestion et analyse financière des entreprises"
,
"800 Communication, Rédaction et Techniques Éditoriales",
801 - Communication écrite,Communication > Communication écrite
"802 - Communication orale, prise de parole",Communication > Communication orale
803 Communication visuelle et digitale,Communication > Communication digitale
804 - Rédaction administrative et technique,Communication >
805 - Pédagogie des adultes,
,
"900 Juridique, Réglementaire et Déontologie",
901 - Droit du travail et statut de la fonction publique,
"902 - Commande publique, marchés",Commande publique et achats publics
903 Gestion publique,Gestion publique
904 - Déontologie et éthique,
905 - Accessibilité et RGAA,
907 Droit autres,Droit
908 Archivage,Gestion et recherche documentaire
,
1000 Transition Écologique et Développement Durable,Développement durable
,
"1100 Langues, Compétences Transversales",
1101 - Langues,Langues
1102 - Soft skills et développement personnel,Efficacité professionnelle - Intelligence relationnelle et développement du potentiel créatif
,
1200 Autres Formations Spécifiques ou Transversales,
1201 - Formations transversales,
1203 Economie générale,Economie générale
1203 - Formations techniques diverses,Immobilier
1201 - Formations transversales,Événements & séminaires
507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF)
507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie A
507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie B
507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie C
507 - Préparation aux concours et examens,Préparations aux concours interministériels
507 - Préparation aux concours et examens,Qualifications informatiques (agents MEF)
507 - Préparation aux concours et examens,Qualifications informatiques (agents MEF) > Préparations de catégorie A - Qualifications informatiques
507 - Préparation aux concours et examens,Qualifications informatiques (agents MEF) > Préparations de catégorie B - Qualifications informatiques
1 Nouvelle nomenclature Nomenclature IGPDE
2 700 – Management, Leadership et Compétences Managériales
3 701 - Management d'équipe Management
4 703 - Communication managériale Communication > Communication managériale
5 704 - Gestion des conflits et des situations difficiles Médiation
6 705 - Accompagnement du changement
7 706 - Intelligence collective et émotionnelle
8 707 - Évaluation et entretien professionnel
9
10 500 – Ressources Humaines, Carrière, Prépa concours
11 501 – Recrutement
12 502 - Gestion de carrière et mobilité Ressources humaines > Mobilité et transitions professionnelles
13 503 - Diversité, inclusion et laïcité, lutte contre les discriminations Ressources humaines > Diversité, laïcité et égalité professionnelle
14 504 - Dialogue social et relations professionnelles
15 505 - Accueil et intégration
16 506 - RH à l’Insee : outils et métiers Ressources humaines
17 507 - Préparation aux concours et examens
18 508 - Jury
19 509 – Formation initiale
20 509 – Formation initiale
21
22 100 – Méthodologie et production statistique
23 101 - Enquêtes entreprises
24 102 - Enquêtes ménages (yc PCS, ISCO)
25 103 - Enquêtes ménages – formations spécifiques enquêteurs
26 104 – Répertoire Sirene
27 105 – Recensement de la population
28 106 – Conjoncture, études économiques
29 107 – Comptabilité nationale
30 108 - Autres outils et logiciels d'enquête
31 109 - Méthodologie d’enquête
32 110 – Communication avec les enquêtés
33
34 200 – Statistique, Analyse de Données et Data Science
35 201 - Statistique descriptive et analyse
36 202 - Programmation statistique (R, Python pour les statisticiens) Big data, IA, DevOps > Langages de programmation
37 203 - Visualisation de données (QGIS et autres)
38 204 - Séries temporelles et prévision
39
40 300 – Informatique, Numérique et Systèmes d'Information
41 301 - Développement et programmation informatique Informatique > Langages et Développement
42 301 - Développement et programmation informatique Big data, IA, DevOps
43 302 - Bases de données et SQL Informatique > Bases de données
44 303 - Outils bureautiques Bureautique
45 304 - Sécurité informatique Informatique > Environnement informatique et sécurité
46 305 - Accessibilité numérique
47 306 - Infrastructure et administration système Informatique > Modélisation de projets SI
48 306 - Infrastructure et administration système Informatique > Réseaux
49 306 - Infrastructure et administration système Informatique > Systèmes d’exploitation – Poste de travail
50 306 - Infrastructure et administration système Informatique > Systèmes d’exploitation – Serveurs
51 307 - Agilité et gestion de projet Informatique > Gestion de projet SI
52 308 - Eco-conception et Green IT
53 309 – Intelligence artificielle Big data, IA, DevOps > Intelligence Artificielle
54 310 – Autres formations au numérique Numérique, innovation, créativité
55
56 600 – Santé, Sécurité et Prévention des Risques Ressources humaines > Qualité de vie, santé au travail et sécurité au travail
57 600 – Santé, Sécurité et Prévention des Risques
58
59 400 – Connaissance de l'Insee, des Régions, de l’État, des entreprises
60 401 - Présentation de l'Insee et de ses missions
61 402 - Connaissance des régions et des territoires
62 403 – Action publique, connaissance de l’administration Action publique – Environnement administratif
63 403 – Europe et international Europe – International
64 404 – Economie d’entreprise Connaître et accompagner les entreprises
65 Comptabilité, gestion et analyse financière des entreprises
66
67 800 – Communication, Rédaction et Techniques Éditoriales
68 801 - Communication écrite Communication > Communication écrite
69 802 - Communication orale, prise de parole Communication > Communication orale
70 803 – Communication visuelle et digitale Communication > Communication digitale
71 804 - Rédaction administrative et technique Communication >
72 805 - Pédagogie des adultes
73
74 900 – Juridique, Réglementaire et Déontologie
75 901 - Droit du travail et statut de la fonction publique
76 902 - Commande publique, marchés Commande publique et achats publics
77 903 – Gestion publique Gestion publique
78 904 - Déontologie et éthique
79 905 - Accessibilité et RGAA
80 907 – Droit – autres Droit
81 908 – Archivage Gestion et recherche documentaire
82
83 1000 – Transition Écologique et Développement Durable Développement durable
84
85 1100 – Langues, Compétences Transversales
86 1101 - Langues Langues
87 1102 - Soft skills et développement personnel Efficacité professionnelle - Intelligence relationnelle et développement du potentiel créatif
88
89 1200 – Autres Formations Spécifiques ou Transversales
90 1201 - Formations transversales
91 1203 – Economie générale Economie générale
92 1203 - Formations techniques diverses Immobilier
93 1201 - Formations transversales Événements & séminaires
94 507 - Préparation aux concours et examens Préparations aux concours et examens professionnels (agents MEF)
95 507 - Préparation aux concours et examens Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie A
96 507 - Préparation aux concours et examens Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie B
97 507 - Préparation aux concours et examens Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie C
98 507 - Préparation aux concours et examens Préparations aux concours interministériels
99 507 - Préparation aux concours et examens Qualifications informatiques (agents MEF)
100 507 - Préparation aux concours et examens Qualifications informatiques (agents MEF) > Préparations de catégorie A - Qualifications informatiques
101 507 - Préparation aux concours et examens Qualifications informatiques (agents MEF) > Préparations de catégorie B - Qualifications informatiques
BIN
View File
Binary file not shown.
+3 -1
View File
@@ -4,4 +4,6 @@ services:
container_name: extracteur-pdf
restart: unless-stopped
ports:
- "5050:5000" # Choix du port à définir en fonction des services déjà existants
- "5050:5000"
volumes:
- .:/app
+1
View File
@@ -3,3 +3,4 @@ pdfplumber
pandas
openpyxl
xlsxwriter
odfpy
+156 -226
View File
@@ -1,277 +1,207 @@
import os
import io
import json
import re
import pandas as pd
import pdfplumber
from flask import Flask, request, render_template_string, Response, send_from_directory
import unicodedata
import traceback
from flask import Flask, request, render_template_string, Response, send_from_directory, redirect
app = Flask(__name__)
# Dossier temporaire pour stocker le fichier généré à l'intérieur du conteneur
# Dossiers et fichiers
TEMP_DIR = "/app/downloads"
if not os.path.exists(TEMP_DIR):
os.makedirs(TEMP_DIR)
if not os.path.exists(TEMP_DIR): os.makedirs(TEMP_DIR)
# --- DESIGN HTML & JAVASCRIPT ---
# Ce bloc est envoyé en premier au navigateur pour préparer l'interface de log
HTML_LAYOUT = """
<!DOCTYPE html>
<html lang="fr">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Extracteur de formations</title>
<style>
body { font-family: 'Segoe UI', Tahoma, Geneva, Verdana, sans-serif; background: #f0f2f5; padding: 20px; color: #2d3748; }
.container { max-width: 900px; margin: 0 auto; background: white; padding: 30px; border-radius: 12px; box-shadow: 0 4px 20px rgba(0,0,0,0.08); }
.log-window { background: #1a202c; color: #cbd5e0; padding: 15px; border-radius: 8px; height: 400px; overflow-y: auto; font-family: 'Courier New', Courier, monospace; font-size: 12px; margin-top: 20px; border: 1px solid #2d3748; line-height: 1.5; }
.progress-bar { width: 100%; background: #e2e8f0; height: 12px; border-radius: 6px; margin-top: 20px; overflow: hidden; }
.progress-fill { height: 100%; background: #4299e1; width: 0%; transition: width 0.3s ease; }
.status-ok { color: #68d391; font-weight: bold; }
.status-skip { color: #718096; }
.status-err { color: #f56565; font-weight: bold; }
.btn-download { display: inline-block; background: #48bb78; color: white; padding: 15px 30px; border-radius: 8px; text-decoration: none; font-weight: bold; margin-top: 25px; transition: 0.2s; box-shadow: 0 4px 6px rgba(0,0,0,0.1); }
.btn-download:hover { background: #38a169; transform: translateY(-2px); }
h1 { margin-top: 0; color: #2d3748; font-size: 1.5rem; }
p { color: #4a5568; }
</style>
</head>
<body>
<div class="container">
<h1>Analyse du catalogue en cours...</h1>
<p>Chaque page du fichier est analysée pour extraire les fiches de formation.</p>
TEMP_DATA_FILE = os.path.join(TEMP_DIR, "temp_data.json")
TEMP_MISSING_FILE = os.path.join(TEMP_DIR, "temp_missing.json")
CSV_FILE = "/app/corresp_IGPDE.csv"
ODS_FILE = "/app/corresp_IGPDE.ods"
RESULT_FILE = os.path.join(TEMP_DIR, "resultat.xlsx")
<div class="progress-bar"><div id="progress" class="progress-fill"></div></div>
# --- GESTION DE LA BASE DE DONNÉES ---
def ensure_database():
if os.path.exists(CSV_FILE): return
if os.path.exists(ODS_FILE):
try:
df = pd.read_excel(ODS_FILE, engine="odf")
df.to_csv(CSV_FILE, index=False, encoding='utf-8-sig')
return
except: pass
df_empty = pd.DataFrame(columns=["Nouvelle nomenclature", "Nomenclature IGPDE"])
df_empty.to_csv(CSV_FILE, index=False, encoding='utf-8-sig')
<div id="logs" class="log-window">
> Initialisation du moteur d'extraction...<br>
</div>
# --- FONCTION DE NORMALISATION ---
def super_norm(texte):
if not texte or not isinstance(texte, str): return ""
t = "".join(c for c in unicodedata.normalize('NFD', texte) if unicodedata.category(c) != 'Mn')
t = re.sub(r'[^a-zA-Z0-9]', '', t)
return t.lower()
<div id="final-link" style="display:none; text-align: center;">
<hr style="margin: 30px 0; border: 0; border-top: 1px solid #e2e8f0;">
<p>✅ <b>Analyse terminée avec succès !</b></p>
<a href="/download" class="btn-download">📥 Télécharger le résultat (Excel)</a>
<br><br><a href="/" style="color: #a0aec0; font-size: 0.9rem;">Lancer une nouvelle analyse</a>
</div>
</div>
<script>
const logs = document.getElementById('logs');
const progress = document.getElementById('progress');
function updateLog(msg, percent) {
const line = document.createElement('div');
line.innerHTML = msg;
logs.appendChild(line);
logs.scrollTop = logs.scrollHeight;
progress.style.width = percent + '%';
}
function showFinal() {
document.getElementById('final-link').style.display = 'block';
document.querySelector('h1').innerText = "Extraction terminée !";
}
</script>
# --- DESIGN CSS ---
COMMON_STYLE = """
<style>
body { font-family: 'Segoe UI', Tahoma, Geneva, Verdana, sans-serif; background: #f4f7f9; color: #333; margin: 0; padding: 20px; }
.nav-bar { display: flex; gap: 10px; margin-bottom: 20px; justify-content: center; }
.nav-btn { background: #fff; border: 1px solid #cbd5e0; padding: 8px 15px; border-radius: 8px; text-decoration: none; color: #4a5568; font-size: 0.9rem; transition: 0.2s; }
.nav-btn:hover { background: #edf2f7; }
.nav-btn.active { background: #3182ce; color: white; border-color: #3182ce; }
.card { background: white; padding: 2rem; border-radius: 15px; box-shadow: 0 4px 15px rgba(0,0,0,0.05); max-width: 1200px; margin: 0 auto; }
h1 { color: #2d3748; margin-top: 0; font-size: 1.5rem; }
.header-box { display: flex; justify-content: space-between; align-items: center; margin-bottom: 20px; }
.btn-submit { background: #3182ce; color: white; border: none; padding: 10px 20px; border-radius: 8px; font-size: 0.9rem; cursor: pointer; font-weight: 600; text-decoration: none; display: inline-block; }
.btn-submit:hover { background: #2b6cb0; }
#log-window { background: #1a202c; color: #cbd5e0; padding: 15px; border-radius: 10px; height: 350px; overflow-y: auto; font-family: 'Courier New', monospace; font-size: 0.85rem; text-align: left; margin-top: 20px; line-height: 1.6; border: 1px solid #2d3748; }
.log-success { color: #68d391; font-weight: bold; }
.log-warn { color: #fc8181; }
.table-container { overflow-x: auto; border-radius: 10px; border: 1px solid #e2e8f0; max-height: 600px; }
table { width: 100%; border-collapse: collapse; background: white; font-size: 0.85rem; }
th { background: #f8fafc; padding: 10px; border-bottom: 2px solid #e2e8f0; text-align: left; position: sticky; top: 0; }
td { padding: 8px 10px; border-bottom: 1px solid #edf2f7; }
</style>
"""
# --- PAGE D'ACCUEIL ---
def get_nav(active_page):
h = "active" if active_page=="home" else ""
c = "active" if active_page=="corresp" else ""
r = "active" if active_page=="result" else ""
return f'<div class="nav-bar"><a href="/" class="nav-btn {h}">🏠 Accueil</a><a href="/view_correspondence" class="nav-btn {c}">📋 Table de Correspondance</a><a href="/view_result" class="nav-btn {r}">📊 Dernier Résultat</a></div>'
@app.route("/")
def home():
return render_template_string("""
<!DOCTYPE html>
<html lang="fr">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Extracteur de formations</title>
<style>
body { font-family: 'Segoe UI', system-ui, sans-serif; background: #f0f2f5; display: flex; justify-content: center; align-items: center; height: 100vh; margin: 0; }
.card { background: white; padding: 2.5rem; border-radius: 20px; box-shadow: 0 10px 30px rgba(0,0,0,0.08); text-align: center; width: 100%; max-width: 420px; }
ensure_database()
return render_template_string(f'<!DOCTYPE html><html><head><meta charset="UTF-8">{COMMON_STYLE}</head><body>{get_nav("home")}<div class="card" style="text-align:center; max-width: 500px;"><div style="font-size: 3rem; margin-bottom: 1rem;">📄</div><h1>Extraction PDF</h1><form action="/process" method="post" enctype="multipart/form-data"><input type="file" name="pdf_file" accept=".pdf" required id="file-input" style="display:none;" onchange="document.getElementById(\'file-label\').innerText = this.files[0].name"><label for="file-input" id="file-label" style="display:block; border: 2px dashed #cbd5e0; padding: 2rem; border-radius: 10px; cursor: pointer; margin-bottom:1.5rem; background:#f8fafc;">📂 Choisir le PDF</label><button type="submit" class="btn-submit">Lancer l\'analyse</button></form></div></body></html>')
h1 { color: #1a202c; margin-bottom: 0.5rem; font-size: 1.8rem; }
.subtitle { color: #718096; margin-bottom: 2rem; font-size: 0.95rem; line-height: 1.4; }
@app.route("/view_correspondence")
def view_correspondence():
ensure_database()
df = pd.read_csv(CSV_FILE, encoding='utf-8-sig').fillna('')
table_html = df.to_html(index=False)
return render_template_string(f'<!DOCTYPE html><html><head>{COMMON_STYLE}</head><body>{get_nav("corresp")}<div class="card"><div class="header-box"><h1>Table de Correspondance (CSV)</h1><a href="/download_db" class="btn-submit">📥 Télécharger la base (CSV)</a></div><div class="table-container">{table_html}</div></div></body></html>')
/* Style du bouton Parcourir personnalisé */
.file-input-container { margin-bottom: 1.5rem; position: relative; }
@app.route("/view_result")
def view_result():
if not os.path.exists(RESULT_FILE): return "Aucun résultat."
df = pd.read_excel(RESULT_FILE).fillna('')
table_html = df.to_html(index=False)
return render_template_string(f'<!DOCTYPE html><html><head>{COMMON_STYLE}</head><body>{get_nav("result")}<div class="card"><div class="header-box"><h1>Dernier Résultat Extrait</h1><a href="/download" class="btn-submit">📥 Télécharger le résultat (Excel)</a></div><div class="table-container">{table_html}</div></div></body></html>')
#pdf_file { display: none; } /* On cache l'input moche par défaut */
.custom-file-upload {
display: flex;
align-items: center;
justify-content: center;
gap: 12px;
border: 2px dashed #cbd5e0;
padding: 1.5rem;
border-radius: 12px;
cursor: pointer;
transition: all 0.2s ease;
background: #f8fafc;
color: #4a5568;
}
.custom-file-upload:hover {
border-color: #3182ce;
background: #ebf8ff;
color: #2b6cb0;
}
.file-icon { font-size: 1.5rem; }
#file-name { font-weight: 500; font-size: 0.9rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; max-width: 250px; }
/* Style du bouton Valider */
.btn-submit {
background: #3182ce;
color: white;
border: none;
padding: 14px 24px;
border-radius: 12px;
font-size: 1rem;
cursor: pointer;
width: 100%;
font-weight: 600;
transition: all 0.2s;
box-shadow: 0 4px 6px rgba(49, 130, 206, 0.2);
}
.btn-submit:hover {
background: #2b6cb0;
transform: translateY(-1px);
box-shadow: 0 6px 12px rgba(49, 130, 206, 0.3);
}
.btn-submit:active { transform: translateY(0); }
</style>
</head>
<body>
<div class="card">
<div style="font-size: 3rem; margin-bottom: 1rem;">📄</div>
<h1>Extracteur de formations</h1>
<p class="subtitle">Sélectionnez le catalogue de formations (PDF) pour extraire les données vers Excel.</p>
<form action="/process" method="post" enctype="multipart/form-data" onsubmit="return validateFile()">
<div class="file-input-container">
<input type="file" name="pdf_file" id="pdf_file" accept=".pdf" required onchange="updateFileName()">
<label for="pdf_file" class="custom-file-upload">
<span class="file-icon">📂</span>
<span id="file-name">Choisir le fichier PDF</span>
</label>
</div>
<button type="submit" class="btn-submit">Lancer l'extraction</button>
</form>
</div>
<script>
// Affiche le nom du fichier une fois sélectionné
function updateFileName() {
const input = document.getElementById('pdf_file');
const fileNameDisplay = document.getElementById('file-name');
if (input.files.length > 0) {
fileNameDisplay.innerText = input.files[0].name;
document.querySelector('.custom-file-upload').style.borderColor = '#48bb78';
document.querySelector('.custom-file-upload').style.background = '#f0fff4';
}
}
// Vérifie que c'est bien un PDF avant d'envoyer
function validateFile() {
const input = document.getElementById('pdf_file');
const file = input.files[0];
if (file && !file.name.toLowerCase().endsWith('.pdf')) {
alert("Erreur : Veuillez sélectionner un fichier au format PDF uniquement.");
return false;
}
return true;
}
</script>
</body>
</html>
""")
# --- MOTEUR D'EXTRACTION ---
@app.route("/process", methods=["POST"])
def process():
ensure_database()
file = request.files['pdf_file']
if not file: return "Fichier manquant"
file_content = file.read() # On garde le fichier en mémoire vive
file_content = file.read()
def generate():
yield HTML_LAYOUT # Envoie l'interface au navigateur
yield f'<!DOCTYPE html><html lang="fr"><head><meta charset="UTF-8">{COMMON_STYLE}</head><body><div class="card"><h1>Analyse en cours...</h1><div style="background:#e2e8f0; height:12px; border-radius:6px; overflow:hidden;"><div id="prog" style="background:#3182ce; width:0%; height:100%; transition:0.3s;"></div></div><div id="log-window"></div></div><script>const logWin = document.getElementById("log-window"); const prog = document.getElementById("prog"); function addLog(msg, type="info", p=null) {{ const div = document.createElement("div"); div.className = "log-" + type; div.innerHTML = "> " + msg; logWin.appendChild(div); logWin.scrollTop = logWin.scrollHeight; if(p !== null) prog.style.width = p + "%"; }}</script>'
all_formations = []
COLONNES_ADMISES = ["Référence", "Durée", "Type", "Publics éligibles", "Niveau", "Sessions", "Tarifs"]
try:
with pdfplumber.open(io.BytesIO(file_content)) as pdf:
total_pages = len(pdf.pages)
yield f"<script>updateLog('<b>Document chargé : {total_pages} pages identifiées.</b>', 0)</script>"
for i, page in enumerate(pdf.pages):
page_num = i + 1
percent = int((page_num / total_pages) * 100)
percent = int((page_num / total_pages) * 85)
try:
# Zone de détection (haut de page)
check_zone = page.crop((page.width * 0.5, 0, page.width, 400))
text = check_zone.extract_text() or ""
if "Référence" in (check_zone.extract_text() or ""):
bandeaux = sorted([r for r in page.rects if (r['x1'] - r['x0']) > page.width * 0.7 and r['top'] < 300 and 15 < r['height'] < 200], key=lambda r: r['top'])
if len(bandeaux) >= 2:
titre_brut = " ".join((page.crop((0, max(0, bandeaux[0]['top'] - 2), page.width, min(page.height, bandeaux[0]['bottom'] + 2))).extract_text() or "").replace('\n', ' ').split())
categorie_str = " ".join((page.crop((0, max(0, bandeaux[1]['top'] - 2), page.width, min(page.height, bandeaux[1]['bottom'] + 2))).extract_text() or "").replace('\n', ' ').split())
else:
header = page.crop((0, 60, page.width, 250))
lines = [l.strip() for l in (header.extract_text() or "").split('\n') if l.strip() and not (len(l) > 90 or l.endswith('.'))]
titre_brut, categorie_str = (" ".join(lines[:-1]), lines[-1]) if len(lines) >= 2 else (lines[0] if lines else "", "")
if "Référence" in text:
# --- EXTRACTION DE LA FICHE ---
header = page.crop((0, 0, page.width, 100))
titre_brut = header.extract_text().split('\n')[0].strip()
# Sécurisation du titre pour JS (on enlève les quotes)
titre_clean = titre_brut.replace("'", "").replace('"', '').strip()[:60]
chapitre, sous_chap = (categorie_str.split(">")[0].strip(), categorie_str.split(">")[1].strip()) if ">" in categorie_str else (categorie_str.strip(), "")
data = {'Nom_IGPDE_Key': f"{chapitre} > {sous_chap}" if sous_chap else chapitre, 'Chapitre': chapitre, 'Sous-chapitre': sous_chap, 'Titre': titre_brut, 'Page_Source': page_num}
data = {'Titre': titre_brut, 'Page_Source': page_num}
# Extraction du tableau à droite
right_side = page.crop((page.width * 0.6, 180, page.width, 600))
table = right_side.extract_table()
table = page.crop((page.width * 0.6, 180, page.width, 600)).extract_table()
if table:
for row in table:
if len(row) >= 2 and row[0]:
cle = row[0].replace('\n', ' ').strip()
for col in COLONNES_ADMISES:
for col in ["Référence", "Durée", "Type", "Publics éligibles", "Niveau", "Sessions", "Tarifs"]:
if col.lower() in cle.lower():
data[col] = row[1].replace('\n', ' ').strip()
break
all_formations.append(data)
yield f"<script>updateLog('<span class=\"status-ok\">[OK] Page {page_num} : formation extraite ({titre_clean}...)</span>', {percent})</script>"
t_log = titre_brut[:40].replace("'", " ").replace('"', ' ')
yield f"<script>addLog('Page {page_num} : {t_log}...', 'success', {percent})</script>"
except Exception: pass
else:
# --- PAGE IGNORÉE ---
yield f"<script>updateLog('<span class=\"status-skip\">[SKIP] Page {page_num} : pas une fiche de formation</span>', {percent})</script>"
# --- MAPPING ---
mapping_dict = {}
df_map = pd.read_csv(CSV_FILE, encoding='utf-8-sig')
for _, row in df_map.dropna(subset=['Nomenclature IGPDE']).iterrows():
mapping_dict[super_norm(str(row['Nomenclature IGPDE']))] = str(row['Nouvelle nomenclature']).strip()
except Exception as page_err:
yield f"<script>updateLog('<span class=\"status-err\">[ERREUR] Page {page_num} : {str(page_err)}</span>', {percent})</script>"
# --- FIN DE BOUCLE : SAUVEGARDE EXCEL ---
yield f"<script>updateLog('<b>Génération du fichier Excel final...</b>', 99)</script>"
if all_formations:
df = pd.DataFrame(all_formations)
output_path = os.path.join(TEMP_DIR, "resultat.xlsx")
# On utilise xlsxwriter pour la stabilité
df.to_excel(output_path, index=False, engine='xlsxwriter')
yield f"<script>updateLog('<b>Extraction terminée ! {len(all_formations)} formations trouvées.</b>', 100)</script>"
yield "<script>showFinal()</script>"
else:
yield f"<script>updateLog('<span class=\"status-err\">Aucune formation trouvée dans le document.</span>', 100)</script>"
except Exception as global_err:
yield f"<script>updateLog('<span class=\"status-err\">ERREUR GLOBALE : {str(global_err)}</span>', 0)</script>"
missing = []
seen_missing_norm = set()
for f in all_formations:
full_norm, chap_norm = super_norm(f['Nom_IGPDE_Key']), super_norm(f['Chapitre'])
if full_norm not in mapping_dict and chap_norm not in mapping_dict:
if full_norm not in seen_missing_norm:
missing.append(f['Nom_IGPDE_Key']); seen_missing_norm.add(full_norm)
with open(TEMP_DATA_FILE, "w", encoding="utf-8") as f: json.dump(all_formations, f)
if missing:
with open(TEMP_MISSING_FILE, "w", encoding="utf-8") as f: json.dump(missing, f)
yield "<script>setTimeout(() => window.location.href='/ask_mapping', 800);</script>"
else:
yield "<script>setTimeout(() => window.location.href='/finalize', 800);</script>"
except Exception as e:
traceback.print_exc()
yield f"<script>addLog('Erreur : {str(e)}', 'warn')</script>"
yield "</body></html>"
return Response(generate(), mimetype='text/html')
# --- TÉLÉCHARGEMENT ---
@app.route("/download")
def download():
return send_from_directory(TEMP_DIR, "resultat.xlsx", as_attachment=True)
@app.route("/ask_mapping")
def ask_mapping():
with open(TEMP_MISSING_FILE, "r", encoding="utf-8") as f: missing = json.load(f)
df_db = pd.read_csv(CSV_FILE, encoding='utf-8-sig')
known = sorted(df_db['Nouvelle nomenclature'].dropna().unique().tolist())
options_html = "".join([f'<option value="{kn}">' for kn in known])
html = f'<!DOCTYPE html><html><head>{COMMON_STYLE}</head><body><div class="card"><h1>Nouveaux Chapitres</h1><form action="/save_mapping" method="POST">'
for i, cat in enumerate(missing):
html += f'<div style="margin-bottom:15px; background:#f7fafc; padding:15px; border-radius:10px;"><label style="display:block; font-weight:600; font-size:0.85rem; margin-bottom:5px;">{cat}</label><input type="hidden" name="igpde_{i}" value="{cat}"><input type="text" name="nouv_{i}" list="known_noms" placeholder="Associer à..." style="width:100%; padding:10px; border:1px solid #cbd5e0; border-radius:6px;"></div>'
html += f'<datalist id="known_noms">{options_html}</datalist><button type="submit" class="btn-submit" style="width:100%">Valider</button></form></div></body></html>'
return render_template_string(html)
if __name__ == "__main__":
# Écoute sur 0.0.0.0 pour Docker
app.run(host='0.0.0.0', port=5000, debug=False)
@app.route("/save_mapping", methods=["POST"])
def save_mapping():
new_rows = []
for k in [k for k in request.form.keys() if k.startswith("igpde_")]:
idx = k.split("_")[1]
new_rows.append({"Nouvelle nomenclature": request.form.get(f"nouv_{idx}", "").strip(), "Nomenclature IGPDE": request.form.get(k)})
if new_rows:
df_final = pd.concat([pd.read_csv(CSV_FILE, encoding='utf-8-sig'), pd.DataFrame(new_rows)], ignore_index=True)
df_final.to_csv(CSV_FILE, index=False, encoding='utf-8-sig')
return redirect("/finalize")
@app.route("/finalize")
def finalize():
with open(TEMP_DATA_FILE, "r", encoding="utf-8") as f: all_formations = json.load(f)
mapping_dict = {}
df_map = pd.read_csv(CSV_FILE, encoding='utf-8-sig').dropna(subset=['Nomenclature IGPDE'])
for _, row in df_map.iterrows():
mapping_dict[super_norm(str(row['Nomenclature IGPDE']))] = str(row['Nouvelle nomenclature']).strip()
for f in all_formations:
key = f.pop('Nom_IGPDE_Key', '')
f['Nouvelle nomenclature'] = mapping_dict.get(super_norm(key), mapping_dict.get(super_norm(f['Chapitre']), ""))
df = pd.DataFrame(all_formations).fillna('')
cols = ['Nouvelle nomenclature', 'Chapitre', 'Sous-chapitre', 'Titre', 'Page_Source'] + [c for c in df.columns if c not in ['Nouvelle nomenclature', 'Chapitre', 'Sous-chapitre', 'Titre', 'Page_Source']]
df_f = df[[c for c in cols if c in df.columns]]
df_f.to_excel(RESULT_FILE, index=False, engine='xlsxwriter')
return redirect("/view_result")
@app.route("/download")
def download(): return send_from_directory(TEMP_DIR, "resultat.xlsx", as_attachment=True)
@app.route("/download_db")
def download_db(): return send_from_directory("/app", "corresp_IGPDE.csv", as_attachment=True)
if __name__ == "__main__": app.run(host='0.0.0.0', port=5000)