diff --git a/corresp_IGPDE.csv b/corresp_IGPDE.csv new file mode 100644 index 0000000..b45a647 --- /dev/null +++ b/corresp_IGPDE.csv @@ -0,0 +1,101 @@ +Nouvelle nomenclature,Nomenclature IGPDE +"700 – Management, Leadership et Compétences Managériales", +701 - Management d'équipe,Management +703 - Communication managériale,Communication > Communication managériale +704 - Gestion des conflits et des situations difficiles,Médiation +705 - Accompagnement du changement, +706 - Intelligence collective et émotionnelle, +707 - Évaluation et entretien professionnel, +, +"500 – Ressources Humaines, Carrière, Prépa concours", +501 – Recrutement, +502 - Gestion de carrière et mobilité,Ressources humaines > Mobilité et transitions professionnelles +"503 - Diversité, inclusion et laïcité, lutte contre les discriminations","Ressources humaines > Diversité, laïcité et égalité professionnelle" +504 - Dialogue social et relations professionnelles, +505 - Accueil et intégration, +506 - RH à l’Insee : outils et métiers,Ressources humaines +507 - Préparation aux concours et examens, +508 - Jury, +509 – Formation initiale, +509 – Formation initiale, +, +100 – Méthodologie et production statistique, +101 - Enquêtes entreprises, +"102 - Enquêtes ménages (yc PCS, ISCO)", +103 - Enquêtes ménages – formations spécifiques enquêteurs, +104 – Répertoire Sirene, +105 – Recensement de la population, +"106 – Conjoncture, études économiques", +107 – Comptabilité nationale, +108 - Autres outils et logiciels d'enquête, +109 - Méthodologie d’enquête , +110 – Communication avec les enquêtés, +, +"200 – Statistique, Analyse de Données et Data Science", +201 - Statistique descriptive et analyse, +"202 - Programmation statistique (R, Python pour les statisticiens)","Big data, IA, DevOps > Langages de programmation" +203 - Visualisation de données (QGIS et autres), +204 - Séries temporelles et prévision, +, +"300 – Informatique, Numérique et Systèmes d'Information", +301 - Développement et programmation informatique,Informatique > Langages et Développement +301 - Développement et programmation informatique,"Big data, IA, DevOps" +302 - Bases de données et SQL,Informatique > Bases de données +303 - Outils bureautiques,Bureautique +304 - Sécurité informatique,Informatique > Environnement informatique et sécurité +305 - Accessibilité numérique, +306 - Infrastructure et administration système,Informatique > Modélisation de projets SI +306 - Infrastructure et administration système,Informatique > Réseaux +306 - Infrastructure et administration système,Informatique > Systèmes d’exploitation – Poste de travail +306 - Infrastructure et administration système,Informatique > Systèmes d’exploitation – Serveurs +307 - Agilité et gestion de projet,Informatique > Gestion de projet SI +308 - Eco-conception et Green IT, +309 – Intelligence artificielle,"Big data, IA, DevOps > Intelligence Artificielle" +310 – Autres formations au numérique,"Numérique, innovation, créativité" +, +"600 – Santé, Sécurité et Prévention des Risques","Ressources humaines > Qualité de vie, santé au travail et sécurité au travail" +"600 – Santé, Sécurité et Prévention des Risques", +, +"400 – Connaissance de l'Insee, des Régions, de l’État, des entreprises", +401 - Présentation de l'Insee et de ses missions, +402 - Connaissance des régions et des territoires, +"403 – Action publique, connaissance de l’administration",Action publique – Environnement administratif +403 – Europe et international,Europe – International +404 – Economie d’entreprise,Connaître et accompagner les entreprises +,"Comptabilité, gestion et analyse financière des entreprises" +, +"800 – Communication, Rédaction et Techniques Éditoriales", +801 - Communication écrite,Communication > Communication écrite +"802 - Communication orale, prise de parole",Communication > Communication orale +803 – Communication visuelle et digitale,Communication > Communication digitale +804 - Rédaction administrative et technique,Communication > +805 - Pédagogie des adultes, +, +"900 – Juridique, Réglementaire et Déontologie", +901 - Droit du travail et statut de la fonction publique, +"902 - Commande publique, marchés",Commande publique et achats publics +903 – Gestion publique,Gestion publique +904 - Déontologie et éthique, +905 - Accessibilité et RGAA, +907 – Droit – autres,Droit +908 – Archivage,Gestion et recherche documentaire +, +1000 – Transition Écologique et Développement Durable,Développement durable +, +"1100 – Langues, Compétences Transversales", +1101 - Langues,Langues +1102 - Soft skills et développement personnel,Efficacité professionnelle - Intelligence relationnelle et développement du potentiel créatif +, +1200 – Autres Formations Spécifiques ou Transversales, +1201 - Formations transversales, +1203 – Economie générale,Economie générale +1203 - Formations techniques diverses,Immobilier +1201 - Formations transversales,Événements & séminaires +507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF) +507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie A +507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie B +507 - Préparation aux concours et examens,Préparations aux concours et examens professionnels (agents MEF) > Préparations de catégorie C +507 - Préparation aux concours et examens,Préparations aux concours interministériels +507 - Préparation aux concours et examens,Qualifications informatiques (agents MEF) +507 - Préparation aux concours et examens,Qualifications informatiques (agents MEF) > Préparations de catégorie A - Qualifications informatiques +507 - Préparation aux concours et examens,Qualifications informatiques (agents MEF) > Préparations de catégorie B - Qualifications informatiques diff --git a/corresp_IGPDE.ods b/corresp_IGPDE.ods new file mode 100644 index 0000000..2714285 Binary files /dev/null and b/corresp_IGPDE.ods differ diff --git a/docker-compose.yml b/docker-compose.yml index de21978..39acb5e 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -4,4 +4,6 @@ services: container_name: extracteur-pdf restart: unless-stopped ports: - - "5050:5000" # Choix du port à définir en fonction des services déjà existants \ No newline at end of file + - "5050:5000" + volumes: + - .:/app diff --git a/requirements.txt b/requirements.txt index 12d2ec2..119b1e2 100644 --- a/requirements.txt +++ b/requirements.txt @@ -3,3 +3,4 @@ pdfplumber pandas openpyxl xlsxwriter +odfpy diff --git a/webapp.py b/webapp.py index 1a5da79..59f2cf6 100644 --- a/webapp.py +++ b/webapp.py @@ -1,277 +1,207 @@ import os import io +import json +import re import pandas as pd import pdfplumber -from flask import Flask, request, render_template_string, Response, send_from_directory +import unicodedata +import traceback +from flask import Flask, request, render_template_string, Response, send_from_directory, redirect app = Flask(__name__) -# Dossier temporaire pour stocker le fichier généré à l'intérieur du conteneur +# Dossiers et fichiers TEMP_DIR = "/app/downloads" -if not os.path.exists(TEMP_DIR): - os.makedirs(TEMP_DIR) +if not os.path.exists(TEMP_DIR): os.makedirs(TEMP_DIR) -# --- DESIGN HTML & JAVASCRIPT --- -# Ce bloc est envoyé en premier au navigateur pour préparer l'interface de log -HTML_LAYOUT = """ - - - - - - Extracteur de formations - - - -
-

Analyse du catalogue en cours...

-

Chaque page du fichier est analysée pour extraire les fiches de formation.

- -
- -
- > Initialisation du moteur d'extraction...
-
+TEMP_DATA_FILE = os.path.join(TEMP_DIR, "temp_data.json") +TEMP_MISSING_FILE = os.path.join(TEMP_DIR, "temp_missing.json") +CSV_FILE = "/app/corresp_IGPDE.csv" +ODS_FILE = "/app/corresp_IGPDE.ods" +RESULT_FILE = os.path.join(TEMP_DIR, "resultat.xlsx") - -
+# --- GESTION DE LA BASE DE DONNÉES --- +def ensure_database(): + if os.path.exists(CSV_FILE): return + if os.path.exists(ODS_FILE): + try: + df = pd.read_excel(ODS_FILE, engine="odf") + df.to_csv(CSV_FILE, index=False, encoding='utf-8-sig') + return + except: pass + df_empty = pd.DataFrame(columns=["Nouvelle nomenclature", "Nomenclature IGPDE"]) + df_empty.to_csv(CSV_FILE, index=False, encoding='utf-8-sig') - +# --- DESIGN CSS --- +COMMON_STYLE = """ + """ -# --- PAGE D'ACCUEIL --- +def get_nav(active_page): + h = "active" if active_page=="home" else "" + c = "active" if active_page=="corresp" else "" + r = "active" if active_page=="result" else "" + return f'' + @app.route("/") def home(): - return render_template_string(""" - - - - - - Extracteur de formations - - - -
-
📄
-

Extracteur de formations

-

Sélectionnez le catalogue de formations (PDF) pour extraire les données vers Excel.

- -
-
- - -
- - -
-
- - - - - """) - -# --- MOTEUR D'EXTRACTION --- @app.route("/process", methods=["POST"]) def process(): + ensure_database() file = request.files['pdf_file'] if not file: return "Fichier manquant" - - file_content = file.read() # On garde le fichier en mémoire vive + file_content = file.read() def generate(): - yield HTML_LAYOUT # Envoie l'interface au navigateur + yield f'{COMMON_STYLE}

Analyse en cours...

' all_formations = [] - COLONNES_ADMISES = ["Référence", "Durée", "Type", "Publics éligibles", "Niveau", "Sessions", "Tarifs"] - try: with pdfplumber.open(io.BytesIO(file_content)) as pdf: total_pages = len(pdf.pages) - yield f"" - for i, page in enumerate(pdf.pages): page_num = i + 1 - percent = int((page_num / total_pages) * 100) - + percent = int((page_num / total_pages) * 85) try: - # Zone de détection (haut de page) check_zone = page.crop((page.width * 0.5, 0, page.width, 400)) - text = check_zone.extract_text() or "" - - if "Référence" in text: - # --- EXTRACTION DE LA FICHE --- - header = page.crop((0, 0, page.width, 100)) - titre_brut = header.extract_text().split('\n')[0].strip() - # Sécurisation du titre pour JS (on enlève les quotes) - titre_clean = titre_brut.replace("'", "").replace('"', '').strip()[:60] + if "Référence" in (check_zone.extract_text() or ""): + bandeaux = sorted([r for r in page.rects if (r['x1'] - r['x0']) > page.width * 0.7 and r['top'] < 300 and 15 < r['height'] < 200], key=lambda r: r['top']) + if len(bandeaux) >= 2: + titre_brut = " ".join((page.crop((0, max(0, bandeaux[0]['top'] - 2), page.width, min(page.height, bandeaux[0]['bottom'] + 2))).extract_text() or "").replace('\n', ' ').split()) + categorie_str = " ".join((page.crop((0, max(0, bandeaux[1]['top'] - 2), page.width, min(page.height, bandeaux[1]['bottom'] + 2))).extract_text() or "").replace('\n', ' ').split()) + else: + header = page.crop((0, 60, page.width, 250)) + lines = [l.strip() for l in (header.extract_text() or "").split('\n') if l.strip() and not (len(l) > 90 or l.endswith('.'))] + titre_brut, categorie_str = (" ".join(lines[:-1]), lines[-1]) if len(lines) >= 2 else (lines[0] if lines else "", "") + + chapitre, sous_chap = (categorie_str.split(">")[0].strip(), categorie_str.split(">")[1].strip()) if ">" in categorie_str else (categorie_str.strip(), "") + data = {'Nom_IGPDE_Key': f"{chapitre} > {sous_chap}" if sous_chap else chapitre, 'Chapitre': chapitre, 'Sous-chapitre': sous_chap, 'Titre': titre_brut, 'Page_Source': page_num} - data = {'Titre': titre_brut, 'Page_Source': page_num} - - # Extraction du tableau à droite - right_side = page.crop((page.width * 0.6, 180, page.width, 600)) - table = right_side.extract_table() + table = page.crop((page.width * 0.6, 180, page.width, 600)).extract_table() if table: for row in table: if len(row) >= 2 and row[0]: cle = row[0].replace('\n', ' ').strip() - for col in COLONNES_ADMISES: + for col in ["Référence", "Durée", "Type", "Publics éligibles", "Niveau", "Sessions", "Tarifs"]: if col.lower() in cle.lower(): data[col] = row[1].replace('\n', ' ').strip() break - all_formations.append(data) - yield f"" - - else: - # --- PAGE IGNORÉE --- - yield f"" - - except Exception as page_err: - yield f"" + t_log = titre_brut[:40].replace("'", " ").replace('"', ' ') + yield f"" + except Exception: pass - # --- FIN DE BOUCLE : SAUVEGARDE EXCEL --- - yield f"" - - if all_formations: - df = pd.DataFrame(all_formations) - output_path = os.path.join(TEMP_DIR, "resultat.xlsx") - # On utilise xlsxwriter pour la stabilité - df.to_excel(output_path, index=False, engine='xlsxwriter') - yield f"" - yield "" - else: - yield f"" + # --- MAPPING --- + mapping_dict = {} + df_map = pd.read_csv(CSV_FILE, encoding='utf-8-sig') + for _, row in df_map.dropna(subset=['Nomenclature IGPDE']).iterrows(): + mapping_dict[super_norm(str(row['Nomenclature IGPDE']))] = str(row['Nouvelle nomenclature']).strip() - except Exception as global_err: - yield f"" + missing = [] + seen_missing_norm = set() + for f in all_formations: + full_norm, chap_norm = super_norm(f['Nom_IGPDE_Key']), super_norm(f['Chapitre']) + if full_norm not in mapping_dict and chap_norm not in mapping_dict: + if full_norm not in seen_missing_norm: + missing.append(f['Nom_IGPDE_Key']); seen_missing_norm.add(full_norm) + with open(TEMP_DATA_FILE, "w", encoding="utf-8") as f: json.dump(all_formations, f) + if missing: + with open(TEMP_MISSING_FILE, "w", encoding="utf-8") as f: json.dump(missing, f) + yield "" + else: + yield "" + except Exception as e: + traceback.print_exc() + yield f"" yield "" - return Response(generate(), mimetype='text/html') -# --- TÉLÉCHARGEMENT --- -@app.route("/download") -def download(): - return send_from_directory(TEMP_DIR, "resultat.xlsx", as_attachment=True) +@app.route("/ask_mapping") +def ask_mapping(): + with open(TEMP_MISSING_FILE, "r", encoding="utf-8") as f: missing = json.load(f) + df_db = pd.read_csv(CSV_FILE, encoding='utf-8-sig') + known = sorted(df_db['Nouvelle nomenclature'].dropna().unique().tolist()) + options_html = "".join([f'