Mise à jour à partir de dev
This commit is contained in:
parent
f39bbdeb7d
commit
e1e288f201
200
IA/00 - fiches_corpus/batch_generate_fiches.py
Normal file
200
IA/00 - fiches_corpus/batch_generate_fiches.py
Normal file
@ -0,0 +1,200 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import yaml
|
||||||
|
import requests
|
||||||
|
import argparse
|
||||||
|
import logging
|
||||||
|
|
||||||
|
# Import des fonctions de génération
|
||||||
|
from app.fiches.generer import (
|
||||||
|
generer_fiche
|
||||||
|
)
|
||||||
|
from app.fiches.utils.fiche_utils import load_seuils
|
||||||
|
from utils.gitea import charger_arborescence_fiches
|
||||||
|
from config import GITEA_TOKEN, FICHES_CRITICITE
|
||||||
|
|
||||||
|
# Configuration du logging
|
||||||
|
logging.basicConfig(
|
||||||
|
level=logging.INFO,
|
||||||
|
format='%(asctime)s - %(levelname)s - %(message)s',
|
||||||
|
handlers=[
|
||||||
|
logging.FileHandler("batch_generation.log"),
|
||||||
|
logging.StreamHandler()
|
||||||
|
]
|
||||||
|
)
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
def get_fiche_type(md_source):
|
||||||
|
"""Extrait le type de fiche depuis le frontmatter YAML."""
|
||||||
|
front_match = re.match(r"(?s)^---\n(.*?)\n---\n", md_source)
|
||||||
|
if not front_match:
|
||||||
|
return "autre"
|
||||||
|
|
||||||
|
context = yaml.safe_load(front_match.group(1))
|
||||||
|
return context.get("type_fiche", "autre")
|
||||||
|
|
||||||
|
def get_indice_court(md_source):
|
||||||
|
"""Extrait l'indice court depuis le frontmatter YAML."""
|
||||||
|
front_match = re.match(r"(?s)^---\n(.*?)\n---\n", md_source)
|
||||||
|
if not front_match:
|
||||||
|
return None
|
||||||
|
|
||||||
|
context = yaml.safe_load(front_match.group(1))
|
||||||
|
return context.get("indice_court")
|
||||||
|
|
||||||
|
def batch_generate_fiches(output_dir="", force_regenerate=False):
|
||||||
|
"""Génère toutes les fiches en batch en suivant un ordre de priorité."""
|
||||||
|
# Assurer que les répertoires de sortie existent
|
||||||
|
if output_dir:
|
||||||
|
os.makedirs(output_dir, exist_ok=True)
|
||||||
|
|
||||||
|
# Toujours créer les répertoires nécessaires
|
||||||
|
os.makedirs(os.path.join("Fiches"), exist_ok=True)
|
||||||
|
os.makedirs(os.path.join("HTML"), exist_ok=True)
|
||||||
|
os.makedirs(os.path.join("static", "Fiches"), exist_ok=True)
|
||||||
|
|
||||||
|
# Charger les seuils
|
||||||
|
try:
|
||||||
|
seuils = load_seuils("assets/config.yaml")
|
||||||
|
logger.info("Seuils chargés avec succès")
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur lors du chargement des seuils: {e}")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Charger l'arborescence des fiches
|
||||||
|
try:
|
||||||
|
arborescence = charger_arborescence_fiches()
|
||||||
|
logger.info(f"Arborescence chargée: {len(arborescence)} dossiers trouvés")
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur lors du chargement de l'arborescence: {e}")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Créer une liste de toutes les fiches avec leurs informations
|
||||||
|
toutes_fiches = []
|
||||||
|
for dossier, fiches in arborescence.items():
|
||||||
|
for fiche in fiches:
|
||||||
|
toutes_fiches.append({
|
||||||
|
"dossier": dossier,
|
||||||
|
"nom": fiche["nom"],
|
||||||
|
"download_url": fiche["download_url"],
|
||||||
|
"type": fiche.get("type", "autre")
|
||||||
|
})
|
||||||
|
|
||||||
|
logger.info(f"Total de {len(toutes_fiches)} fiches à traiter")
|
||||||
|
|
||||||
|
# Organisation des fiches par type pour respecter l'ordre de priorité
|
||||||
|
fiches_par_type = {
|
||||||
|
"criticite": [],
|
||||||
|
"assemblage": [],
|
||||||
|
"fabrication": [],
|
||||||
|
"minerai": [],
|
||||||
|
"autre": []
|
||||||
|
}
|
||||||
|
|
||||||
|
# Télécharger et catégoriser les fiches
|
||||||
|
headers = {"Authorization": f"token {GITEA_TOKEN}"}
|
||||||
|
|
||||||
|
# Création des listes spécifiques pour faciliter le traitement ordonné
|
||||||
|
fiches_criticite = []
|
||||||
|
autres_fiches = []
|
||||||
|
|
||||||
|
# Première étape : identifier les fiches de criticité
|
||||||
|
logger.info("Identification des fiches de criticité...")
|
||||||
|
for fiche in toutes_fiches:
|
||||||
|
# Vérifier si c'est une fiche de criticité par deux méthodes
|
||||||
|
est_criticite = False
|
||||||
|
|
||||||
|
# Méthode 1: vérifier par le nom du fichier
|
||||||
|
if fiche["nom"] in FICHES_CRITICITE:
|
||||||
|
est_criticite = True
|
||||||
|
logger.info(f"Fiche de criticité identifiée par nom: {fiche['nom']}")
|
||||||
|
|
||||||
|
# Méthode 2: vérifier par le dossier
|
||||||
|
elif fiche["dossier"] == "Criticités":
|
||||||
|
est_criticite = True
|
||||||
|
logger.info(f"Fiche de criticité identifiée par dossier: {fiche['nom']}")
|
||||||
|
|
||||||
|
if est_criticite:
|
||||||
|
fiches_criticite.append(fiche)
|
||||||
|
else:
|
||||||
|
autres_fiches.append(fiche)
|
||||||
|
|
||||||
|
# Traiter d'abord les fiches de criticité
|
||||||
|
logger.info(f"Traitement prioritaire des fiches de criticité ({len(fiches_criticite)} fiches)")
|
||||||
|
criticite_count = 0
|
||||||
|
|
||||||
|
for fiche in fiches_criticite:
|
||||||
|
try:
|
||||||
|
# Télécharger la fiche de criticité
|
||||||
|
reponse = requests.get(fiche["download_url"], headers=headers)
|
||||||
|
reponse.raise_for_status()
|
||||||
|
md_source = reponse.text
|
||||||
|
|
||||||
|
# Générer immédiatement la fiche de criticité
|
||||||
|
logger.info(f"Génération prioritaire de la fiche de criticité: {fiche['dossier']}/{fiche['nom']}")
|
||||||
|
generer_fiche(md_source, fiche["dossier"], fiche["nom"], seuils)
|
||||||
|
|
||||||
|
# Ajouter à la liste pour le décompte
|
||||||
|
fiches_par_type["criticite"].append(fiche)
|
||||||
|
criticite_count += 1
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur lors de la génération de la fiche de criticité {fiche['nom']}: {e}")
|
||||||
|
|
||||||
|
logger.info(f"{criticite_count} fiches de criticité générées")
|
||||||
|
|
||||||
|
# Maintenant catégoriser et traiter les fiches restantes (non-criticité)
|
||||||
|
for fiche in autres_fiches:
|
||||||
|
try:
|
||||||
|
# Télécharger le contenu pour déterminer le type
|
||||||
|
reponse = requests.get(fiche["download_url"], headers=headers)
|
||||||
|
reponse.raise_for_status()
|
||||||
|
md_source = reponse.text
|
||||||
|
|
||||||
|
# Déterminer le type
|
||||||
|
type_fiche = get_fiche_type(md_source)
|
||||||
|
if type_fiche in fiches_par_type:
|
||||||
|
fiches_par_type[type_fiche].append({**fiche, "md_source": md_source})
|
||||||
|
else:
|
||||||
|
fiches_par_type["autre"].append({**fiche, "md_source": md_source})
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur lors du traitement de {fiche['nom']}: {e}")
|
||||||
|
|
||||||
|
# Ordre de traitement pour les fiches restantes
|
||||||
|
ordre_types = ["assemblage", "fabrication", "minerai", "autre"]
|
||||||
|
|
||||||
|
# Générer les fiches restantes dans l'ordre spécifié
|
||||||
|
for type_fiche in ordre_types:
|
||||||
|
logger.info(f"Traitement des fiches de type '{type_fiche}' ({len(fiches_par_type[type_fiche])} fiches)")
|
||||||
|
|
||||||
|
for fiche in fiches_par_type[type_fiche]:
|
||||||
|
try:
|
||||||
|
md_source = fiche["md_source"]
|
||||||
|
dossier = fiche["dossier"]
|
||||||
|
nom_fichier = fiche["nom"]
|
||||||
|
|
||||||
|
# Vérifier si la génération est nécessaire
|
||||||
|
html_path = os.path.join("HTML", dossier, os.path.splitext(nom_fichier)[0] + ".html")
|
||||||
|
if force_regenerate or not os.path.exists(html_path):
|
||||||
|
logger.info(f"Génération de {dossier}/{nom_fichier}")
|
||||||
|
generer_fiche(md_source, dossier, nom_fichier, seuils)
|
||||||
|
else:
|
||||||
|
logger.info(f"Ignoré (déjà généré): {dossier}/{nom_fichier}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur lors de la génération de {fiche['nom']}: {e}")
|
||||||
|
|
||||||
|
logger.info("Génération batch terminée")
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description="Générateur batch de fiches")
|
||||||
|
parser.add_argument("--output", "-o", default="", help="Répertoire de sortie (laissez vide pour utiliser les dossiers de l'application)")
|
||||||
|
parser.add_argument("--force", "-f", action="store_true", help="Forcer la régénération de toutes les fiches")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
logger.info(f"Démarrage de la génération batch (output: {args.output if args.output else 'dossiers par défaut'}, force: {args.force})")
|
||||||
|
batch_generate_fiches(output_dir=args.output, force_regenerate=args.force)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
96
IA/00 - fiches_corpus/generate_corpus.py
Normal file
96
IA/00 - fiches_corpus/generate_corpus.py
Normal file
@ -0,0 +1,96 @@
|
|||||||
|
import os
|
||||||
|
import re
|
||||||
|
import shutil
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
EXCLUDE_DIRS = {"Local"}
|
||||||
|
MAX_SECTION_LENGTH = 1200 # non utilisé ici car découpe selon présence de ###
|
||||||
|
|
||||||
|
def slugify(text):
|
||||||
|
return re.sub(r'\W+', '-', text.strip()).strip('-').lower()
|
||||||
|
|
||||||
|
def split_markdown_sections_refined(content):
|
||||||
|
lines = content.splitlines()
|
||||||
|
sections = []
|
||||||
|
header_level_2 = None
|
||||||
|
section_lines = []
|
||||||
|
subsections = []
|
||||||
|
current_subsection = None
|
||||||
|
inside_section = False
|
||||||
|
|
||||||
|
for line in lines:
|
||||||
|
if line.startswith("## "):
|
||||||
|
if header_level_2:
|
||||||
|
if current_subsection:
|
||||||
|
subsections.append(current_subsection)
|
||||||
|
sections.append((header_level_2, section_lines, subsections))
|
||||||
|
section_lines, subsections = [], []
|
||||||
|
current_subsection = None
|
||||||
|
header_level_2 = line[3:].strip()
|
||||||
|
inside_section = True
|
||||||
|
elif line.startswith("### ") and inside_section:
|
||||||
|
if current_subsection:
|
||||||
|
subsections.append(current_subsection)
|
||||||
|
current_subsection = (line[4:].strip(), [])
|
||||||
|
elif inside_section:
|
||||||
|
if current_subsection:
|
||||||
|
current_subsection[1].append(line)
|
||||||
|
else:
|
||||||
|
section_lines.append(line)
|
||||||
|
|
||||||
|
if header_level_2:
|
||||||
|
if current_subsection:
|
||||||
|
subsections.append(current_subsection)
|
||||||
|
sections.append((header_level_2, section_lines, subsections))
|
||||||
|
return sections
|
||||||
|
|
||||||
|
def process_markdown_file(md_path, rel_output_dir):
|
||||||
|
with open(md_path, encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
sections = split_markdown_sections_refined(content)
|
||||||
|
|
||||||
|
for idx, (sec_title, sec_lines, subsections) in enumerate(sections):
|
||||||
|
base_name = f"{idx:02d}-{slugify(sec_title)}"
|
||||||
|
if subsections:
|
||||||
|
sec_dir = rel_output_dir / base_name
|
||||||
|
sec_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
with open(sec_dir / "_intro.md", "w", encoding="utf-8") as f_out:
|
||||||
|
f_out.write(f"## {sec_title}\n")
|
||||||
|
f_out.write("\n".join(sec_lines).strip())
|
||||||
|
for sub_idx, (sub_title, sub_lines) in enumerate(subsections):
|
||||||
|
sub_name = f"{sub_idx:02d}-{slugify(sub_title)}.md"
|
||||||
|
with open(sec_dir / sub_name, "w", encoding="utf-8") as f_out:
|
||||||
|
f_out.write(f"### {sub_title}\n")
|
||||||
|
f_out.write("\n".join(sub_lines).strip())
|
||||||
|
else:
|
||||||
|
with open(rel_output_dir / f"{base_name}.md", "w", encoding="utf-8") as f_out:
|
||||||
|
f_out.write(f"## {sec_title}\n")
|
||||||
|
f_out.write("\n".join(sec_lines).strip())
|
||||||
|
|
||||||
|
def build_corpus_structure():
|
||||||
|
BASE_DIR = Path(__file__).resolve().parent.parent.parent
|
||||||
|
print(BASE_DIR)
|
||||||
|
SOURCE_DIR = BASE_DIR / "Fiches"
|
||||||
|
DEST_DIR = BASE_DIR / "Corpus"
|
||||||
|
|
||||||
|
if DEST_DIR.exists():
|
||||||
|
shutil.rmtree(DEST_DIR)
|
||||||
|
DEST_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
for root, _, files in os.walk(SOURCE_DIR):
|
||||||
|
rel_path = Path(root).relative_to(SOURCE_DIR)
|
||||||
|
if any(part in EXCLUDE_DIRS for part in rel_path.parts):
|
||||||
|
continue
|
||||||
|
for file in files:
|
||||||
|
if not file.endswith(".md") or ".md." in file:
|
||||||
|
continue
|
||||||
|
input_file = Path(root) / file
|
||||||
|
subdir = rel_path
|
||||||
|
filename_no_ext = Path(file).stem
|
||||||
|
output_dir = DEST_DIR / subdir / filename_no_ext
|
||||||
|
output_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
process_markdown_file(input_file, output_dir)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
build_corpus_structure()
|
||||||
|
print("✅ Corpus généré avec succès dans le dossier 'Corpus/'")
|
||||||
209
IA/01 - corpus_rapport_factuel/analyze_graph.py
Normal file
209
IA/01 - corpus_rapport_factuel/analyze_graph.py
Normal file
@ -0,0 +1,209 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
"""
|
||||||
|
Script d'analyse de la structure du graphe DOT pour comprendre
|
||||||
|
comment intégrer l'ISG dans le générateur de template.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
from networkx.drawing.nx_agraph import read_dot
|
||||||
|
|
||||||
|
# Chemins
|
||||||
|
BASE_DIR = Path(__file__).resolve().parent
|
||||||
|
GRAPH_PATH = BASE_DIR / "graphe.dot"
|
||||||
|
|
||||||
|
def analyze_graph_structure(dot_path):
|
||||||
|
"""Analyse la structure du graphe et affiche ses caractéristiques."""
|
||||||
|
print(f"Analyse du fichier: {dot_path}")
|
||||||
|
|
||||||
|
# Lire le graphe
|
||||||
|
G = read_dot(dot_path)
|
||||||
|
|
||||||
|
# Informations de base
|
||||||
|
print(f"Nombre total de nœuds: {len(G.nodes())}")
|
||||||
|
print(f"Nombre total d'arêtes: {len(G.edges())}")
|
||||||
|
|
||||||
|
# Analyse des attributs des nœuds
|
||||||
|
node_attrs = {}
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
for key in attrs:
|
||||||
|
if key not in node_attrs:
|
||||||
|
node_attrs[key] = set()
|
||||||
|
node_attrs[key].add(attrs[key])
|
||||||
|
|
||||||
|
print("\nAttributs des nœuds:")
|
||||||
|
for attr, values in node_attrs.items():
|
||||||
|
print(f"- {attr}: {len(values)} valeurs différentes")
|
||||||
|
if len(values) < 20: # Afficher seulement si le nombre de valeurs est raisonnable
|
||||||
|
print(f" Valeurs: {', '.join(sorted(values))}")
|
||||||
|
|
||||||
|
# Analyse des niveaux (si l'attribut existe)
|
||||||
|
if 'level' in node_attrs:
|
||||||
|
print("\nAnalyse par niveau:")
|
||||||
|
levels = {}
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
if 'level' in attrs:
|
||||||
|
level = attrs['level']
|
||||||
|
if level not in levels:
|
||||||
|
levels[level] = []
|
||||||
|
levels[level].append(node)
|
||||||
|
|
||||||
|
for level, nodes in sorted(levels.items()):
|
||||||
|
print(f"- Niveau {level}: {len(nodes)} nœuds")
|
||||||
|
# Afficher quelques exemples
|
||||||
|
if len(nodes) < 5:
|
||||||
|
print(f" Exemples: {', '.join(nodes)}")
|
||||||
|
else:
|
||||||
|
print(f" Exemples: {', '.join(nodes[:3])}... (et {len(nodes)-3} autres)")
|
||||||
|
|
||||||
|
# Analyse des attributs ISG
|
||||||
|
print("\nRecherche des attributs ISG:")
|
||||||
|
isg_nodes = []
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
if 'isg' in attrs:
|
||||||
|
isg_nodes.append((node, attrs['isg']))
|
||||||
|
|
||||||
|
if isg_nodes:
|
||||||
|
print(f"- {len(isg_nodes)} nœuds avec attribut ISG")
|
||||||
|
print(" Exemples:")
|
||||||
|
for node, isg in isg_nodes[:5]:
|
||||||
|
print(f" - {node}: ISG = {isg}")
|
||||||
|
else:
|
||||||
|
print("- Aucun nœud avec attribut ISG trouvé")
|
||||||
|
|
||||||
|
# Analyse des connexions pour les nœuds critiques (IHH)
|
||||||
|
print("\nAnalyse des nœuds avec IHH:")
|
||||||
|
ihh_nodes = []
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
if 'ihh_pays' in attrs or 'ihh_acteurs' in attrs:
|
||||||
|
ihh_value_pays = attrs.get('ihh_pays', 'N/A')
|
||||||
|
ihh_value_acteurs = attrs.get('ihh_acteurs', 'N/A')
|
||||||
|
ihh_nodes.append((node, ihh_value_pays, ihh_value_acteurs))
|
||||||
|
|
||||||
|
if ihh_nodes:
|
||||||
|
print(f"- {len(ihh_nodes)} nœuds avec attributs IHH")
|
||||||
|
print(" Exemples:")
|
||||||
|
for node, ihh_pays, ihh_acteurs in ihh_nodes[:5]:
|
||||||
|
print(f" - {node}: IHH pays = {ihh_pays}, IHH acteurs = {ihh_acteurs}")
|
||||||
|
|
||||||
|
# Analyser les connexions de ce nœud
|
||||||
|
print(f" Connexions sortantes:")
|
||||||
|
out_edges = list(G.out_edges(node))
|
||||||
|
if out_edges:
|
||||||
|
for i, (_, target) in enumerate(out_edges[:3]):
|
||||||
|
print(f" - Vers {target}")
|
||||||
|
if len(out_edges) > 3:
|
||||||
|
print(f" - ... et {len(out_edges)-3} autres")
|
||||||
|
else:
|
||||||
|
print(" - Aucune connexion sortante")
|
||||||
|
|
||||||
|
print(f" Connexions entrantes:")
|
||||||
|
in_edges = list(G.in_edges(node))
|
||||||
|
if in_edges:
|
||||||
|
for i, (source, _) in enumerate(in_edges[:3]):
|
||||||
|
print(f" - Depuis {source}")
|
||||||
|
if len(in_edges) > 3:
|
||||||
|
print(f" - ... et {len(in_edges)-3} autres")
|
||||||
|
else:
|
||||||
|
print(" - Aucune connexion entrante")
|
||||||
|
else:
|
||||||
|
print("- Aucun nœud avec attributs IHH trouvé")
|
||||||
|
|
||||||
|
# Vérifier si un nœud a un attribut de niveau 99 (ISG supposé)
|
||||||
|
print("\nRecherche des nœuds de niveau 99 (ISG):")
|
||||||
|
level_99_nodes = []
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
if attrs.get('level') == '99':
|
||||||
|
level_99_nodes.append(node)
|
||||||
|
|
||||||
|
if level_99_nodes:
|
||||||
|
print(f"- {len(level_99_nodes)} nœuds de niveau 99")
|
||||||
|
print(" Exemples:")
|
||||||
|
for node in level_99_nodes[:5]:
|
||||||
|
print(f" - {node}")
|
||||||
|
# Analyser les connexions de ce nœud
|
||||||
|
print(f" Connexions entrantes:")
|
||||||
|
in_edges = list(G.in_edges(node))
|
||||||
|
if in_edges:
|
||||||
|
for i, (source, _) in enumerate(in_edges[:3]):
|
||||||
|
print(f" - Depuis {source}")
|
||||||
|
if len(in_edges) > 3:
|
||||||
|
print(f" - ... et {len(in_edges)-3} autres")
|
||||||
|
else:
|
||||||
|
print(" - Aucune connexion entrante")
|
||||||
|
else:
|
||||||
|
print("- Aucun nœud de niveau 99 trouvé")
|
||||||
|
|
||||||
|
def check_isg_paths(dot_path):
|
||||||
|
"""Vérifie les chemins entre les nœuds critiques (IHH) et les nœuds ISG."""
|
||||||
|
print("\nAnalyse des chemins entre nœuds IHH et nœuds ISG:")
|
||||||
|
|
||||||
|
# Lire le graphe
|
||||||
|
G = read_dot(dot_path)
|
||||||
|
|
||||||
|
# Identifier les nœuds avec IHH
|
||||||
|
ihh_nodes = []
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
ihh_pays = attrs.get('ihh_pays', 0)
|
||||||
|
ihh_acteurs = attrs.get('ihh_acteurs', 0)
|
||||||
|
try:
|
||||||
|
ihh_pays = float(ihh_pays)
|
||||||
|
ihh_acteurs = float(ihh_acteurs)
|
||||||
|
if ihh_pays > 25 or ihh_acteurs > 25: # Seuil critique
|
||||||
|
ihh_nodes.append(node)
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
if not ihh_nodes:
|
||||||
|
print("- Aucun nœud IHH critique trouvé")
|
||||||
|
return
|
||||||
|
|
||||||
|
print(f"- {len(ihh_nodes)} nœuds IHH critiques identifiés")
|
||||||
|
|
||||||
|
# Pour chaque nœud IHH critique, chercher des chemins vers des nœuds ISG
|
||||||
|
for node in ihh_nodes[:5]: # Limiter à 5 exemples
|
||||||
|
print(f"\n Analyse des chemins pour {node}:")
|
||||||
|
|
||||||
|
# Analyser les voisins directs
|
||||||
|
successors = list(G.successors(node))
|
||||||
|
print(f" - {len(successors)} successeurs directs")
|
||||||
|
|
||||||
|
if successors:
|
||||||
|
for succ in successors[:3]:
|
||||||
|
print(f" - Vers {succ}")
|
||||||
|
|
||||||
|
# Vérifier les attributs de ce successeur
|
||||||
|
succ_attrs = G.nodes[succ]
|
||||||
|
print(f" Attributs: {', '.join(f'{k}={v}' for k, v in succ_attrs.items() if k in ['level', 'isg'])}")
|
||||||
|
|
||||||
|
# Chercher les successeurs de niveau 2
|
||||||
|
succ2 = list(G.successors(succ))
|
||||||
|
print(f" {len(succ2)} successeurs de niveau 2")
|
||||||
|
|
||||||
|
if succ2:
|
||||||
|
for s2 in succ2[:2]:
|
||||||
|
print(f" - Vers {s2}")
|
||||||
|
s2_attrs = G.nodes[s2]
|
||||||
|
print(f" Attributs: {', '.join(f'{k}={v}' for k, v in s2_attrs.items() if k in ['level', 'isg'])}")
|
||||||
|
|
||||||
|
# Chercher encore plus loin si nécessaire
|
||||||
|
succ3 = list(G.successors(s2))
|
||||||
|
print(f" {len(succ3)} successeurs de niveau 3")
|
||||||
|
|
||||||
|
if succ3:
|
||||||
|
for s3 in succ3[:2]:
|
||||||
|
print(f" - Vers {s3}")
|
||||||
|
s3_attrs = G.nodes[s3]
|
||||||
|
print(f" Attributs: {', '.join(f'{k}={v}' for k, v in s3_attrs.items() if k in ['level', 'isg'])}")
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""Fonction principale."""
|
||||||
|
print("=== Analyse de la structure du graphe ===")
|
||||||
|
analyze_graph_structure(GRAPH_PATH)
|
||||||
|
|
||||||
|
print("\n=== Analyse des chemins entre nœuds critiques et ISG ===")
|
||||||
|
check_isg_paths(GRAPH_PATH)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
146
IA/01 - corpus_rapport_factuel/check_paths.py
Normal file
146
IA/01 - corpus_rapport_factuel/check_paths.py
Normal file
@ -0,0 +1,146 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from collections import defaultdict
|
||||||
|
|
||||||
|
def extract_paths(file_path):
|
||||||
|
"""Extrait tous les chemins du fichier rapport_template.md"""
|
||||||
|
paths = []
|
||||||
|
try:
|
||||||
|
with open(file_path, 'r', encoding='utf-8') as f:
|
||||||
|
for line in f:
|
||||||
|
# Extraire les lignes qui commencent par "Corpus/"
|
||||||
|
if line.strip().startswith("Corpus/"):
|
||||||
|
paths.append(line.strip())
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Erreur lors de la lecture du fichier {file_path}: {e}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
return paths
|
||||||
|
|
||||||
|
def check_paths(paths, base_dir):
|
||||||
|
"""Vérifie si les chemins existent dans le système de fichiers"""
|
||||||
|
results = {
|
||||||
|
"existing": [],
|
||||||
|
"missing": [],
|
||||||
|
"problematic": [] # Chemins qui pourraient nécessiter des corrections
|
||||||
|
}
|
||||||
|
|
||||||
|
for path in paths:
|
||||||
|
# Vérifier si le chemin est absolu ou relatif
|
||||||
|
abs_path = os.path.join(base_dir, path)
|
||||||
|
|
||||||
|
if os.path.exists(abs_path):
|
||||||
|
results["existing"].append(path)
|
||||||
|
else:
|
||||||
|
# Essayer de détecter des problèmes potentiels
|
||||||
|
problem_detected = False
|
||||||
|
|
||||||
|
# Vérifier les chemins avec "Fiche minerai" ou "Fiche fabrication"
|
||||||
|
if "Fiche minerai" in path or "Fiche fabrication" in path:
|
||||||
|
# Problème courant: mauvaise casse ou absence du mot "minerai"
|
||||||
|
path_lower = path.lower()
|
||||||
|
if "minerai" not in path_lower and "/minerai/" in path_lower:
|
||||||
|
corrected_path = path.replace("/Fiche ", "/Fiche minerai ")
|
||||||
|
if os.path.exists(os.path.join(base_dir, corrected_path)):
|
||||||
|
results["problematic"].append((path, corrected_path, "Mot 'minerai' manquant"))
|
||||||
|
problem_detected = True
|
||||||
|
|
||||||
|
# Vérifier les chemins SSD
|
||||||
|
if "SSD25" in path:
|
||||||
|
corrected_path = path.replace("SSD25", "SSD 2.5")
|
||||||
|
if os.path.exists(os.path.join(base_dir, corrected_path)):
|
||||||
|
results["problematic"].append((path, corrected_path, "Format 'SSD25' au lieu de 'SSD 2.5'"))
|
||||||
|
problem_detected = True
|
||||||
|
|
||||||
|
# Si aucun problème spécifique n'a été détecté, marquer comme manquant
|
||||||
|
if not problem_detected:
|
||||||
|
results["missing"].append(path)
|
||||||
|
|
||||||
|
return results
|
||||||
|
|
||||||
|
def find_similar_paths(missing_path, base_dir):
|
||||||
|
"""Essaie de trouver des chemins similaires pour aider à diagnostiquer le problème"""
|
||||||
|
missing_parts = missing_path.split('/')
|
||||||
|
similar_paths = []
|
||||||
|
|
||||||
|
# Rechercher dans les sous-répertoires correspondants
|
||||||
|
search_dir = os.path.join(base_dir, *missing_parts[:-1])
|
||||||
|
if os.path.exists(search_dir):
|
||||||
|
for file in os.listdir(search_dir):
|
||||||
|
if file.endswith('.md'):
|
||||||
|
similar_path = os.path.join(search_dir, file).replace(base_dir + '/', '')
|
||||||
|
similar_paths.append(similar_path)
|
||||||
|
|
||||||
|
# Si aucun chemin similaire n'est trouvé, remonter d'un niveau
|
||||||
|
if not similar_paths and len(missing_parts) > 2:
|
||||||
|
parent_dir = os.path.join(base_dir, *missing_parts[:-2])
|
||||||
|
if os.path.exists(parent_dir):
|
||||||
|
for dir_name in os.listdir(parent_dir):
|
||||||
|
if dir_name.lower() in missing_parts[-2].lower():
|
||||||
|
dir_path = os.path.join(parent_dir, dir_name)
|
||||||
|
if os.path.isdir(dir_path):
|
||||||
|
for file in os.listdir(dir_path):
|
||||||
|
if file.endswith('.md'):
|
||||||
|
similar_path = os.path.join(dir_path, file).replace(base_dir + '/', '')
|
||||||
|
similar_paths.append(similar_path)
|
||||||
|
|
||||||
|
return similar_paths
|
||||||
|
|
||||||
|
def main():
|
||||||
|
# Vérifier que nous sommes dans le bon répertoire
|
||||||
|
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
base_dir = script_dir
|
||||||
|
|
||||||
|
# Chemin vers le rapport_template.md
|
||||||
|
template_path = os.path.join(base_dir, "Corpus", "rapport_template.md")
|
||||||
|
|
||||||
|
if not os.path.exists(template_path):
|
||||||
|
print(f"Erreur: Le fichier {template_path} n'existe pas.")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
print("=== Vérification des chemins dans rapport_template.md ===")
|
||||||
|
|
||||||
|
# Extraire les chemins
|
||||||
|
paths = extract_paths(template_path)
|
||||||
|
print(f"Nombre total de chemins trouvés: {len(paths)}")
|
||||||
|
|
||||||
|
# Vérifier les chemins
|
||||||
|
results = check_paths(paths, base_dir)
|
||||||
|
|
||||||
|
# Afficher les résultats
|
||||||
|
print("\n=== Résultats ===")
|
||||||
|
print(f"Chemins existants: {len(results['existing'])}")
|
||||||
|
print(f"Chemins manquants: {len(results['missing'])}")
|
||||||
|
print(f"Chemins problématiques: {len(results['problematic'])}")
|
||||||
|
|
||||||
|
# Afficher les chemins manquants
|
||||||
|
if results["missing"]:
|
||||||
|
print("\n=== Chemins manquants ===")
|
||||||
|
for path in results["missing"]:
|
||||||
|
print(f"- {path}")
|
||||||
|
similar = find_similar_paths(path, base_dir)
|
||||||
|
if similar:
|
||||||
|
print(" Chemins similaires trouvés:")
|
||||||
|
for sim_path in similar[:3]: # Limiter à 3 suggestions
|
||||||
|
print(f" * {sim_path}")
|
||||||
|
|
||||||
|
# Afficher les chemins problématiques avec suggestions
|
||||||
|
if results["problematic"]:
|
||||||
|
print("\n=== Chemins problématiques ===")
|
||||||
|
for orig, corrected, reason in results["problematic"]:
|
||||||
|
print(f"- {orig}")
|
||||||
|
print(f" Suggestion: {corrected}")
|
||||||
|
print(f" Raison: {reason}")
|
||||||
|
|
||||||
|
# Résumé
|
||||||
|
if not results["missing"] and not results["problematic"]:
|
||||||
|
print("\nTous les chemins dans le rapport sont valides !")
|
||||||
|
else:
|
||||||
|
print("\nDes chemins problématiques ont été détectés. Veuillez corriger les erreurs.")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
1267
IA/01 - corpus_rapport_factuel/generate_template.py
Normal file
1267
IA/01 - corpus_rapport_factuel/generate_template.py
Normal file
File diff suppressed because it is too large
Load Diff
147
IA/01 - corpus_rapport_factuel/replace_paths.py
Normal file
147
IA/01 - corpus_rapport_factuel/replace_paths.py
Normal file
@ -0,0 +1,147 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
"""
|
||||||
|
Script pour remplacer les références de chemins dans le rapport par le contenu des fichiers.
|
||||||
|
Ajuste automatiquement les niveaux de titres pour maintenir la hiérarchie.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
# Chemins de base
|
||||||
|
BASE_DIR = Path(__file__).resolve().parent
|
||||||
|
BASE_DIR = BASE_DIR / ".."
|
||||||
|
CORPUS_DIR = BASE_DIR / "Corpus"
|
||||||
|
INPUT_PATH = CORPUS_DIR / "rapport_template.md"
|
||||||
|
OUTPUT_PATH = CORPUS_DIR / "rapport_final.md"
|
||||||
|
|
||||||
|
def determine_heading_level(line):
|
||||||
|
"""Détermine le niveau de titre d'une ligne."""
|
||||||
|
match = re.match(r'^(#+)\s+', line)
|
||||||
|
if match:
|
||||||
|
return len(match.group(1))
|
||||||
|
return 0
|
||||||
|
|
||||||
|
def determine_parent_level(lines, current_index):
|
||||||
|
"""Détermine le niveau de titre parent pour une ligne donnée."""
|
||||||
|
# Remonter dans les lignes précédentes pour trouver le titre parent
|
||||||
|
for i in range(current_index - 1, -1, -1):
|
||||||
|
level = determine_heading_level(lines[i])
|
||||||
|
if level > 0:
|
||||||
|
return level
|
||||||
|
return 0
|
||||||
|
|
||||||
|
def adjust_heading_levels(content, parent_level, is_intro_file=False, is_ivc_section=False):
|
||||||
|
"""Ajuste les niveaux de titres dans le contenu pour s'adapter à la hiérarchie."""
|
||||||
|
lines = content.split('\n')
|
||||||
|
|
||||||
|
# Si le contenu est vide, retourner une chaîne vide
|
||||||
|
if not lines:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
# Déterminer le niveau minimum de titre dans le contenu original
|
||||||
|
min_level = 10
|
||||||
|
for line in lines:
|
||||||
|
level = determine_heading_level(line)
|
||||||
|
if level > 0 and level < min_level:
|
||||||
|
min_level = level
|
||||||
|
|
||||||
|
# Si aucun titre trouvé, simplement supprimer la première ligne si nécessaire
|
||||||
|
if min_level == 10:
|
||||||
|
if not is_intro_file and not is_ivc_section and lines:
|
||||||
|
return '\n'.join(lines[1:])
|
||||||
|
return content
|
||||||
|
|
||||||
|
# Traitement spécial pour les fichiers IVC et intro
|
||||||
|
if is_ivc_section or is_intro_file:
|
||||||
|
adjusted_lines = []
|
||||||
|
# Pour les fichiers IVC ou intro, on garde toutes les lignes mais on ajuste les niveaux des titres
|
||||||
|
for line in lines:
|
||||||
|
level = determine_heading_level(line)
|
||||||
|
if level > 0:
|
||||||
|
# Nouveau niveau = niveau parent + 1 + (niveau actuel - min_level)
|
||||||
|
new_level = parent_level + 1 + (level - min_level)
|
||||||
|
# S'assurer que le niveau ne dépasse pas 6 (limite en markdown)
|
||||||
|
new_level = min(new_level, 6)
|
||||||
|
line = re.sub(r'^#+\s+', '#' * new_level + ' ', line)
|
||||||
|
adjusted_lines.append(line)
|
||||||
|
else:
|
||||||
|
# Pour les fichiers standards, on supprime la première ligne
|
||||||
|
lines = lines[1:]
|
||||||
|
adjusted_lines = []
|
||||||
|
# Ajuster les niveaux de titres pour les lignes restantes
|
||||||
|
for line in lines:
|
||||||
|
level = determine_heading_level(line)
|
||||||
|
if level > 0:
|
||||||
|
# Nouveau niveau = niveau parent + 1 + (niveau actuel - min_level)
|
||||||
|
new_level = parent_level + 1 + (level - min_level)
|
||||||
|
# S'assurer que le niveau ne dépasse pas 6 (limite en markdown)
|
||||||
|
new_level = min(new_level, 6)
|
||||||
|
line = re.sub(r'^#+\s+', '#' * new_level + ' ', line)
|
||||||
|
adjusted_lines.append(line)
|
||||||
|
|
||||||
|
return '\n'.join(adjusted_lines)
|
||||||
|
|
||||||
|
def process_report():
|
||||||
|
"""Traite le rapport pour remplacer les chemins par le contenu."""
|
||||||
|
if not os.path.exists(INPUT_PATH):
|
||||||
|
print(f"Fichier d'entrée introuvable: {INPUT_PATH}")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Lire le rapport
|
||||||
|
with open(INPUT_PATH, 'r', encoding='utf-8') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
|
||||||
|
output_lines = []
|
||||||
|
i = 0
|
||||||
|
while i < len(lines):
|
||||||
|
line = lines[i].strip()
|
||||||
|
|
||||||
|
# Vérifier si la ligne est un chemin
|
||||||
|
if line.startswith('Corpus/'):
|
||||||
|
path = line
|
||||||
|
full_path = BASE_DIR / path
|
||||||
|
|
||||||
|
# Déterminer le niveau de titre parent
|
||||||
|
parent_level = determine_parent_level(lines, i)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Lire le contenu du fichier
|
||||||
|
if os.path.exists(full_path):
|
||||||
|
# Vérifier si c'est un fichier _intro.md
|
||||||
|
is_intro_file = os.path.basename(full_path) == "_intro.md"
|
||||||
|
|
||||||
|
# Vérifier si c'est une section d'Indice de Vulnérabilité de Concurrence
|
||||||
|
is_ivc_section = "Vulnérabilité de Concurrence" in line or "/ivc-" in path.lower() or "/fiche technique ivc/" in path.lower()
|
||||||
|
|
||||||
|
with open(full_path, 'r', encoding='utf-8') as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Ajuster les niveaux de titres
|
||||||
|
adjusted_content = adjust_heading_levels(content, parent_level, is_intro_file, is_ivc_section)
|
||||||
|
|
||||||
|
# Ajouter le contenu ajusté
|
||||||
|
output_lines.append(f"<!-- Contenu du fichier {path} -->")
|
||||||
|
output_lines.append(adjusted_content)
|
||||||
|
output_lines.append(f"<!-- Fin du contenu de {path} -->")
|
||||||
|
else:
|
||||||
|
output_lines.append(f"<!-- Fichier non trouvé: {path} -->")
|
||||||
|
output_lines.append(line)
|
||||||
|
except Exception as e:
|
||||||
|
output_lines.append(f"<!-- Erreur lors de la lecture du fichier {path}: {str(e)} -->")
|
||||||
|
output_lines.append(line)
|
||||||
|
else:
|
||||||
|
# Conserver la ligne telle quelle
|
||||||
|
output_lines.append(line)
|
||||||
|
|
||||||
|
i += 1
|
||||||
|
|
||||||
|
# Écrire le rapport final
|
||||||
|
with open(OUTPUT_PATH, 'w', encoding='utf-8') as f:
|
||||||
|
f.write('\n'.join(output_lines))
|
||||||
|
|
||||||
|
print(f"Rapport final généré: {OUTPUT_PATH}")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
process_report()
|
||||||
90
IA/02 - injection_fiches/README.md
Normal file
90
IA/02 - injection_fiches/README.md
Normal file
@ -0,0 +1,90 @@
|
|||||||
|
# Script d'Injection Automatique pour PrivateGPT
|
||||||
|
|
||||||
|
Ce script permet d'automatiser l'injection de documents dans PrivateGPT à partir d'un répertoire local. Au lieu d'utiliser l'interface utilisateur pour télécharger les fichiers un par un, vous pouvez injecter un dossier entier en une seule commande.
|
||||||
|
|
||||||
|
## Prérequis
|
||||||
|
|
||||||
|
- Python 3.7 ou supérieur
|
||||||
|
- PrivateGPT installé et fonctionnel sous Docker
|
||||||
|
- Accès à l'API REST de PrivateGPT (port 8001 par défaut)
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
1. Clonez ce dépôt ou téléchargez les fichiers dans un dossier
|
||||||
|
|
||||||
|
2. Installez les dépendances nécessaires :
|
||||||
|
```bash
|
||||||
|
pip install requests
|
||||||
|
```
|
||||||
|
|
||||||
|
## Utilisation
|
||||||
|
|
||||||
|
Le script s'utilise en ligne de commande avec différentes options :
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python auto_ingest.py -d REPERTOIRE [-u URL] [-r] [-t THREADS] [--retry RETRY] [--retry-delay RETRY_DELAY] [--timeout TIMEOUT] [--extensions EXT1 EXT2 ...]
|
||||||
|
```
|
||||||
|
|
||||||
|
### Options
|
||||||
|
|
||||||
|
- `-d, --directory` : Chemin du répertoire contenant les fichiers à injecter (obligatoire)
|
||||||
|
- `-u, --url` : URL de l'API PrivateGPT (défaut: http://localhost:8001)
|
||||||
|
- `-r, --recursive` : Parcourir récursivement les sous-répertoires
|
||||||
|
- `-t, --threads` : Nombre de threads pour les injections parallèles (défaut: 5)
|
||||||
|
- `--retry` : Nombre de tentatives en cas d'échec (défaut: 3)
|
||||||
|
- `--retry-delay` : Délai entre les tentatives en secondes (défaut: 5)
|
||||||
|
- `--timeout` : Délai d'attente pour chaque requête en secondes (défaut: 300)
|
||||||
|
- `--extensions` : Liste d'extensions spécifiques à injecter (ex: pdf txt)
|
||||||
|
|
||||||
|
## Exemples d'utilisation
|
||||||
|
|
||||||
|
### Injection simple d'un répertoire
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python auto_ingest.py -d /chemin/vers/documents
|
||||||
|
```
|
||||||
|
|
||||||
|
### Injection récursive avec extensions spécifiques
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python auto_ingest.py -d /chemin/vers/documents -r --extensions pdf docx txt
|
||||||
|
```
|
||||||
|
|
||||||
|
### Injection avec paramètres avancés
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python auto_ingest.py -d /chemin/vers/documents -r -t 10 --timeout 600 --retry 5
|
||||||
|
```
|
||||||
|
|
||||||
|
## Formats de fichiers supportés
|
||||||
|
|
||||||
|
Par défaut, le script reconnaît et traite les formats suivants :
|
||||||
|
- PDF (.pdf)
|
||||||
|
- Documents texte (.txt, .md)
|
||||||
|
- Documents Microsoft Office (.doc, .docx, .ppt, .pptx, .xls, .xlsx)
|
||||||
|
- CSV (.csv)
|
||||||
|
- EPUB (.epub)
|
||||||
|
- HTML (.html, .htm)
|
||||||
|
|
||||||
|
## Résolution des problèmes
|
||||||
|
|
||||||
|
### Erreur de connexion
|
||||||
|
|
||||||
|
Si vous obtenez des erreurs de connexion, vérifiez que :
|
||||||
|
1. PrivateGPT est bien en cours d'exécution
|
||||||
|
2. L'URL est correcte (par défaut: http://localhost:8001)
|
||||||
|
3. Le port 8001 est accessible et n'est pas bloqué par un pare-feu
|
||||||
|
|
||||||
|
### Erreurs d'injection
|
||||||
|
|
||||||
|
- Si un fichier spécifique ne peut pas être injecté, vérifiez qu'il est d'un format supporté par PrivateGPT
|
||||||
|
- Pour les fichiers volumineux, vous pouvez augmenter la valeur de `--timeout`
|
||||||
|
- En cas d'erreurs répétées, augmentez les valeurs de `--retry` et `--retry-delay`
|
||||||
|
|
||||||
|
## Logs
|
||||||
|
|
||||||
|
Le script génère des logs dans :
|
||||||
|
- La console (stdout)
|
||||||
|
- Un fichier pgpt_auto_ingest.log dans le répertoire courant
|
||||||
|
|
||||||
|
Ces logs contiennent des informations détaillées sur le processus d'injection.
|
||||||
254
IA/02 - injection_fiches/auto_ingest.py
Executable file
254
IA/02 - injection_fiches/auto_ingest.py
Executable file
@ -0,0 +1,254 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
"""
|
||||||
|
Script d'injection automatique de documents pour PrivateGPT
|
||||||
|
|
||||||
|
Ce script parcourt un répertoire spécifié et injecte tous les fichiers
|
||||||
|
compatibles dans PrivateGPT via son API REST.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
import argparse
|
||||||
|
import logging
|
||||||
|
import requests
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import List, Dict, Tuple, Set
|
||||||
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
|
|
||||||
|
# Configuration du logging
|
||||||
|
logging.basicConfig(
|
||||||
|
level=logging.INFO,
|
||||||
|
format='%(asctime)s - %(levelname)s - %(message)s',
|
||||||
|
handlers=[
|
||||||
|
logging.StreamHandler(),
|
||||||
|
logging.FileHandler("pgpt_auto_ingest.log")
|
||||||
|
]
|
||||||
|
)
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Extensions de fichiers couramment supportées par PrivateGPT
|
||||||
|
SUPPORTED_EXTENSIONS = {
|
||||||
|
'.pdf', '.txt', '.md', '.doc', '.docx', '.ppt', '.pptx',
|
||||||
|
'.xls', '.xlsx', '.csv', '.epub', '.html', '.htm'
|
||||||
|
}
|
||||||
|
|
||||||
|
def parse_arguments():
|
||||||
|
"""Parse les arguments de ligne de commande."""
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Injecte automatiquement tous les fichiers d'un répertoire dans PrivateGPT"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"-d", "--directory",
|
||||||
|
type=str,
|
||||||
|
required=True,
|
||||||
|
help="Chemin du répertoire contenant les fichiers à injecter"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"-u", "--url",
|
||||||
|
type=str,
|
||||||
|
default="http://localhost:8001",
|
||||||
|
help="URL de l'API PrivateGPT (défaut: http://localhost:8001)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"-r", "--recursive",
|
||||||
|
action="store_true",
|
||||||
|
help="Parcourir récursivement les sous-répertoires"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"-t", "--threads",
|
||||||
|
type=int,
|
||||||
|
default=5,
|
||||||
|
help="Nombre de threads pour les injections parallèles (défaut: 5)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--retry",
|
||||||
|
type=int,
|
||||||
|
default=3,
|
||||||
|
help="Nombre de tentatives en cas d'échec (défaut: 3)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--retry-delay",
|
||||||
|
type=int,
|
||||||
|
default=5,
|
||||||
|
help="Délai entre les tentatives en secondes (défaut: 5)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--timeout",
|
||||||
|
type=int,
|
||||||
|
default=300,
|
||||||
|
help="Délai d'attente pour chaque requête en secondes (défaut: 300)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--extensions",
|
||||||
|
nargs="+",
|
||||||
|
help="Liste d'extensions spécifiques à injecter (ex: .pdf .txt)"
|
||||||
|
)
|
||||||
|
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
def find_files(directory: str, recursive: bool = False,
|
||||||
|
extensions: Set[str] = SUPPORTED_EXTENSIONS) -> List[Path]:
|
||||||
|
"""
|
||||||
|
Trouve tous les fichiers avec les extensions spécifiées dans le répertoire.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
directory: Répertoire à scanner
|
||||||
|
recursive: Si True, parcourt aussi les sous-répertoires
|
||||||
|
extensions: Ensemble d'extensions de fichier à inclure
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Liste des chemins de fichiers trouvés
|
||||||
|
"""
|
||||||
|
directory_path = Path(directory)
|
||||||
|
|
||||||
|
if not directory_path.exists() or not directory_path.is_dir():
|
||||||
|
logger.error(f"Le répertoire {directory} n'existe pas ou n'est pas un répertoire.")
|
||||||
|
return []
|
||||||
|
|
||||||
|
files = []
|
||||||
|
|
||||||
|
if recursive:
|
||||||
|
# Parcours récursif
|
||||||
|
for root, _, filenames in os.walk(directory):
|
||||||
|
for filename in filenames:
|
||||||
|
file_path = Path(root) / filename
|
||||||
|
if file_path.suffix.lower() in extensions:
|
||||||
|
files.append(file_path)
|
||||||
|
else:
|
||||||
|
# Parcours non récursif
|
||||||
|
for file_path in directory_path.iterdir():
|
||||||
|
if file_path.is_file() and file_path.suffix.lower() in extensions:
|
||||||
|
files.append(file_path)
|
||||||
|
|
||||||
|
return files
|
||||||
|
|
||||||
|
def ingest_file(file_path: Path, pgpt_url: str, timeout: int,
|
||||||
|
retry_count: int, retry_delay: int) -> Tuple[Path, bool, str]:
|
||||||
|
"""
|
||||||
|
Injecte un fichier dans PrivateGPT.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_path: Chemin du fichier à injecter
|
||||||
|
pgpt_url: URL de base de l'API PrivateGPT
|
||||||
|
timeout: Délai d'attente pour la requête
|
||||||
|
retry_count: Nombre de tentatives en cas d'échec
|
||||||
|
retry_delay: Délai entre les tentatives en secondes
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Tuple contenant (chemin_fichier, succès, message)
|
||||||
|
"""
|
||||||
|
ingest_url = f"{pgpt_url}/v1/ingest/file"
|
||||||
|
|
||||||
|
for attempt in range(retry_count):
|
||||||
|
try:
|
||||||
|
logger.info(f"Injection de {file_path} (tentative {attempt + 1}/{retry_count})")
|
||||||
|
|
||||||
|
with open(file_path, 'rb') as file:
|
||||||
|
files = {'file': (file_path.name, file, 'application/octet-stream')}
|
||||||
|
response = requests.post(ingest_url, files=files, timeout=timeout)
|
||||||
|
|
||||||
|
if response.status_code == 200:
|
||||||
|
result = response.json()
|
||||||
|
doc_ids = result.get('document_ids', [])
|
||||||
|
logger.info(f"Succès! {file_path} -> {len(doc_ids)} documents créés")
|
||||||
|
return file_path, True, f"{len(doc_ids)} documents créés"
|
||||||
|
else:
|
||||||
|
error_msg = f"Erreur HTTP {response.status_code}: {response.text}"
|
||||||
|
logger.warning(error_msg)
|
||||||
|
if attempt < retry_count - 1:
|
||||||
|
logger.info(f"Nouvelle tentative dans {retry_delay} secondes...")
|
||||||
|
time.sleep(retry_delay)
|
||||||
|
else:
|
||||||
|
return file_path, False, error_msg
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
error_msg = f"Exception: {str(e)}"
|
||||||
|
logger.warning(error_msg)
|
||||||
|
if attempt < retry_count - 1:
|
||||||
|
logger.info(f"Nouvelle tentative dans {retry_delay} secondes...")
|
||||||
|
time.sleep(retry_delay)
|
||||||
|
else:
|
||||||
|
return file_path, False, error_msg
|
||||||
|
|
||||||
|
return file_path, False, "Nombre maximum de tentatives atteint"
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""Fonction principale."""
|
||||||
|
args = parse_arguments()
|
||||||
|
|
||||||
|
# Préparation des extensions si spécifiées
|
||||||
|
extensions = set(args.extensions) if args.extensions else SUPPORTED_EXTENSIONS
|
||||||
|
# Assurer que les extensions commencent par un point
|
||||||
|
extensions = {ext if ext.startswith('.') else f'.{ext}' for ext in extensions}
|
||||||
|
|
||||||
|
logger.info(f"Démarrage de l'injection automatique depuis {args.directory}")
|
||||||
|
logger.info(f"URL PrivateGPT: {args.url}")
|
||||||
|
logger.info(f"Mode récursif: {args.recursive}")
|
||||||
|
logger.info(f"Extensions: {', '.join(extensions)}")
|
||||||
|
|
||||||
|
# Trouver les fichiers
|
||||||
|
files = find_files(args.directory, args.recursive, extensions)
|
||||||
|
total_files = len(files)
|
||||||
|
|
||||||
|
if total_files == 0:
|
||||||
|
logger.warning(f"Aucun fichier trouvé avec les extensions {', '.join(extensions)} dans {args.directory}")
|
||||||
|
return
|
||||||
|
|
||||||
|
logger.info(f"Trouvé {total_files} fichiers à injecter")
|
||||||
|
|
||||||
|
# Statistiques
|
||||||
|
successful = 0
|
||||||
|
failed = 0
|
||||||
|
failed_files = []
|
||||||
|
|
||||||
|
# Injection des fichiers en parallèle
|
||||||
|
with ThreadPoolExecutor(max_workers=args.threads) as executor:
|
||||||
|
futures = {
|
||||||
|
executor.submit(
|
||||||
|
ingest_file,
|
||||||
|
file_path,
|
||||||
|
args.url,
|
||||||
|
args.timeout,
|
||||||
|
args.retry,
|
||||||
|
args.retry_delay
|
||||||
|
): file_path for file_path in files
|
||||||
|
}
|
||||||
|
|
||||||
|
for future in as_completed(futures):
|
||||||
|
file_path, success, message = future.result()
|
||||||
|
if success:
|
||||||
|
successful += 1
|
||||||
|
else:
|
||||||
|
failed += 1
|
||||||
|
failed_files.append((file_path, message))
|
||||||
|
|
||||||
|
# Afficher la progression
|
||||||
|
progress = (successful + failed) / total_files * 100
|
||||||
|
logger.info(f"Progression: {progress:.1f}% ({successful + failed}/{total_files})")
|
||||||
|
|
||||||
|
# Rapport final
|
||||||
|
logger.info("="*50)
|
||||||
|
logger.info("RAPPORT D'INJECTION")
|
||||||
|
logger.info("="*50)
|
||||||
|
logger.info(f"Total des fichiers: {total_files}")
|
||||||
|
logger.info(f"Succès: {successful}")
|
||||||
|
logger.info(f"Échecs: {failed}")
|
||||||
|
|
||||||
|
if failed > 0:
|
||||||
|
logger.info("\nDétails des échecs:")
|
||||||
|
for file_path, message in failed_files:
|
||||||
|
logger.info(f"- {file_path}: {message}")
|
||||||
|
|
||||||
|
logger.info("="*50)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
try:
|
||||||
|
main()
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
logger.info("\nInterruption par l'utilisateur. Arrêt du processus.")
|
||||||
|
sys.exit(1)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur non gérée: {str(e)}", exc_info=True)
|
||||||
|
sys.exit(1)
|
||||||
90
IA/02 - injection_fiches/auto_ingest.sh
Executable file
90
IA/02 - injection_fiches/auto_ingest.sh
Executable file
@ -0,0 +1,90 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Script d'exécution pour auto_ingest.py
|
||||||
|
|
||||||
|
# Vérification des dépendances
|
||||||
|
check_dependencies() {
|
||||||
|
# Vérifier Python
|
||||||
|
if ! command -v python3 &> /dev/null; then
|
||||||
|
echo "Erreur: Python 3 n'est pas installé ou n'est pas dans le PATH"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Vérifier pip et requests
|
||||||
|
if ! python3 -c "import requests" &> /dev/null; then
|
||||||
|
echo "Installation de la bibliothèque requests..."
|
||||||
|
pip3 install requests
|
||||||
|
if [ $? -ne 0 ]; then
|
||||||
|
echo "Erreur: Impossible d'installer la bibliothèque requests"
|
||||||
|
echo "Exécutez manuellement: pip3 install requests"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
# Fonction d'aide
|
||||||
|
show_help() {
|
||||||
|
echo "Script d'injection automatique pour PrivateGPT"
|
||||||
|
echo ""
|
||||||
|
echo "Usage: $0 [options] -d RÉPERTOIRE"
|
||||||
|
echo ""
|
||||||
|
echo "Options:"
|
||||||
|
echo " -d, --directory RÉPERTOIRE Répertoire contenant les fichiers à injecter (obligatoire)"
|
||||||
|
echo " -u, --url URL URL de l'API PrivateGPT (défaut: http://localhost:8001)"
|
||||||
|
echo " -r, --recursive Parcourir récursivement les sous-répertoires"
|
||||||
|
echo " -t, --threads N Nombre de threads pour les injections parallèles (défaut: 5)"
|
||||||
|
echo " --retry N Nombre de tentatives en cas d'échec (défaut: 3)"
|
||||||
|
echo " --retry-delay N Délai entre les tentatives en secondes (défaut: 5)"
|
||||||
|
echo " --timeout N Délai d'attente pour chaque requête en secondes (défaut: 300)"
|
||||||
|
echo " --extensions EXT1 EXT2 ... Liste d'extensions spécifiques à injecter"
|
||||||
|
echo " -h, --help Afficher cette aide"
|
||||||
|
echo ""
|
||||||
|
echo "Exemple: $0 -d /documents -r --extensions pdf docx"
|
||||||
|
}
|
||||||
|
|
||||||
|
# Vérifier si aucun argument n'est fourni
|
||||||
|
if [ $# -eq 0 ]; then
|
||||||
|
show_help
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Vérifier l'argument d'aide
|
||||||
|
for arg in "$@"; do
|
||||||
|
if [ "$arg" = "-h" ] || [ "$arg" = "--help" ]; then
|
||||||
|
show_help
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
# Vérifier la présence de l'argument obligatoire (-d ou --directory)
|
||||||
|
directory_specified=false
|
||||||
|
for ((i=1; i<=$#; i++)); do
|
||||||
|
if [ "${!i}" = "-d" ] || [ "${!i}" = "--directory" ]; then
|
||||||
|
directory_specified=true
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [ "$directory_specified" = false ]; then
|
||||||
|
echo "Erreur: L'option -d/--directory est obligatoire"
|
||||||
|
echo "Utilisez -h ou --help pour afficher l'aide"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Vérifier les dépendances
|
||||||
|
check_dependencies
|
||||||
|
|
||||||
|
# Chemin au script Python
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
PYTHON_SCRIPT="${SCRIPT_DIR}/auto_ingest.py"
|
||||||
|
|
||||||
|
# Vérifier que le script Python existe
|
||||||
|
if [ ! -f "$PYTHON_SCRIPT" ]; then
|
||||||
|
echo "Erreur: Le script auto_ingest.py n'existe pas dans $SCRIPT_DIR"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Rendre le script Python exécutable
|
||||||
|
chmod +x "$PYTHON_SCRIPT"
|
||||||
|
|
||||||
|
# Exécuter le script Python avec tous les arguments
|
||||||
|
python3 "$PYTHON_SCRIPT" "$@"
|
||||||
40
IA/02 - injection_fiches/docker-compose.yml.example
Normal file
40
IA/02 - injection_fiches/docker-compose.yml.example
Normal file
@ -0,0 +1,40 @@
|
|||||||
|
version: '3.8'
|
||||||
|
|
||||||
|
services:
|
||||||
|
privategpt:
|
||||||
|
image: ghcr.io/zylon-ai/private-gpt:latest
|
||||||
|
container_name: privategpt
|
||||||
|
ports:
|
||||||
|
- "8001:8001"
|
||||||
|
environment:
|
||||||
|
- PGPT_PROFILES=local
|
||||||
|
# Décommentez et modifiez ces variables si vous voulez utiliser un modèle différent
|
||||||
|
# - PGPT_SETTINGS_LLMS_DEFAULT__MODEL=/models/custom-model.gguf
|
||||||
|
# - PGPT_SETTINGS_EMBEDDING_DEFAULT__MODEL=/models/custom-embedding-model
|
||||||
|
volumes:
|
||||||
|
# Volume persistant pour les données
|
||||||
|
- privategpt-data:/app/local_data
|
||||||
|
# Montage du répertoire d'auto-injection
|
||||||
|
- ./documents_to_ingest:/app/documents_to_ingest
|
||||||
|
# Montage des modèles personnalisés (décommentez si nécessaire)
|
||||||
|
# - ./custom_models:/app/models
|
||||||
|
restart: unless-stopped
|
||||||
|
# Décommentez ces lignes si vous avez un GPU NVIDIA
|
||||||
|
# deploy:
|
||||||
|
# resources:
|
||||||
|
# reservations:
|
||||||
|
# devices:
|
||||||
|
# - driver: nvidia
|
||||||
|
# count: 1
|
||||||
|
# capabilities: [gpu]
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
privategpt-data:
|
||||||
|
name: privategpt-data
|
||||||
|
|
||||||
|
# Instructions d'utilisation:
|
||||||
|
# 1. Copiez ce fichier sous le nom "docker-compose.yml"
|
||||||
|
# 2. Créez un répertoire "documents_to_ingest" à côté du fichier docker-compose.yml
|
||||||
|
# 3. Placez vos documents à injecter dans ce répertoire
|
||||||
|
# 4. Lancez avec la commande: docker-compose up -d
|
||||||
|
# 5. Utilisez le script d'injection: ./auto_ingest.sh -d documents_to_ingest -u http://localhost:8001
|
||||||
224
IA/02 - injection_fiches/nettoyer_pgpt.py
Normal file
224
IA/02 - injection_fiches/nettoyer_pgpt.py
Normal file
@ -0,0 +1,224 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Script de nettoyage pour PrivateGPT
|
||||||
|
|
||||||
|
Ce script permet de lister et supprimer les documents ingérés dans PrivateGPT.
|
||||||
|
Options:
|
||||||
|
- Lister tous les documents
|
||||||
|
- Supprimer des documents par préfixe (ex: "temp_section_")
|
||||||
|
- Supprimer des documents par motif
|
||||||
|
- Supprimer tous les documents
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python nettoyer_pgpt.py --list
|
||||||
|
python nettoyer_pgpt.py --delete-prefix "temp_section_"
|
||||||
|
python nettoyer_pgpt.py --delete-pattern "rapport_.*\.md"
|
||||||
|
python nettoyer_pgpt.py --delete-all
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import requests
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
import uuid
|
||||||
|
from typing import List, Dict, Any, Optional
|
||||||
|
|
||||||
|
# Configuration de l'API PrivateGPT
|
||||||
|
PGPT_URL = "http://127.0.0.1:8001"
|
||||||
|
API_URL = f"{PGPT_URL}/v1"
|
||||||
|
|
||||||
|
|
||||||
|
def check_api_availability() -> bool:
|
||||||
|
"""Vérifie si l'API PrivateGPT est disponible"""
|
||||||
|
try:
|
||||||
|
response = requests.get(f"{PGPT_URL}/health")
|
||||||
|
if response.status_code == 200:
|
||||||
|
print("✅ API PrivateGPT disponible")
|
||||||
|
return True
|
||||||
|
else:
|
||||||
|
print(f"❌ L'API PrivateGPT a retourné le code d'état {response.status_code}")
|
||||||
|
return False
|
||||||
|
except requests.RequestException as e:
|
||||||
|
print(f"❌ Erreur de connexion à l'API PrivateGPT: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def list_documents() -> List[Dict[str, Any]]:
|
||||||
|
"""Liste tous les documents ingérés et renvoie la liste des métadonnées"""
|
||||||
|
try:
|
||||||
|
# Récupérer la liste des documents
|
||||||
|
response = requests.get(f"{API_URL}/ingest/list")
|
||||||
|
response.raise_for_status()
|
||||||
|
data = response.json()
|
||||||
|
|
||||||
|
# Format de réponse OpenAI
|
||||||
|
if "data" in data:
|
||||||
|
documents = data.get("data", [])
|
||||||
|
# Format alternatif
|
||||||
|
else:
|
||||||
|
documents = data.get("documents", [])
|
||||||
|
|
||||||
|
# Construire une liste normalisée des documents
|
||||||
|
normalized_docs = []
|
||||||
|
for doc in documents:
|
||||||
|
doc_id = doc.get("doc_id") or doc.get("id")
|
||||||
|
metadata = doc.get("doc_metadata", {})
|
||||||
|
filename = metadata.get("file_name") or metadata.get("filename", "Inconnu")
|
||||||
|
|
||||||
|
normalized_docs.append({
|
||||||
|
"id": doc_id,
|
||||||
|
"filename": filename,
|
||||||
|
"metadata": metadata
|
||||||
|
})
|
||||||
|
|
||||||
|
return normalized_docs
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Erreur lors de la récupération des documents: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def print_documents(documents: List[Dict[str, Any]]) -> None:
|
||||||
|
"""Affiche la liste des documents de façon lisible"""
|
||||||
|
if not documents:
|
||||||
|
print("📋 Aucun document trouvé dans PrivateGPT")
|
||||||
|
return
|
||||||
|
|
||||||
|
print(f"📋 {len(documents)} documents trouvés dans PrivateGPT:")
|
||||||
|
|
||||||
|
# Regrouper par nom de fichier pour un affichage plus compact
|
||||||
|
files_grouped = {}
|
||||||
|
for doc in documents:
|
||||||
|
filename = doc["filename"]
|
||||||
|
if filename not in files_grouped:
|
||||||
|
files_grouped[filename] = []
|
||||||
|
files_grouped[filename].append(doc["id"])
|
||||||
|
|
||||||
|
# Afficher les résultats groupés
|
||||||
|
for i, (filename, ids) in enumerate(files_grouped.items(), 1):
|
||||||
|
print(f"{i}. {filename} ({len(ids)} chunks)")
|
||||||
|
if args.verbose:
|
||||||
|
for j, doc_id in enumerate(ids, 1):
|
||||||
|
print(f" {j}. ID: {doc_id}")
|
||||||
|
|
||||||
|
|
||||||
|
def delete_document(doc_id: str) -> bool:
|
||||||
|
"""Supprime un document par son ID"""
|
||||||
|
try:
|
||||||
|
response = requests.delete(f"{API_URL}/ingest/{doc_id}")
|
||||||
|
if response.status_code == 200:
|
||||||
|
return True
|
||||||
|
else:
|
||||||
|
print(f"⚠️ Échec de la suppression de l'ID {doc_id}: Code {response.status_code}")
|
||||||
|
return False
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Erreur lors de la suppression de l'ID {doc_id}: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def delete_documents_by_criteria(documents: List[Dict[str, Any]],
|
||||||
|
prefix: Optional[str] = None,
|
||||||
|
pattern: Optional[str] = None,
|
||||||
|
delete_all: bool = False) -> int:
|
||||||
|
"""
|
||||||
|
Supprime des documents selon différents critères
|
||||||
|
Retourne le nombre de documents supprimés
|
||||||
|
"""
|
||||||
|
if not documents:
|
||||||
|
print("❌ Aucun document à supprimer")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
if not (prefix or pattern or delete_all):
|
||||||
|
print("❌ Aucun critère de suppression spécifié")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Comptage des suppressions réussies
|
||||||
|
success_count = 0
|
||||||
|
|
||||||
|
# Filtrer les documents à supprimer
|
||||||
|
docs_to_delete = []
|
||||||
|
|
||||||
|
if delete_all:
|
||||||
|
docs_to_delete = documents
|
||||||
|
print(f"🗑️ Suppression de tous les documents ({len(documents)} chunks)...")
|
||||||
|
elif prefix:
|
||||||
|
docs_to_delete = [doc for doc in documents if doc["filename"].startswith(prefix)]
|
||||||
|
print(f"🗑️ Suppression des documents dont le nom commence par '{prefix}' ({len(docs_to_delete)} chunks)...")
|
||||||
|
elif pattern:
|
||||||
|
try:
|
||||||
|
regex = re.compile(pattern)
|
||||||
|
docs_to_delete = [doc for doc in documents if regex.search(doc["filename"])]
|
||||||
|
print(f"🗑️ Suppression des documents correspondant au motif '{pattern}' ({len(docs_to_delete)} chunks)...")
|
||||||
|
except re.error as e:
|
||||||
|
print(f"❌ Expression régulière invalide: {e}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Demander confirmation si beaucoup de documents
|
||||||
|
if len(docs_to_delete) > 5 and not args.force:
|
||||||
|
confirm = input(f"⚠️ Vous êtes sur le point de supprimer {len(docs_to_delete)} chunks. Confirmer ? (o/N) ")
|
||||||
|
if confirm.lower() != 'o':
|
||||||
|
print("❌ Opération annulée")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Supprimer les documents
|
||||||
|
for doc in docs_to_delete:
|
||||||
|
if delete_document(doc["id"]):
|
||||||
|
success_count += 1
|
||||||
|
if args.verbose:
|
||||||
|
print(f"✅ Document supprimé: {doc['filename']} (ID: {doc['id']})")
|
||||||
|
|
||||||
|
# Petite pause pour éviter de surcharger l'API
|
||||||
|
time.sleep(0.1)
|
||||||
|
|
||||||
|
print(f"✅ {success_count}/{len(docs_to_delete)} documents supprimés avec succès")
|
||||||
|
return success_count
|
||||||
|
|
||||||
|
|
||||||
|
def generate_unique_prefix() -> str:
|
||||||
|
"""Génère un préfixe unique basé sur un UUID pour différencier les fichiers temporaires"""
|
||||||
|
unique_id = str(uuid.uuid4())[:8] # Prendre les 8 premiers caractères de l'UUID
|
||||||
|
return f"temp_{unique_id}_"
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
parser = argparse.ArgumentParser(description="Utilitaire de nettoyage pour PrivateGPT")
|
||||||
|
|
||||||
|
# Options principales
|
||||||
|
group = parser.add_mutually_exclusive_group(required=True)
|
||||||
|
group.add_argument("--list", action="store_true", help="Lister tous les documents ingérés")
|
||||||
|
group.add_argument("--delete-prefix", type=str, help="Supprimer les documents dont le nom commence par PREFIX")
|
||||||
|
group.add_argument("--delete-pattern", type=str, help="Supprimer les documents dont le nom correspond au motif PATTERN (regex)")
|
||||||
|
group.add_argument("--delete-all", action="store_true", help="Supprimer tous les documents (⚠️ DANGER)")
|
||||||
|
group.add_argument("--generate-prefix", action="store_true", help="Générer un préfixe unique pour les fichiers temporaires")
|
||||||
|
|
||||||
|
# Options additionnelles
|
||||||
|
parser.add_argument("--force", action="store_true", help="Ne pas demander de confirmation")
|
||||||
|
parser.add_argument("--verbose", action="store_true", help="Afficher plus de détails")
|
||||||
|
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
# Vérifier la disponibilité de l'API
|
||||||
|
if not check_api_availability():
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
# Générer un préfixe unique
|
||||||
|
if args.generate_prefix:
|
||||||
|
unique_prefix = generate_unique_prefix()
|
||||||
|
print(f"🔑 Préfixe unique généré: {unique_prefix}")
|
||||||
|
print(f"Utilisez ce préfixe pour les fichiers temporaires de votre script.")
|
||||||
|
sys.exit(0)
|
||||||
|
|
||||||
|
# Récupérer la liste des documents
|
||||||
|
documents = list_documents()
|
||||||
|
|
||||||
|
# Traiter selon l'option choisie
|
||||||
|
if args.list:
|
||||||
|
print_documents(documents)
|
||||||
|
elif args.delete_prefix:
|
||||||
|
delete_documents_by_criteria(documents, prefix=args.delete_prefix)
|
||||||
|
elif args.delete_pattern:
|
||||||
|
delete_documents_by_criteria(documents, pattern=args.delete_pattern)
|
||||||
|
elif args.delete_all:
|
||||||
|
delete_documents_by_criteria(documents, delete_all=True)
|
||||||
263
IA/02 - injection_fiches/watch_directory.py
Executable file
263
IA/02 - injection_fiches/watch_directory.py
Executable file
@ -0,0 +1,263 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
"""
|
||||||
|
Script de surveillance de répertoire pour l'injection automatique dans PrivateGPT
|
||||||
|
|
||||||
|
Ce script surveille un répertoire et injecte automatiquement les nouveaux fichiers
|
||||||
|
dans PrivateGPT dès qu'ils sont ajoutés ou modifiés.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
import logging
|
||||||
|
import argparse
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Set, Dict, List
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
try:
|
||||||
|
from watchdog.observers import Observer
|
||||||
|
from watchdog.events import FileSystemEventHandler, FileCreatedEvent, FileModifiedEvent
|
||||||
|
except ImportError:
|
||||||
|
print("Bibliothèque 'watchdog' non installée. Installation en cours...")
|
||||||
|
subprocess.run([sys.executable, "-m", "pip", "install", "watchdog"])
|
||||||
|
from watchdog.observers import Observer
|
||||||
|
from watchdog.events import FileSystemEventHandler, FileCreatedEvent, FileModifiedEvent
|
||||||
|
|
||||||
|
# Configuration du logging
|
||||||
|
logging.basicConfig(
|
||||||
|
level=logging.INFO,
|
||||||
|
format='%(asctime)s - %(levelname)s - %(message)s',
|
||||||
|
handlers=[
|
||||||
|
logging.StreamHandler(),
|
||||||
|
logging.FileHandler("pgpt_watch_directory.log")
|
||||||
|
]
|
||||||
|
)
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Extensions de fichiers couramment supportées par PrivateGPT
|
||||||
|
SUPPORTED_EXTENSIONS = {
|
||||||
|
'.pdf', '.txt', '.md', '.doc', '.docx', '.ppt', '.pptx',
|
||||||
|
'.xls', '.xlsx', '.csv', '.epub', '.html', '.htm'
|
||||||
|
}
|
||||||
|
|
||||||
|
class DocumentHandler(FileSystemEventHandler):
|
||||||
|
"""Gestionnaire d'événements pour les fichiers de documents."""
|
||||||
|
|
||||||
|
def __init__(self, watch_dir: str, ingest_script: str, pgpt_url: str,
|
||||||
|
extensions: Set[str], delay: int = 5):
|
||||||
|
"""
|
||||||
|
Initialise le gestionnaire d'événements.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
watch_dir: Répertoire à surveiller
|
||||||
|
ingest_script: Chemin vers le script d'injection
|
||||||
|
pgpt_url: URL de l'API PrivateGPT
|
||||||
|
extensions: Extensions de fichiers à traiter
|
||||||
|
delay: Délai en secondes à attendre avant le traitement (évite de traiter des fichiers partiellement écrits)
|
||||||
|
"""
|
||||||
|
self.watch_dir = os.path.abspath(watch_dir)
|
||||||
|
self.ingest_script = os.path.abspath(ingest_script)
|
||||||
|
self.pgpt_url = pgpt_url
|
||||||
|
self.extensions = extensions
|
||||||
|
self.delay = delay
|
||||||
|
|
||||||
|
# Queue pour les fichiers en attente de traitement
|
||||||
|
self.pending_files: Dict[str, float] = {}
|
||||||
|
|
||||||
|
# Vérifier que le script d'injection existe
|
||||||
|
if not os.path.exists(self.ingest_script):
|
||||||
|
logger.error(f"Le script d'injection {self.ingest_script} n'existe pas!")
|
||||||
|
raise FileNotFoundError(f"Script d'injection introuvable: {self.ingest_script}")
|
||||||
|
|
||||||
|
def on_created(self, event):
|
||||||
|
"""Appelé lorsqu'un fichier est créé."""
|
||||||
|
if not event.is_directory:
|
||||||
|
self._handle_file_event(event)
|
||||||
|
|
||||||
|
def on_modified(self, event):
|
||||||
|
"""Appelé lorsqu'un fichier est modifié."""
|
||||||
|
if not event.is_directory:
|
||||||
|
self._handle_file_event(event)
|
||||||
|
|
||||||
|
def _handle_file_event(self, event):
|
||||||
|
"""Traite un événement de fichier (création ou modification)."""
|
||||||
|
file_path = event.src_path
|
||||||
|
file_ext = os.path.splitext(file_path)[1].lower()
|
||||||
|
|
||||||
|
# Ignorer les fichiers non supportés
|
||||||
|
if file_ext not in self.extensions:
|
||||||
|
return
|
||||||
|
|
||||||
|
# Ignorer les fichiers temporaires et cachés
|
||||||
|
file_name = os.path.basename(file_path)
|
||||||
|
if file_name.startswith('.') or file_name.startswith('~') or file_name.endswith('.tmp'):
|
||||||
|
return
|
||||||
|
|
||||||
|
# Ajouter à la queue avec l'horodatage actuel
|
||||||
|
self.pending_files[file_path] = time.time()
|
||||||
|
logger.info(f"Fichier détecté: {file_path} (en attente pendant {self.delay} secondes)")
|
||||||
|
|
||||||
|
def process_pending_files(self):
|
||||||
|
"""Traite les fichiers en attente qui ont dépassé le délai d'attente."""
|
||||||
|
current_time = time.time()
|
||||||
|
files_to_process: List[str] = []
|
||||||
|
|
||||||
|
# Identifier les fichiers prêts à être traités
|
||||||
|
for file_path, timestamp in list(self.pending_files.items()):
|
||||||
|
if current_time - timestamp >= self.delay:
|
||||||
|
if os.path.exists(file_path): # Vérifier que le fichier existe toujours
|
||||||
|
files_to_process.append(file_path)
|
||||||
|
self.pending_files.pop(file_path)
|
||||||
|
|
||||||
|
# Traiter les fichiers par lot
|
||||||
|
if files_to_process:
|
||||||
|
self._ingest_files(files_to_process)
|
||||||
|
|
||||||
|
def _ingest_files(self, files: List[str]):
|
||||||
|
"""
|
||||||
|
Injecte une liste de fichiers en utilisant le script d'injection.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
files: Liste des chemins de fichiers à injecter
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Créer un répertoire temporaire pour stocker la liste des fichiers
|
||||||
|
temp_dir = os.path.join(os.path.dirname(self.ingest_script), "temp")
|
||||||
|
os.makedirs(temp_dir, exist_ok=True)
|
||||||
|
|
||||||
|
# Créer un fichier de liste
|
||||||
|
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||||
|
list_file = os.path.join(temp_dir, f"files_to_ingest_{timestamp}.txt")
|
||||||
|
|
||||||
|
with open(list_file, "w") as f:
|
||||||
|
for file_path in files:
|
||||||
|
f.write(f"{file_path}\n")
|
||||||
|
|
||||||
|
# Construire la commande pour le script d'injection
|
||||||
|
for file_path in files:
|
||||||
|
file_dir = os.path.dirname(file_path)
|
||||||
|
logger.info(f"Injection de {file_path}...")
|
||||||
|
|
||||||
|
cmd = [
|
||||||
|
sys.executable,
|
||||||
|
self.ingest_script,
|
||||||
|
"-d", file_dir,
|
||||||
|
"-u", self.pgpt_url,
|
||||||
|
"--extensions", os.path.splitext(file_path)[1][1:] # Extension sans le point
|
||||||
|
]
|
||||||
|
|
||||||
|
# Exécuter la commande
|
||||||
|
process = subprocess.run(
|
||||||
|
cmd,
|
||||||
|
capture_output=True,
|
||||||
|
text=True
|
||||||
|
)
|
||||||
|
|
||||||
|
if process.returncode == 0:
|
||||||
|
logger.info(f"Injection réussie de {file_path}")
|
||||||
|
else:
|
||||||
|
logger.error(f"Échec de l'injection de {file_path}: {process.stderr}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur lors de l'injection des fichiers: {str(e)}")
|
||||||
|
|
||||||
|
def parse_arguments():
|
||||||
|
"""Parse les arguments de ligne de commande."""
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Surveille un répertoire et injecte automatiquement les nouveaux fichiers dans PrivateGPT"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"-d", "--directory",
|
||||||
|
type=str,
|
||||||
|
required=True,
|
||||||
|
help="Chemin du répertoire à surveiller"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"-s", "--script",
|
||||||
|
type=str,
|
||||||
|
default=None,
|
||||||
|
help="Chemin vers le script auto_ingest.py (par défaut: détection automatique)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"-u", "--url",
|
||||||
|
type=str,
|
||||||
|
default="http://localhost:8001",
|
||||||
|
help="URL de l'API PrivateGPT (défaut: http://localhost:8001)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--delay",
|
||||||
|
type=int,
|
||||||
|
default=5,
|
||||||
|
help="Délai en secondes avant de traiter un nouveau fichier (défaut: 5)"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--extensions",
|
||||||
|
nargs="+",
|
||||||
|
help="Liste d'extensions spécifiques à surveiller (ex: pdf txt)"
|
||||||
|
)
|
||||||
|
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""Fonction principale."""
|
||||||
|
args = parse_arguments()
|
||||||
|
|
||||||
|
# Préparation des extensions si spécifiées
|
||||||
|
extensions = set(args.extensions) if args.extensions else SUPPORTED_EXTENSIONS
|
||||||
|
# Assurer que les extensions commencent par un point
|
||||||
|
extensions = {ext if ext.startswith('.') else f'.{ext}' for ext in extensions}
|
||||||
|
|
||||||
|
# Déterminer le chemin du script d'injection
|
||||||
|
if args.script:
|
||||||
|
ingest_script = args.script
|
||||||
|
else:
|
||||||
|
# Utiliser le script auto_ingest.py dans le même répertoire que ce script
|
||||||
|
ingest_script = os.path.join(os.path.dirname(os.path.abspath(__file__)), "auto_ingest.py")
|
||||||
|
|
||||||
|
# Créer le répertoire de surveillance s'il n'existe pas
|
||||||
|
watch_dir = os.path.abspath(args.directory)
|
||||||
|
if not os.path.exists(watch_dir):
|
||||||
|
logger.info(f"Création du répertoire de surveillance: {watch_dir}")
|
||||||
|
os.makedirs(watch_dir, exist_ok=True)
|
||||||
|
|
||||||
|
logger.info(f"Démarrage de la surveillance de {watch_dir}")
|
||||||
|
logger.info(f"URL PrivateGPT: {args.url}")
|
||||||
|
logger.info(f"Extensions surveillées: {', '.join(extensions)}")
|
||||||
|
logger.info(f"Délai de traitement: {args.delay} secondes")
|
||||||
|
|
||||||
|
# Initialiser le gestionnaire et l'observateur
|
||||||
|
event_handler = DocumentHandler(
|
||||||
|
watch_dir=watch_dir,
|
||||||
|
ingest_script=ingest_script,
|
||||||
|
pgpt_url=args.url,
|
||||||
|
extensions=extensions,
|
||||||
|
delay=args.delay
|
||||||
|
)
|
||||||
|
|
||||||
|
observer = Observer()
|
||||||
|
observer.schedule(event_handler, path=watch_dir, recursive=True)
|
||||||
|
observer.start()
|
||||||
|
|
||||||
|
try:
|
||||||
|
logger.info("Surveillance en cours... (Ctrl+C pour quitter)")
|
||||||
|
|
||||||
|
while True:
|
||||||
|
# Traiter les fichiers en attente
|
||||||
|
event_handler.process_pending_files()
|
||||||
|
time.sleep(1)
|
||||||
|
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
logger.info("\nInterruption par l'utilisateur. Arrêt de la surveillance.")
|
||||||
|
observer.stop()
|
||||||
|
|
||||||
|
observer.join()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
try:
|
||||||
|
main()
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Erreur non gérée: {str(e)}", exc_info=True)
|
||||||
|
sys.exit(1)
|
||||||
94
IA/02 - injection_fiches/watch_directory.sh
Executable file
94
IA/02 - injection_fiches/watch_directory.sh
Executable file
@ -0,0 +1,94 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Script d'exécution pour watch_directory.py
|
||||||
|
|
||||||
|
# Vérification des dépendances
|
||||||
|
check_dependencies() {
|
||||||
|
# Vérifier Python
|
||||||
|
if ! command -v python3 &> /dev/null; then
|
||||||
|
echo "Erreur: Python 3 n'est pas installé ou n'est pas dans le PATH"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Vérifier les bibliothèques Python requises
|
||||||
|
python3 -c "
|
||||||
|
try:
|
||||||
|
import watchdog
|
||||||
|
except ImportError:
|
||||||
|
print('Installation de la bibliothèque watchdog...')
|
||||||
|
import subprocess, sys
|
||||||
|
subprocess.run([sys.executable, '-m', 'pip', 'install', 'watchdog'])
|
||||||
|
" || {
|
||||||
|
echo "Erreur: Impossible d'installer les dépendances"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# Fonction d'aide
|
||||||
|
show_help() {
|
||||||
|
echo "Script de surveillance de répertoire pour PrivateGPT"
|
||||||
|
echo ""
|
||||||
|
echo "Usage: $0 [options] -d RÉPERTOIRE"
|
||||||
|
echo ""
|
||||||
|
echo "Options:"
|
||||||
|
echo " -d, --directory RÉPERTOIRE Répertoire à surveiller (obligatoire)"
|
||||||
|
echo " -s, --script CHEMIN Chemin vers le script auto_ingest.py (facultatif)"
|
||||||
|
echo " -u, --url URL URL de l'API PrivateGPT (défaut: http://localhost:8001)"
|
||||||
|
echo " --delay N Délai en secondes avant de traiter un nouveau fichier (défaut: 5)"
|
||||||
|
echo " --extensions EXT1 EXT2 ... Liste d'extensions spécifiques à surveiller"
|
||||||
|
echo " -h, --help Afficher cette aide"
|
||||||
|
echo ""
|
||||||
|
echo "Exemple: $0 -d /documents/à/surveiller --extensions pdf docx"
|
||||||
|
}
|
||||||
|
|
||||||
|
# Vérifier si aucun argument n'est fourni
|
||||||
|
if [ $# -eq 0 ]; then
|
||||||
|
show_help
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Vérifier l'argument d'aide
|
||||||
|
for arg in "$@"; do
|
||||||
|
if [ "$arg" = "-h" ] || [ "$arg" = "--help" ]; then
|
||||||
|
show_help
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
# Vérifier la présence de l'argument obligatoire (-d ou --directory)
|
||||||
|
directory_specified=false
|
||||||
|
for ((i=1; i<=$#; i++)); do
|
||||||
|
if [ "${!i}" = "-d" ] || [ "${!i}" = "--directory" ]; then
|
||||||
|
directory_specified=true
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [ "$directory_specified" = false ]; then
|
||||||
|
echo "Erreur: L'option -d/--directory est obligatoire"
|
||||||
|
echo "Utilisez -h ou --help pour afficher l'aide"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Vérifier les dépendances
|
||||||
|
check_dependencies
|
||||||
|
|
||||||
|
# Chemin au script Python
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
PYTHON_SCRIPT="${SCRIPT_DIR}/watch_directory.py"
|
||||||
|
|
||||||
|
# Vérifier que le script Python existe
|
||||||
|
if [ ! -f "$PYTHON_SCRIPT" ]; then
|
||||||
|
echo "Erreur: Le script watch_directory.py n'existe pas dans $SCRIPT_DIR"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Rendre le script Python exécutable
|
||||||
|
chmod +x "$PYTHON_SCRIPT"
|
||||||
|
|
||||||
|
# Message d'information sur l'arrêt
|
||||||
|
echo "Démarrage de la surveillance..."
|
||||||
|
echo "Appuyez sur Ctrl+C pour arrêter la surveillance"
|
||||||
|
echo ""
|
||||||
|
|
||||||
|
# Exécuter le script Python avec tous les arguments
|
||||||
|
python3 "$PYTHON_SCRIPT" "$@"
|
||||||
171
IA/get_regeneration_plan.py
Normal file
171
IA/get_regeneration_plan.py
Normal file
@ -0,0 +1,171 @@
|
|||||||
|
from datetime import datetime
|
||||||
|
from collections import defaultdict, deque
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
from datetime import timezone
|
||||||
|
|
||||||
|
sys.path.append(os.path.dirname(os.path.dirname(__file__)))
|
||||||
|
|
||||||
|
# À adapter dans ton environnement
|
||||||
|
from config import GITEA_URL, ORGANISATION, DEPOT_FICHES, ENV
|
||||||
|
from utils.gitea import recuperer_date_dernier_commit
|
||||||
|
from IA.make_config import MAKE # MAKE doit être importé depuis un fichier de config
|
||||||
|
|
||||||
|
def get_mtime(path):
|
||||||
|
try:
|
||||||
|
return datetime.fromtimestamp(os.path.getmtime(path), tz=timezone.utc)
|
||||||
|
except FileNotFoundError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
def get_commit_time(path_relative):
|
||||||
|
commits_url = f"{GITEA_URL}/repos/{ORGANISATION}/{DEPOT_FICHES}/commits?path={path_relative.replace("Fiches", "Documents")}&sha={ENV}"
|
||||||
|
return recuperer_date_dernier_commit(commits_url)
|
||||||
|
|
||||||
|
def resolve_path_from_where(where_str):
|
||||||
|
parts = where_str.split(".")
|
||||||
|
current = MAKE
|
||||||
|
path_stack = []
|
||||||
|
|
||||||
|
for part in parts:
|
||||||
|
if isinstance(current, dict) and part in current:
|
||||||
|
path_stack.append((part, current))
|
||||||
|
current = current[part]
|
||||||
|
else:
|
||||||
|
return None
|
||||||
|
|
||||||
|
if not isinstance(current, str):
|
||||||
|
return None
|
||||||
|
|
||||||
|
for i in range(len(path_stack) - 1, -1, -1):
|
||||||
|
key, context = path_stack[i]
|
||||||
|
if "directory" in context:
|
||||||
|
directory = context["directory"]
|
||||||
|
if "fiches" in where_str:
|
||||||
|
return os.path.join("Fiches", directory, current)
|
||||||
|
else:
|
||||||
|
return os.path.join(directory, current)
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
def identifier_type_fiche(path):
|
||||||
|
for type_fiche, data in MAKE["fiches"].items():
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
continue
|
||||||
|
directory = data.get("directory", "")
|
||||||
|
prefix = data.get("prefix", "")
|
||||||
|
base = os.path.join("Fiches", directory, prefix)
|
||||||
|
if path.startswith(base):
|
||||||
|
return type_fiche, data
|
||||||
|
raise ValueError("Type de fiche non reconnu")
|
||||||
|
|
||||||
|
def doit_regenerer(fichier, doc_deps, fiche_data=None):
|
||||||
|
mtime_fichier = get_mtime(fichier)
|
||||||
|
|
||||||
|
if fiche_data:
|
||||||
|
gitea_dep = fiche_data.get("depends_on", {}).get("gitea", {})
|
||||||
|
if gitea_dep.get("compare") == "file2commit":
|
||||||
|
commit_time = get_commit_time(fichier)
|
||||||
|
if commit_time and mtime_fichier and commit_time > mtime_fichier:
|
||||||
|
return True
|
||||||
|
|
||||||
|
for _, dep in doc_deps.items():
|
||||||
|
if isinstance(dep, dict) and "where" in dep:
|
||||||
|
source_path = resolve_path_from_where(dep["where"])
|
||||||
|
if source_path:
|
||||||
|
if dep["compare"] == "file2file":
|
||||||
|
mtime_source = get_mtime(source_path)
|
||||||
|
elif dep["compare"] == "file2commit":
|
||||||
|
mtime_source = get_commit_time(source_path)
|
||||||
|
else:
|
||||||
|
continue
|
||||||
|
if mtime_source and mtime_fichier and mtime_source > mtime_fichier:
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
def get_regeneration_plan(fiche_path):
|
||||||
|
def build_dependency_graph_complete(path, graph=None, visited=None):
|
||||||
|
if graph is None:
|
||||||
|
graph = defaultdict(set)
|
||||||
|
if visited is None:
|
||||||
|
visited = set()
|
||||||
|
if path in visited:
|
||||||
|
return graph
|
||||||
|
visited.add(path)
|
||||||
|
|
||||||
|
try:
|
||||||
|
_, fiche_data = identifier_type_fiche(path)
|
||||||
|
except ValueError:
|
||||||
|
return graph
|
||||||
|
|
||||||
|
depends = fiche_data.get("depends_on", {})
|
||||||
|
doc_deps = depends.get("document", {})
|
||||||
|
|
||||||
|
for _, dep_info in doc_deps.items():
|
||||||
|
if isinstance(dep_info, dict) and "where" in dep_info:
|
||||||
|
dep_path = resolve_path_from_where(dep_info["where"])
|
||||||
|
if dep_path:
|
||||||
|
graph[path].add(dep_path)
|
||||||
|
build_dependency_graph_complete(dep_path, graph, visited)
|
||||||
|
|
||||||
|
if "file" in fiche_data:
|
||||||
|
for fichier in fiche_data["file"].values():
|
||||||
|
dir_fiche = fiche_data.get("directory", "")
|
||||||
|
fichier_path = os.path.join("Fiches", dir_fiche, fichier)
|
||||||
|
if fichier_path not in graph:
|
||||||
|
graph[fichier_path] = set()
|
||||||
|
|
||||||
|
return graph
|
||||||
|
|
||||||
|
def topological_sort(graph):
|
||||||
|
in_degree = defaultdict(int)
|
||||||
|
for node in graph:
|
||||||
|
for dep in graph[node]:
|
||||||
|
in_degree[dep] += 1
|
||||||
|
queue = deque([node for node in graph if in_degree[node] == 0])
|
||||||
|
result = []
|
||||||
|
|
||||||
|
while queue:
|
||||||
|
node = queue.popleft()
|
||||||
|
result.append(node)
|
||||||
|
for dep in graph[node]:
|
||||||
|
in_degree[dep] -= 1
|
||||||
|
if in_degree[dep] == 0:
|
||||||
|
queue.append(dep)
|
||||||
|
|
||||||
|
all_nodes = set(graph.keys()).union(*graph.values())
|
||||||
|
for node in all_nodes:
|
||||||
|
if node not in result:
|
||||||
|
result.append(node)
|
||||||
|
|
||||||
|
return result[::-1]
|
||||||
|
|
||||||
|
graph = build_dependency_graph_complete(fiche_path)
|
||||||
|
sorted_fiches = topological_sort(graph)
|
||||||
|
if fiche_path not in sorted_fiches:
|
||||||
|
sorted_fiches.append(fiche_path)
|
||||||
|
|
||||||
|
|
||||||
|
to_regen = []
|
||||||
|
regen_flags = {}
|
||||||
|
|
||||||
|
for fiche in sorted_fiches:
|
||||||
|
print(f"=> {fiche}")
|
||||||
|
try:
|
||||||
|
_, fiche_data = identifier_type_fiche(fiche)
|
||||||
|
except ValueError:
|
||||||
|
fiche_data = None
|
||||||
|
depends = fiche_data.get("depends_on", {}) if fiche_data else {}
|
||||||
|
doc_deps = depends.get("document", {}) if depends else {}
|
||||||
|
|
||||||
|
doit = doit_regenerer(fiche, doc_deps, fiche_data)
|
||||||
|
if any(regen_flags.get(dep, False) for dep in graph.get(fiche, [])):
|
||||||
|
doit = True
|
||||||
|
|
||||||
|
regen_flags[fiche] = doit
|
||||||
|
if doit:
|
||||||
|
to_regen.append(fiche)
|
||||||
|
|
||||||
|
return to_regen
|
||||||
|
|
||||||
|
plan = get_regeneration_plan("Fiches/Minerai/Fiche minerai antimoine.md")
|
||||||
|
print(plan)
|
||||||
141
IA/make_config.py
Normal file
141
IA/make_config.py
Normal file
@ -0,0 +1,141 @@
|
|||||||
|
from utils.gitea import recuperer_date_dernier_commit
|
||||||
|
#
|
||||||
|
# from config import GITEA_URL, GITEA_TOKEN, ORGANISATION, DEPOT_FICHES, DEPOT_CODE, ENV, ENV_CODE, DOT_FILE
|
||||||
|
#
|
||||||
|
#def recuperer_date_dernier_commit(url):
|
||||||
|
# headers = {"Authorization": f"token {GITEA_TOKEN}"}
|
||||||
|
# try:
|
||||||
|
# response = requests.get(url, headers=headers, timeout=10)
|
||||||
|
# response.raise_for_status()
|
||||||
|
# commits = response.json()
|
||||||
|
# if commits:
|
||||||
|
# return parser.isoparse(commits[0]["commit"]["author"]["date"])
|
||||||
|
# except Exception as e:
|
||||||
|
# logging.error(f"Erreur récupération commit schema : {e}")
|
||||||
|
# return None
|
||||||
|
#
|
||||||
|
# path_relative = f"Documents/{dossier_choisi}/{fiche_choisie}"
|
||||||
|
# commits_url = f"{GITEA_URL}/repos/{ORGANISATION}/{DEPOT_FICHES}/commits?path={path_relative}&sha={ENV}"
|
||||||
|
#
|
||||||
|
# local_mtime = datetime.fromtimestamp(os.path.getmtime(path_relative), tz=timezone.utc)
|
||||||
|
# remote_mtime = recuperer_date_dernier_commit(commit_url)
|
||||||
|
|
||||||
|
MAKE = {
|
||||||
|
"assets": {
|
||||||
|
"directory": "assets",
|
||||||
|
"seuils": {
|
||||||
|
"depends_on": "None",
|
||||||
|
},
|
||||||
|
"file": {
|
||||||
|
"seuils": "config.yaml"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"fiches": {
|
||||||
|
"directory": "Fiches",
|
||||||
|
"criticites": {
|
||||||
|
"directory": "Criticités",
|
||||||
|
"préfix": "Fiche technique ",
|
||||||
|
"depends_on": {
|
||||||
|
"gitea": {
|
||||||
|
"compare": "file2commit"
|
||||||
|
},
|
||||||
|
"document": {
|
||||||
|
"seuils": {
|
||||||
|
"where": "assets.file.seuils",
|
||||||
|
"compare": "file2file"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"file": {
|
||||||
|
"ihh": "Fiche technique IHH.md",
|
||||||
|
"isg": "Fiche technique ISG.md",
|
||||||
|
"ivc": "Fiche technique IVC.md",
|
||||||
|
"ics": "Fiche technique ICS.md"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"assemblage": {
|
||||||
|
"directory": "Assemblage",
|
||||||
|
"prefix": "Fiche assemblage ",
|
||||||
|
"depends_on": {
|
||||||
|
"gitea": {
|
||||||
|
"compare": "file2commit"
|
||||||
|
},
|
||||||
|
"document": {
|
||||||
|
"ihh": {
|
||||||
|
"where": "fiches.criticites.file.ihh",
|
||||||
|
"compare": "file2file"
|
||||||
|
},
|
||||||
|
"isg": {
|
||||||
|
"where": "fiches.criticites.file.isg",
|
||||||
|
"compare": "file2file"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"fabrication": {
|
||||||
|
"directory": "Fabrication",
|
||||||
|
"prefix": "Fiche fabrication ",
|
||||||
|
"depends_on": {
|
||||||
|
"gitea": {
|
||||||
|
"compare": "file2commit"
|
||||||
|
},
|
||||||
|
"document": {
|
||||||
|
"ihh": {
|
||||||
|
"where": "fiches.criticites.file.ihh",
|
||||||
|
"compare": "file2file"
|
||||||
|
},
|
||||||
|
"isg": {
|
||||||
|
"where": "fiches.criticites.file.isg",
|
||||||
|
"compare": "file2file"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"connexe": {
|
||||||
|
"directory": "Connexe",
|
||||||
|
"prefix": "Fiche assemblage ",
|
||||||
|
"depends_on": {
|
||||||
|
"gitea": {
|
||||||
|
"compare": "file2commit"
|
||||||
|
},
|
||||||
|
"document": {
|
||||||
|
"ihh": {
|
||||||
|
"where": "fiches.criticites.file.ihh",
|
||||||
|
"compare": "file2file"
|
||||||
|
},
|
||||||
|
"isg": {
|
||||||
|
"where": "fiches.criticites.file.isg",
|
||||||
|
"compare": "file2file"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"minerai": {
|
||||||
|
"directory": "Minerai",
|
||||||
|
"prefix": "Fiche minerai ",
|
||||||
|
"depends_on": {
|
||||||
|
"gitea": {
|
||||||
|
"compare": "file2commit"
|
||||||
|
},
|
||||||
|
"document": {
|
||||||
|
"ihh": {
|
||||||
|
"where": "fiches.criticites.file.ihh",
|
||||||
|
"compare": "file2file"
|
||||||
|
},
|
||||||
|
"isg": {
|
||||||
|
"where": "fiches.criticites.file.isg",
|
||||||
|
"compare": "file2file"
|
||||||
|
},
|
||||||
|
"ics": {
|
||||||
|
"where": "fiches.criticites.file.ics",
|
||||||
|
"compare": "file2file"
|
||||||
|
},
|
||||||
|
"ivc": {
|
||||||
|
"where": "fiches.criticites.file.ivc",
|
||||||
|
"compare": "file2file"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
106
README.md
106
README.md
@ -14,7 +14,7 @@ Le code proposé répond à la partie outillage, avec une architecture modulaire
|
|||||||
|
|
||||||
Le projet est bâti sur un backeng Gitea pour la gestion des fiches et des tickets d'évolution. (Accéder au backend)[https://fabnum-git.peccini.fr/FabNum/Fiches]
|
Le projet est bâti sur un backeng Gitea pour la gestion des fiches et des tickets d'évolution. (Accéder au backend)[https://fabnum-git.peccini.fr/FabNum/Fiches]
|
||||||
|
|
||||||
Le serveur qui héberge l'application héberge aussi le service Gitea, ce qui permet d'éliminer les temps de latence dus au réseau.
|
Le serveur qui héberge l'application héberge aussi le service Gitea, ce qui permet d'éliminer les temps de latence dus au réseau. Les fiches sont toutefois mise en cache localement en générant les fichiers markdown (référence), html (affichage) et pdf (download).
|
||||||
|
|
||||||
L'application est écrite en python et utilise majoritairement streamlit.
|
L'application est écrite en python et utilise majoritairement streamlit.
|
||||||
|
|
||||||
@ -26,11 +26,14 @@ Le fichier **requirements.txt** permet d'installer tout ce qui est nécessaire p
|
|||||||
|
|
||||||
python -m venv venv
|
python -m venv venv
|
||||||
source venv/bin/activate
|
source venv/bin/activate
|
||||||
|
# Installation depuis le fichier
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
|
# Génération du fichier
|
||||||
|
pipreqs requirements.txt
|
||||||
|
|
||||||
### Environnement
|
### Environnement
|
||||||
|
|
||||||
Le fichier **.env.local** qui contient GITEA_TOKEN n'est pas dans le dépôt car il contient la clé pour accéder au backend.
|
Le fichier **.env.local** qui contient GITEA_TOKEN n'est pas dans le dépôt car il contient la clé pour accéder au backend. Il doit donc être créé ainsi que le Token. Le TOken doit permetre l'accès en lecture au dépôt et en lecture/écriture au gestionnaire des tickets.
|
||||||
|
|
||||||
Pour l'environnement de pré-production, (https://fabnum-dev.peccini.fr)[https://fabnum-dev.peccini.fr] :
|
Pour l'environnement de pré-production, (https://fabnum-dev.peccini.fr)[https://fabnum-dev.peccini.fr] :
|
||||||
|
|
||||||
@ -52,7 +55,7 @@ Pour l'environnement de production, (https://fabnum.peccini.fr)[https://fabnum.p
|
|||||||
|
|
||||||
PORT=8501
|
PORT=8501
|
||||||
|
|
||||||
La différence entre les deux environnements se fait au travers de la configuration du proxy Nginx (ici celle de dev ; il suffit de changer dev en public pour l'environnement de production) :
|
La différence entre les deux environnements se fait au travers de la configuration du proxy Nginx (ci-après celle de dev ; il suffit de changer dev en public pour l'environnement de production) :
|
||||||
|
|
||||||
# Ajout d'un en-tête personnalisé pour indiquer l'environnement
|
# Ajout d'un en-tête personnalisé pour indiquer l'environnement
|
||||||
add_header X-Environment "dev" always;
|
add_header X-Environment "dev" always;
|
||||||
@ -106,11 +109,19 @@ Le cœur de l'application. Ce script sert de point d'entrée et d'orchestrateur
|
|||||||
|
|
||||||
Ce fichier est conçu de manière modulaire, déléguant les fonctionnalités spécifiques aux modules spécialisés, ce qui facilite la maintenance et les évolutions futures.
|
Ce fichier est conçu de manière modulaire, déléguant les fonctionnalités spécifiques aux modules spécialisés, ce qui facilite la maintenance et les évolutions futures.
|
||||||
|
|
||||||
|
Dans la sidebar, l'application indique une estimation des émissions de gaz à effet de serre. Pour cela, un mécanisme est mis en place dans la configuration Nginx pour enregistrer dans le fichier :
|
||||||
|
/var/log/nginx/fabnum-dev.access.log
|
||||||
|
les informations d'octets transférés.
|
||||||
|
|
||||||
|
Ce système est basé sur la création d'un cookie de session, utilisé ensuite pour distinguer les utilisations.
|
||||||
|
|
||||||
|
<<<<<<<<<<<à compléter avec la configuration Nginx>>>>>>>>>>>
|
||||||
|
|
||||||
## Architecture et principes de conception
|
## Architecture et principes de conception
|
||||||
|
|
||||||
### Modularité et simplification
|
### Modularité et simplification
|
||||||
|
|
||||||
L'application a été restructurée selon les principes suivants :
|
L'application a été structurée selon les principes suivants :
|
||||||
|
|
||||||
1. **Séparation des responsabilités** : Chaque module a une fonction bien définie
|
1. **Séparation des responsabilités** : Chaque module a une fonction bien définie
|
||||||
2. **Modularité** : Les fonctionnalités sont décomposées en composants réutilisables
|
2. **Modularité** : Les fonctionnalités sont décomposées en composants réutilisables
|
||||||
@ -149,43 +160,56 @@ L'application est organisée de façon modulaire, avec une structure simplifiée
|
|||||||
|
|
||||||
```
|
```
|
||||||
fabnum-dev/
|
fabnum-dev/
|
||||||
├── fabnum.py # Point d'entrée principal
|
├── fabnum.py # Point d'entrée principal
|
||||||
├── config.py # Configuration et variables d'environnement
|
├── config.py # Configuration et variables d'environnement
|
||||||
├── app/ # Modules fonctionnels principaux
|
├── app/ # Modules fonctionnels principaux
|
||||||
│ ├── analyse/ # Module d'analyse des chaînes de dépendance
|
│ ├── analyse/ # Module d'analyse des chaînes de dépendance
|
||||||
│ │ ├── interface.py # Interface utilisateur pour l'analyse
|
│ │ ├── interface.py # Interface utilisateur pour l'analyse
|
||||||
│ │ ├── sankey.py # Génération des diagrammes Sankey
|
│ │ ├── sankey.py # Génération des diagrammes Sankey
|
||||||
│ │ └── README.md # Documentation du module
|
│ │ └── README.md # Documentation du module
|
||||||
│ ├── fiches/ # Gestion et affichage des fiches
|
│ ├── fiches/ # Gestion et affichage des fiches
|
||||||
│ │ ├── interface.py # Interface utilisateur pour les fiches
|
│ │ ├── interface.py # Interface utilisateur pour les fiches
|
||||||
│ │ ├── generer.py # Génération des fiches
|
│ │ ├── generer.py # Génération des fiches
|
||||||
│ │ ├── utils/ # Utilitaires spécifiques aux fiches
|
│ │ ├── utils/ # Utilitaires spécifiques aux fiches
|
||||||
│ │ └── README.md # Documentation du module
|
│ │ ├── utils/dynamic # Gestion de la génération et affichage des fiches par type d'opération
|
||||||
│ ├── personnalisation/ # Personnalisation de la chaîne
|
│ │ ├── utils/tickets # Gestion de l'affichage et de la création des tickets
|
||||||
│ │ ├── interface.py # Interface principale
|
│ │ ├── utils/fiches_utils.py # Outils de gestion et rendu des fiches
|
||||||
│ │ ├── ajout.py # Ajout de produits
|
│ │ └── README.md # Documentation du module
|
||||||
│ │ ├── modification.py # Modification de produits
|
│ ├── personnalisation/ # Personnalisation de la chaîne
|
||||||
│ │ ├── import_export.py # Import/export de configurations
|
│ │ ├── interface.py # Interface principale
|
||||||
│ │ └── README.md # Documentation du module
|
│ │ ├── ajout.py # Ajout de produits
|
||||||
│ └── visualisations/ # Visualisations graphiques
|
│ │ ├── modification.py # Modification de produits
|
||||||
│ ├── interface.py # Interface des visualisations
|
│ │ ├── import_export.py # Import/export de configurations
|
||||||
│ ├── graphes.py # Gestion des graphes à visualiser
|
│ │ └── README.md # Documentation du module
|
||||||
│ └── README.md # Documentation du module
|
│ └── visualisations/ # Visualisations graphiques
|
||||||
├── components/ # Composants d'interface réutilisables
|
│ ├── interface.py # Interface des visualisations
|
||||||
│ ├── sidebar.py # Barre latérale de navigation
|
│ ├── graphes.py # Gestion des graphes à visualiser
|
||||||
│ ├── header.py # En-tête de l'application
|
│ └── README.md # Documentation du module
|
||||||
│ ├── footer.py # Pied de page
|
├── components/ # Composants d'interface réutilisables
|
||||||
│ └── README.md # Documentation des composants
|
│ ├── sidebar.py # Barre latérale de navigation
|
||||||
├── utils/ # Utilitaires partagés
|
│ ├── header.py # En-tête de l'application
|
||||||
│ ├── gitea.py # Connexion API Gitea
|
│ ├── footer.py # Pied de page
|
||||||
│ ├── graph_utils.py # Manipulation des graphes
|
│ ├── connexion.py # Module de connexion à partir d'un token Gitea
|
||||||
│ └── README.md # Documentation des utilitaires
|
│ └── README.md # Documentation des composants
|
||||||
├── assets/ # Ressources statiques
|
├── utils/ # Utilitaires partagés
|
||||||
│ ├── styles/ # Feuilles de style CSS
|
│ ├── gitea.py # Connexion API Gitea
|
||||||
│ └── impact_co2.js # Calcul d'impact environnemental
|
│ ├── graph_utils.py # Manipulation des graphes
|
||||||
├── .env # Configuration versionnée
|
│ ├── translations.py # Module de gestion de l'internationalisation
|
||||||
├── .env.local # Configuration locale (non versionnée)
|
│ ├── visualisations.py # Manipulation des graphes de l'onglet Visualisations
|
||||||
└── requirements.txt # Dépendances Python
|
│ └── README.md # Documentation des utilitaires
|
||||||
|
├── assets/ # Ressources statiques
|
||||||
|
│ ├── locales/ # Gestion de l'internationalisation
|
||||||
|
│ ├── styles/ # Feuilles de style CSS
|
||||||
|
│ ├── confir.yaml # Définition des seuils pour les indices
|
||||||
|
│ ├── fiches_labels.csv # Dictionnaire d'association entre les fiches et leurs labels (tickets)
|
||||||
|
│ ├── impact_co2.js # Calcul d'impact environnemental
|
||||||
|
│ ├── licence.md # Licence ajoutée à toutes les fiches
|
||||||
|
│ └── weakness.png # Icône de l'onglet dans le navigateur
|
||||||
|
├── .env # Configuration versionnée
|
||||||
|
├── .env.local # Configuration locale (non versionnée)
|
||||||
|
└── requirements.txt # Dépendances Python
|
||||||
|
├── .streamlit/ # Configuration streamlit côté serveur
|
||||||
|
│ └── config.toml # Fichier de configuration : important theme.base = light
|
||||||
```
|
```
|
||||||
|
|
||||||
Chaque module dispose de sa propre documentation détaillée dans un fichier README.md.
|
Chaque module dispose de sa propre documentation détaillée dans un fichier README.md.
|
||||||
|
|||||||
@ -1,5 +1,6 @@
|
|||||||
import streamlit as st
|
import streamlit as st
|
||||||
from utils.translations import _
|
from utils.translations import _
|
||||||
|
from utils.widgets import html_expander
|
||||||
|
|
||||||
from .sankey import afficher_sankey
|
from .sankey import afficher_sankey
|
||||||
|
|
||||||
@ -123,8 +124,7 @@ def configurer_filtres_vulnerabilite():
|
|||||||
|
|
||||||
def interface_analyse(G_temp):
|
def interface_analyse(G_temp):
|
||||||
st.markdown(f"# {str(_('pages.analyse.title'))}")
|
st.markdown(f"# {str(_('pages.analyse.title'))}")
|
||||||
with st.expander(str(_("pages.analyse.help")), expanded=False):
|
html_expander(f"{str(_('pages.analyse.help'))}", content="\n".join(_("pages.analyse.help_content")), open_by_default=False, details_class="details_introduction")
|
||||||
st.markdown("\n".join(_("pages.analyse.help_content")))
|
|
||||||
st.markdown("---")
|
st.markdown("---")
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@ -148,7 +148,7 @@ def interface_analyse(G_temp):
|
|||||||
|
|
||||||
# Lancement de l'analyse
|
# Lancement de l'analyse
|
||||||
st.markdown("---")
|
st.markdown("---")
|
||||||
if st.button(str(_("pages.analyse.run_analysis")), type="primary", key="analyse_lancer"):
|
if st.button(str(_("pages.analyse.run_analysis")), type="primary", key="analyse_lancer", icon=":material/graph_4:"):
|
||||||
afficher_sankey(
|
afficher_sankey(
|
||||||
G_temp,
|
G_temp,
|
||||||
niveau_depart=niveau_depart,
|
niveau_depart=niveau_depart,
|
||||||
|
|||||||
@ -37,14 +37,14 @@ def extraire_niveaux(G):
|
|||||||
logging.warning(f"Niveau non entier pour le noeud {node}: {niveau_str}")
|
logging.warning(f"Niveau non entier pour le noeud {node}: {niveau_str}")
|
||||||
return niveaux
|
return niveaux
|
||||||
|
|
||||||
def extraire_criticite(G, u, v):
|
def extraire_ics(G, u, v):
|
||||||
"""Extrait la criticité d'un lien entre deux nœuds"""
|
"""Extrait la criticité d'un lien entre deux nœuds"""
|
||||||
data = G.get_edge_data(u, v)
|
data = G.get_edge_data(u, v)
|
||||||
if not data:
|
if not data:
|
||||||
return 0
|
return 0
|
||||||
if isinstance(data, dict) and all(isinstance(k, int) for k in data):
|
if isinstance(data, dict) and all(isinstance(k, int) for k in data):
|
||||||
return float(data[0].get("criticite", 0))
|
return float(data[0].get("ics", 0))
|
||||||
return float(data.get("criticite", 0))
|
return float(data.get("ics", 0))
|
||||||
|
|
||||||
def extraire_chemins_selon_criteres(G, niveaux, niveau_depart, noeuds_depart, noeuds_arrivee, minerais):
|
def extraire_chemins_selon_criteres(G, niveaux, niveau_depart, noeuds_depart, noeuds_arrivee, minerais):
|
||||||
"""Extrait les chemins selon les critères spécifiés"""
|
"""Extrait les chemins selon les critères spécifiés"""
|
||||||
@ -102,7 +102,7 @@ def verifier_critere_ics(G, chemin, niveaux):
|
|||||||
|
|
||||||
if ((niveau_u == 1 and niveau_v == 2) or
|
if ((niveau_u == 1 and niveau_v == 2) or
|
||||||
(niveau_u == 1001 and niveau_v == 1002) or
|
(niveau_u == 1001 and niveau_v == 1002) or
|
||||||
(niveau_u == 10 and niveau_v in (1000, 1001))) and extraire_criticite(G, u, v) > 0.66:
|
(niveau_u == 10 and niveau_v in (1000, 1001))) and extraire_ics(G, u, v) > 0.66:
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
@ -153,7 +153,7 @@ def filtrer_chemins_par_criteres(G, chemins, niveaux, niveau_depart, niveau_arri
|
|||||||
# Vérification des critères pour ce chemin
|
# Vérification des critères pour ce chemin
|
||||||
has_ihh = filtrer_ihh and verifier_critere_ihh(G, chemin, niveaux, ihh_type)
|
has_ihh = filtrer_ihh and verifier_critere_ihh(G, chemin, niveaux, ihh_type)
|
||||||
has_ivc = filtrer_ivc and verifier_critere_ivc(G, chemin, niveaux)
|
has_ivc = filtrer_ivc and verifier_critere_ivc(G, chemin, niveaux)
|
||||||
has_criticite = filtrer_ics and verifier_critere_ics(G, chemin, niveaux)
|
has_ics = filtrer_ics and verifier_critere_ics(G, chemin, niveaux)
|
||||||
has_isg_critique = filtrer_isg and verifier_critere_isg(G, chemin, niveaux)
|
has_isg_critique = filtrer_isg and verifier_critere_isg(G, chemin, niveaux)
|
||||||
|
|
||||||
# Appliquer la logique de filtrage
|
# Appliquer la logique de filtrage
|
||||||
@ -161,12 +161,12 @@ def filtrer_chemins_par_criteres(G, chemins, niveaux, niveau_depart, niveau_arri
|
|||||||
keep = True
|
keep = True
|
||||||
if filtrer_ihh: keep = keep and has_ihh
|
if filtrer_ihh: keep = keep and has_ihh
|
||||||
if filtrer_ivc: keep = keep and has_ivc
|
if filtrer_ivc: keep = keep and has_ivc
|
||||||
if filtrer_ics: keep = keep and has_criticite
|
if filtrer_ics: keep = keep and has_ics
|
||||||
if filtrer_isg: keep = keep and has_isg_critique
|
if filtrer_isg: keep = keep and has_isg_critique
|
||||||
if keep:
|
if keep:
|
||||||
chemins_filtres.add(tuple(chemin))
|
chemins_filtres.add(tuple(chemin))
|
||||||
elif logique_filtrage == "OU":
|
elif logique_filtrage == "OU":
|
||||||
if has_ihh or has_ivc or has_criticite or has_isg_critique:
|
if has_ihh or has_ivc or has_ics or has_isg_critique:
|
||||||
chemins_filtres.add(tuple(chemin))
|
chemins_filtres.add(tuple(chemin))
|
||||||
|
|
||||||
# Extraction des liens après filtrage
|
# Extraction des liens après filtrage
|
||||||
@ -176,7 +176,7 @@ def filtrer_chemins_par_criteres(G, chemins, niveaux, niveau_depart, niveau_arri
|
|||||||
|
|
||||||
return liens_filtres, chemins_filtres
|
return liens_filtres, chemins_filtres
|
||||||
|
|
||||||
def couleur_criticite(p):
|
def couleur_ics(p):
|
||||||
"""Retourne la couleur en fonction du niveau de criticité"""
|
"""Retourne la couleur en fonction du niveau de criticité"""
|
||||||
if p <= 0.33:
|
if p <= 0.33:
|
||||||
return "darkgreen"
|
return "darkgreen"
|
||||||
@ -206,8 +206,8 @@ def preparer_donnees_sankey(G, liens_chemins, niveaux, chemins):
|
|||||||
df_liens = pd.DataFrame(list(liens_chemins), columns=["source", "target"])
|
df_liens = pd.DataFrame(list(liens_chemins), columns=["source", "target"])
|
||||||
df_liens = df_liens.groupby(["source", "target"]).size().reset_index(name="value")
|
df_liens = df_liens.groupby(["source", "target"]).size().reset_index(name="value")
|
||||||
|
|
||||||
df_liens["criticite"] = df_liens.apply(
|
df_liens["ics"] = df_liens.apply(
|
||||||
lambda row: extraire_criticite(G, row["source"], row["target"]), axis=1)
|
lambda row: extraire_ics(G, row["source"], row["target"]), axis=1)
|
||||||
df_liens["value"] = 0.1
|
df_liens["value"] = 0.1
|
||||||
|
|
||||||
# Ne garder que les nœuds effectivement connectés
|
# Ne garder que les nœuds effectivement connectés
|
||||||
@ -221,7 +221,7 @@ def preparer_donnees_sankey(G, liens_chemins, niveaux, chemins):
|
|||||||
noeuds_utilises.add(n)
|
noeuds_utilises.add(n)
|
||||||
|
|
||||||
df_liens["color"] = df_liens.apply(
|
df_liens["color"] = df_liens.apply(
|
||||||
lambda row: couleur_criticite(row["criticite"]) if row["criticite"] > 0 else "white",
|
lambda row: couleur_ics(row["ics"]) if row["ics"] > 0 else "white",
|
||||||
axis=1
|
axis=1
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@ -2,7 +2,6 @@
|
|||||||
import streamlit as st
|
import streamlit as st
|
||||||
import requests
|
import requests
|
||||||
import os
|
import os
|
||||||
import pathlib
|
|
||||||
from utils.translations import _
|
from utils.translations import _
|
||||||
|
|
||||||
from .utils.tickets.display import afficher_tickets_par_fiche
|
from .utils.tickets.display import afficher_tickets_par_fiche
|
||||||
@ -16,11 +15,11 @@ from utils.gitea import charger_arborescence_fiches
|
|||||||
from .utils.fiche_utils import load_seuils, doit_regenerer_fiche
|
from .utils.fiche_utils import load_seuils, doit_regenerer_fiche
|
||||||
|
|
||||||
from .generer import generer_fiche
|
from .generer import generer_fiche
|
||||||
|
from utils.widgets import html_expander
|
||||||
|
|
||||||
def interface_fiches():
|
def interface_fiches():
|
||||||
st.markdown(f"# {str(_('pages.fiches.title'))}")
|
st.markdown(f"# {str(_('pages.fiches.title'))}")
|
||||||
with st.expander(str(_("pages.fiches.help")), expanded=False):
|
html_expander(f"{str(_('pages.fiches.help'))}", content="\n".join(_("pages.fiches.help_content")), open_by_default=False, details_class="details_introduction")
|
||||||
st.markdown("\n".join(_("pages.fiches.help_content")))
|
|
||||||
st.markdown("---")
|
st.markdown("---")
|
||||||
|
|
||||||
if "fiches_arbo" not in st.session_state:
|
if "fiches_arbo" not in st.session_state:
|
||||||
@ -102,4 +101,4 @@ def interface_fiches():
|
|||||||
formulaire_creation_ticket_dynamique(fiche_choisie)
|
formulaire_creation_ticket_dynamique(fiche_choisie)
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
st.error(f"{str(_('pages.fiches.loading_error', 'Erreur lors du chargement de la fiche :'))} {e}")
|
st.error(f"{str(_('pages.fiches.loading_error'))} {e}")
|
||||||
|
|||||||
@ -567,6 +567,8 @@ def build_minerai_sections(md: str) -> str:
|
|||||||
md,
|
md,
|
||||||
flags=re.DOTALL
|
flags=re.DOTALL
|
||||||
)
|
)
|
||||||
|
# suppression pour le dernier minerai dans la fiche IHH
|
||||||
|
md = re.sub(r"# Tableaux de synthèse.*<!---- AUTO-END:SECTION-IHH-TRAITEMENT -->", "", md, flags=re.DOTALL)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
st.error(f"Erreur lors de la génération des sections IHH: {e}")
|
st.error(f"Erreur lors de la génération des sections IHH: {e}")
|
||||||
|
|
||||||
|
|||||||
@ -110,10 +110,6 @@ def creer_ticket_gitea(titre, corps, labels):
|
|||||||
|
|
||||||
reponse = gitea_request("post", url, headers={"Content-Type": "application/json"}, data=json.dumps(data))
|
reponse = gitea_request("post", url, headers={"Content-Type": "application/json"}, data=json.dumps(data))
|
||||||
if not reponse:
|
if not reponse:
|
||||||
return
|
return False
|
||||||
|
|
||||||
issue_url = reponse.json().get("html_url", "")
|
|
||||||
if issue_url:
|
|
||||||
st.success(f"{str(_('pages.fiches.tickets.created_success'))} [Voir le ticket]({issue_url})")
|
|
||||||
else:
|
else:
|
||||||
st.success(str(_('pages.fiches.tickets.created')))
|
return True
|
||||||
|
|||||||
@ -74,6 +74,8 @@ def afficher_controles_formulaire():
|
|||||||
col1, col2 = st.columns(2)
|
col1, col2 = st.columns(2)
|
||||||
if col1.button(str(_("pages.fiches.tickets.preview"))):
|
if col1.button(str(_("pages.fiches.tickets.preview"))):
|
||||||
st.session_state.previsualiser = True
|
st.session_state.previsualiser = True
|
||||||
|
# S'assurer que l'expander reste ouvert en mode prévisualisation
|
||||||
|
st.session_state.expander_state = True
|
||||||
if col2.button(str(_("pages.fiches.tickets.cancel"))):
|
if col2.button(str(_("pages.fiches.tickets.cancel"))):
|
||||||
st.session_state.previsualiser = False
|
st.session_state.previsualiser = False
|
||||||
st.rerun()
|
st.rerun()
|
||||||
@ -81,6 +83,23 @@ def afficher_controles_formulaire():
|
|||||||
|
|
||||||
def gerer_previsualisation_et_soumission(reponses, labels, selected_ops, cible):
|
def gerer_previsualisation_et_soumission(reponses, labels, selected_ops, cible):
|
||||||
"""Gère la prévisualisation et la soumission du ticket."""
|
"""Gère la prévisualisation et la soumission du ticket."""
|
||||||
|
# Si nous avons tenté de créer un ticket (succès ou erreur)
|
||||||
|
if st.session_state.get("ticket_cree", False) or st.session_state.get("ticket_erreur", False):
|
||||||
|
if st.session_state.get("ticket_cree", False):
|
||||||
|
st.success(str(_("pages.fiches.tickets.created_success")))
|
||||||
|
else:
|
||||||
|
st.error(str(_("pages.fiches.tickets.creation_error")))
|
||||||
|
|
||||||
|
if st.button(str(_("pages.fiches.tickets.continue"))):
|
||||||
|
# Réinitialiser le formulaire et cacher l'expander
|
||||||
|
st.session_state.ticket_cree = False
|
||||||
|
st.session_state.ticket_erreur = False
|
||||||
|
st.session_state.previsualiser = False
|
||||||
|
st.session_state.expander_state = False
|
||||||
|
st.rerun()
|
||||||
|
return
|
||||||
|
|
||||||
|
# Si nous ne sommes pas en mode prévisualisation, ne rien afficher
|
||||||
if not st.session_state.get("previsualiser", False):
|
if not st.session_state.get("previsualiser", False):
|
||||||
return
|
return
|
||||||
|
|
||||||
@ -101,15 +120,31 @@ def gerer_previsualisation_et_soumission(reponses, labels, selected_ops, cible):
|
|||||||
labels_ids.append(labels_existants["Backlog"])
|
labels_ids.append(labels_existants["Backlog"])
|
||||||
|
|
||||||
corps = construire_corps_ticket_markdown(reponses)
|
corps = construire_corps_ticket_markdown(reponses)
|
||||||
creer_ticket_gitea(titre_ticket, corps, labels_ids)
|
resultat = creer_ticket_gitea(titre_ticket, corps, labels_ids)
|
||||||
|
|
||||||
|
# Marquer le résultat et ouvrir l'expander pour afficher le résultat
|
||||||
|
st.session_state.ticket_cree = resultat
|
||||||
|
st.session_state.ticket_erreur = not resultat
|
||||||
st.session_state.previsualiser = False
|
st.session_state.previsualiser = False
|
||||||
st.success(str(_("pages.fiches.tickets.created")))
|
st.session_state.expander_state = True
|
||||||
|
st.rerun()
|
||||||
|
|
||||||
|
|
||||||
def formulaire_creation_ticket_dynamique(fiche_selectionnee):
|
def formulaire_creation_ticket_dynamique(fiche_selectionnee):
|
||||||
"""Fonction principale pour le formulaire de création de ticket."""
|
"""Fonction principale pour le formulaire de création de ticket."""
|
||||||
with st.expander(str(_("pages.fiches.tickets.create_new")), expanded=False):
|
# Initialiser l'état de l'expander si ce n'est pas déjà fait
|
||||||
|
if "expander_state" not in st.session_state:
|
||||||
|
st.session_state.expander_state = False
|
||||||
|
|
||||||
|
with st.expander(str(_("pages.fiches.tickets.create_new")), expanded=st.session_state.expander_state):
|
||||||
|
# Initialiser les états si ce n'est pas déjà fait
|
||||||
|
if "ticket_cree" not in st.session_state:
|
||||||
|
st.session_state.ticket_cree = False
|
||||||
|
if "ticket_erreur" not in st.session_state:
|
||||||
|
st.session_state.ticket_erreur = False
|
||||||
|
if "previsualiser" not in st.session_state:
|
||||||
|
st.session_state.previsualiser = False
|
||||||
|
|
||||||
# Chargement et vérification du modèle
|
# Chargement et vérification du modèle
|
||||||
contenu_modele = charger_modele_ticket()
|
contenu_modele = charger_modele_ticket()
|
||||||
if not contenu_modele:
|
if not contenu_modele:
|
||||||
@ -119,11 +154,21 @@ def formulaire_creation_ticket_dynamique(fiche_selectionnee):
|
|||||||
# Traitement du modèle et génération du formulaire
|
# Traitement du modèle et génération du formulaire
|
||||||
sections = parser_modele_ticket(contenu_modele)
|
sections = parser_modele_ticket(contenu_modele)
|
||||||
labels, selected_ops, cible = generer_labels(fiche_selectionnee)
|
labels, selected_ops, cible = generer_labels(fiche_selectionnee)
|
||||||
reponses = creer_champs_formulaire(sections, fiche_selectionnee)
|
|
||||||
|
|
||||||
# Gestion des contrôles et de la prévisualisation
|
# Créer le formulaire et gérer ses états
|
||||||
afficher_controles_formulaire()
|
if st.session_state.ticket_cree or st.session_state.ticket_erreur:
|
||||||
gerer_previsualisation_et_soumission(reponses, labels, selected_ops, cible)
|
# Si le ticket a été créé ou a échoué, afficher le message approprié et le bouton continuer
|
||||||
|
gerer_previsualisation_et_soumission({}, labels, selected_ops, cible)
|
||||||
|
else:
|
||||||
|
# Sinon afficher le formulaire normal
|
||||||
|
reponses = creer_champs_formulaire(sections, fiche_selectionnee)
|
||||||
|
|
||||||
|
# Afficher les contrôles uniquement si nous ne sommes pas en mode prévisualisation
|
||||||
|
if not st.session_state.previsualiser:
|
||||||
|
afficher_controles_formulaire()
|
||||||
|
|
||||||
|
# Gérer la prévisualisation et soumission
|
||||||
|
gerer_previsualisation_et_soumission(reponses, labels, selected_ops, cible)
|
||||||
|
|
||||||
|
|
||||||
def charger_modele_ticket():
|
def charger_modele_ticket():
|
||||||
|
|||||||
@ -6,7 +6,6 @@ import re
|
|||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from dateutil import parser
|
from dateutil import parser
|
||||||
from utils.translations import _
|
from utils.translations import _
|
||||||
from .core import rechercher_tickets_gitea
|
|
||||||
|
|
||||||
|
|
||||||
def extraire_statut_par_label(ticket):
|
def extraire_statut_par_label(ticket):
|
||||||
|
|||||||
42
app/ia_nalyse/README.md
Normal file
42
app/ia_nalyse/README.md
Normal file
@ -0,0 +1,42 @@
|
|||||||
|
# Module d'Analyse
|
||||||
|
|
||||||
|
Ce module permet d'analyser les relations entre les différentes parties de la chaîne de fabrication du numérique. Il offre des outils pour visualiser les flux et identifier les vulnérabilités potentielles dans la chaîne d'approvisionnement.
|
||||||
|
|
||||||
|
## Structure du module
|
||||||
|
|
||||||
|
Le module d'analyse comprend deux composants principaux :
|
||||||
|
|
||||||
|
- **interface.py** : Gère l'interface utilisateur pour paramétrer les analyses
|
||||||
|
- **sankey.py** : Génère les diagrammes de flux (Sankey) pour visualiser les relations entre les éléments
|
||||||
|
|
||||||
|
## Fonctionnalités
|
||||||
|
|
||||||
|
### Interface d'analyse
|
||||||
|
L'interface permet de :
|
||||||
|
- Sélectionner les niveaux de départ et d'arrivée pour l'analyse (produits, composants, minerais, opérations, etc.)
|
||||||
|
- Filtrer les données par minerais spécifiques
|
||||||
|
- Effectuer une sélection fine des nœuds de départ et d'arrivée
|
||||||
|
- Appliquer des filtres pour identifier les vulnérabilités :
|
||||||
|
- Filtres ICS (criticité pour un composant)
|
||||||
|
- Filtres IVC (criticité par rapport à la concurrence sectorielle)
|
||||||
|
- Filtres IHH (concentration géographique ou industrielle)
|
||||||
|
- Filtres ISG (instabilité des pays)
|
||||||
|
- Choisir la logique de filtrage (OU, ET)
|
||||||
|
|
||||||
|
### Visualisation Sankey
|
||||||
|
Le module génère des diagrammes Sankey qui :
|
||||||
|
- Affichent les flux entre les différents éléments de la chaîne
|
||||||
|
- Mettent en évidence les relations de dépendance
|
||||||
|
- Permettent d'identifier visuellement les goulots d'étranglement potentiels
|
||||||
|
- Sont interactifs et permettent d'explorer la chaîne de valeur
|
||||||
|
|
||||||
|
## Utilisation
|
||||||
|
|
||||||
|
1. Sélectionnez un niveau de départ (ex : Produit final, Composant)
|
||||||
|
2. Choisissez un niveau d'arrivée (ex : Pays géographique, Acteur d'opération)
|
||||||
|
3. Si nécessaire, filtrez par minerais spécifiques
|
||||||
|
4. Affinez votre sélection avec des nœuds de départ et d'arrivée spécifiques
|
||||||
|
5. Appliquez les filtres de vulnérabilité souhaités
|
||||||
|
6. Lancez l'analyse pour générer le diagramme Sankey
|
||||||
|
|
||||||
|
Le diagramme résultant permet d'identifier visuellement les relations et points de vulnérabilité dans la chaîne d'approvisionnement du numérique.
|
||||||
2
app/ia_nalyse/__init__.py
Normal file
2
app/ia_nalyse/__init__.py
Normal file
@ -0,0 +1,2 @@
|
|||||||
|
# __init__.py – app/fiches
|
||||||
|
from .interface import interface_ia_nalyse
|
||||||
195
app/ia_nalyse/interface.py
Normal file
195
app/ia_nalyse/interface.py
Normal file
@ -0,0 +1,195 @@
|
|||||||
|
import streamlit as st
|
||||||
|
import networkx as nx
|
||||||
|
from utils.translations import _
|
||||||
|
from utils.widgets import html_expander
|
||||||
|
|
||||||
|
from utils.graph_utils import (
|
||||||
|
extraire_chemins_depuis,
|
||||||
|
extraire_chemins_vers
|
||||||
|
)
|
||||||
|
|
||||||
|
from batch_ia.batch_utils import soumettre_batch, statut_utilisateur, nettoyage_post_telechargement
|
||||||
|
|
||||||
|
niveau_labels = {
|
||||||
|
0: "Produit final",
|
||||||
|
1: "Composant",
|
||||||
|
2: "Minerai",
|
||||||
|
10: "Opération",
|
||||||
|
11: "Pays d'opération",
|
||||||
|
12: "Acteur d'opération",
|
||||||
|
99: "Pays géographique"
|
||||||
|
}
|
||||||
|
|
||||||
|
inverse_niveau_labels = {v: k for k, v in niveau_labels.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def preparer_graphe(G):
|
||||||
|
"""Nettoie et prépare le graphe pour l'analyse."""
|
||||||
|
niveaux_temp = {
|
||||||
|
node: int(str(attrs.get("niveau")).strip('"'))
|
||||||
|
for node, attrs in G.nodes(data=True)
|
||||||
|
if attrs.get("niveau") and str(attrs.get("niveau")).strip('"').isdigit()
|
||||||
|
}
|
||||||
|
G.remove_nodes_from([n for n in G.nodes() if n not in niveaux_temp])
|
||||||
|
G.remove_nodes_from(
|
||||||
|
[n for n in G.nodes() if niveaux_temp.get(n) == 10 and 'Reserves' in n])
|
||||||
|
return G, niveaux_temp
|
||||||
|
|
||||||
|
|
||||||
|
def selectionner_minerais(G):
|
||||||
|
"""Interface pour sélectionner les minerais si nécessaire."""
|
||||||
|
minerais_selection = None
|
||||||
|
|
||||||
|
st.markdown(f"## {str(_('pages.ia_nalyse.select_minerals'))}")
|
||||||
|
# Tous les nœuds de niveau 2 (minerai)
|
||||||
|
minerais_nodes = sorted([
|
||||||
|
n for n, d in G.nodes(data=True)
|
||||||
|
if d.get("niveau") and int(str(d.get("niveau")).strip('"')) == 2
|
||||||
|
])
|
||||||
|
|
||||||
|
minerais_selection = st.multiselect(
|
||||||
|
str(_("pages.ia_nalyse.filter_by_minerals")),
|
||||||
|
minerais_nodes,
|
||||||
|
key="analyse_minerais"
|
||||||
|
)
|
||||||
|
|
||||||
|
return minerais_selection
|
||||||
|
|
||||||
|
|
||||||
|
def selectionner_noeuds(G, niveaux_temp, niveau_depart):
|
||||||
|
"""Interface pour sélectionner les nœuds spécifiques de départ et d'arrivée."""
|
||||||
|
st.markdown("---")
|
||||||
|
st.markdown(f"## {str(_('pages.ia_nalyse.fine_selection'))}")
|
||||||
|
|
||||||
|
depart_nodes = [n for n in G.nodes() if niveaux_temp.get(n) == niveau_depart]
|
||||||
|
noeuds_arrivee = [n for n in G.nodes() if niveaux_temp.get(n) == 99]
|
||||||
|
|
||||||
|
noeuds_depart = st.multiselect(str(_("pages.ia_nalyse.filter_start_nodes")),
|
||||||
|
sorted(depart_nodes),
|
||||||
|
key="analyse_noeuds_depart")
|
||||||
|
|
||||||
|
noeuds_depart = noeuds_depart if noeuds_depart else None
|
||||||
|
|
||||||
|
return noeuds_depart, noeuds_arrivee
|
||||||
|
|
||||||
|
def extraire_niveaux(G):
|
||||||
|
"""Extrait les niveaux des nœuds du graphe"""
|
||||||
|
niveaux = {}
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
niveau_str = attrs.get("niveau")
|
||||||
|
if niveau_str:
|
||||||
|
niveaux[node] = int(str(niveau_str).strip('"'))
|
||||||
|
return niveaux
|
||||||
|
|
||||||
|
def extraire_chemins_selon_criteres(G, niveaux, niveau_depart, noeuds_depart, noeuds_arrivee, minerais):
|
||||||
|
"""Extrait les chemins selon les critères spécifiés"""
|
||||||
|
chemins = []
|
||||||
|
if noeuds_depart and noeuds_arrivee:
|
||||||
|
for nd in noeuds_depart:
|
||||||
|
for na in noeuds_arrivee:
|
||||||
|
tous_chemins = extraire_chemins_depuis(G, nd)
|
||||||
|
chemins.extend([chemin for chemin in tous_chemins if na in chemin])
|
||||||
|
elif noeuds_depart:
|
||||||
|
for nd in noeuds_depart:
|
||||||
|
chemins.extend(extraire_chemins_depuis(G, nd))
|
||||||
|
elif noeuds_arrivee:
|
||||||
|
for na in noeuds_arrivee:
|
||||||
|
chemins.extend(extraire_chemins_vers(G, na, niveau_depart))
|
||||||
|
else:
|
||||||
|
sources_depart = [n for n in G.nodes() if niveaux.get(n) == niveau_depart]
|
||||||
|
for nd in sources_depart:
|
||||||
|
chemins.extend(extraire_chemins_depuis(G, nd))
|
||||||
|
|
||||||
|
if minerais:
|
||||||
|
chemins = [chemin for chemin in chemins if any(n in minerais for n in chemin)]
|
||||||
|
|
||||||
|
return chemins
|
||||||
|
|
||||||
|
def exporter_graphe_filtre(G, liens_chemins):
|
||||||
|
"""Gère l'export du graphe filtré au format DOT"""
|
||||||
|
if not st.session_state.get("logged_in", False) or not liens_chemins:
|
||||||
|
return
|
||||||
|
|
||||||
|
G_export = nx.DiGraph()
|
||||||
|
for u, v in liens_chemins:
|
||||||
|
G_export.add_node(u, **G.nodes[u])
|
||||||
|
G_export.add_node(v, **G.nodes[v])
|
||||||
|
data = G.get_edge_data(u, v)
|
||||||
|
if isinstance(data, dict) and all(isinstance(k, int) for k in data):
|
||||||
|
G_export.add_edge(u, v, **data[0])
|
||||||
|
elif isinstance(data, dict):
|
||||||
|
G_export.add_edge(u, v, **data)
|
||||||
|
else:
|
||||||
|
G_export.add_edge(u, v)
|
||||||
|
|
||||||
|
return(G_export)
|
||||||
|
|
||||||
|
def extraire_liens_filtres(chemins, niveaux, niveau_depart, niveau_arrivee, niveaux_speciaux):
|
||||||
|
"""Extrait les liens des chemins en respectant les niveaux"""
|
||||||
|
liens = set()
|
||||||
|
for chemin in chemins:
|
||||||
|
for i in range(len(chemin) - 1):
|
||||||
|
u, v = chemin[i], chemin[i + 1]
|
||||||
|
niveau_u = niveaux.get(u, 999)
|
||||||
|
niveau_v = niveaux.get(v, 999)
|
||||||
|
if (
|
||||||
|
(niveau_depart <= niveau_u <= niveau_arrivee or niveau_u in niveaux_speciaux)
|
||||||
|
and (niveau_depart <= niveau_v <= niveau_arrivee or niveau_v in niveaux_speciaux)
|
||||||
|
):
|
||||||
|
liens.add((u, v))
|
||||||
|
return liens
|
||||||
|
|
||||||
|
def interface_ia_nalyse(G_temp):
|
||||||
|
st.markdown(f"# {str(_('pages.ia_nalyse.title'))}")
|
||||||
|
html_expander(f"{str(_('pages.ia_nalyse.help'))}", content="\n".join(_("pages.ia_nalyse.help_content")), open_by_default=False, details_class="details_introduction")
|
||||||
|
st.markdown("---")
|
||||||
|
|
||||||
|
resultat = statut_utilisateur(st.session_state.username)
|
||||||
|
st.info(resultat["message"])
|
||||||
|
|
||||||
|
if resultat["statut"] is None:
|
||||||
|
# Préparation du graphe
|
||||||
|
G_temp, niveaux_temp = preparer_graphe(G_temp)
|
||||||
|
|
||||||
|
# Sélection des niveaux
|
||||||
|
niveau_depart = 0
|
||||||
|
niveau_arrivee = 99
|
||||||
|
|
||||||
|
# Sélection fine des noeuds
|
||||||
|
noeuds_depart, noeuds_arrivee = selectionner_noeuds(G_temp, niveaux_temp, niveau_depart)
|
||||||
|
|
||||||
|
# Sélection des minerais si nécessaire
|
||||||
|
minerais = selectionner_minerais(G_temp)
|
||||||
|
|
||||||
|
# Étape 1 : Extraction des niveaux des nœuds
|
||||||
|
niveaux = extraire_niveaux(G_temp)
|
||||||
|
|
||||||
|
# Étape 2 : Extraction des chemins selon les critères
|
||||||
|
chemins = extraire_chemins_selon_criteres(G_temp, niveaux, niveau_depart, noeuds_depart, noeuds_arrivee, minerais)
|
||||||
|
|
||||||
|
niveaux_speciaux = [1000, 1001, 1002, 1010, 1011, 1012]
|
||||||
|
# Extraction des liens sans filtrage
|
||||||
|
liens_chemins = extraire_liens_filtres(chemins, niveaux, niveau_depart, niveau_arrivee, niveaux_speciaux)
|
||||||
|
|
||||||
|
if liens_chemins:
|
||||||
|
G_final = exporter_graphe_filtre(G_temp, liens_chemins)
|
||||||
|
if st.button(str(_("pages.ia_nalyse.submit_request")), icon=":material/send:"):
|
||||||
|
soumettre_batch(st.session_state.username, G_final)
|
||||||
|
st.rerun()
|
||||||
|
else:
|
||||||
|
st.info(str(_("pages.ia_nalyse.empty_graph")))
|
||||||
|
|
||||||
|
elif resultat["statut"] == "terminé" and resultat["telechargement"]:
|
||||||
|
if not st.session_state.get("telechargement_confirme"):
|
||||||
|
st.download_button(str(_("buttons.download")), resultat["telechargement"], file_name="analyse.zip", icon=":material/download:")
|
||||||
|
if st.button(str(_("pages.ia_nalyse.confirm_download")), icon=":material/task_alt:"):
|
||||||
|
nettoyage_post_telechargement(st.session_state.username)
|
||||||
|
st.session_state["telechargement_confirme"] = True
|
||||||
|
st.rerun()
|
||||||
|
else:
|
||||||
|
st.success("Résultat supprimé. Vous pouvez relancer une nouvelle analyse.")
|
||||||
|
if st.button(str(_("buttons.refresh")), icon=":material/refresh:"):
|
||||||
|
st.rerun()
|
||||||
|
else:
|
||||||
|
if st.button(str(_("buttons.refresh")), icon=":material/refresh:"):
|
||||||
|
st.rerun()
|
||||||
@ -4,7 +4,7 @@ from utils.translations import get_translation as _
|
|||||||
|
|
||||||
def importer_exporter_graph(G):
|
def importer_exporter_graph(G):
|
||||||
st.markdown(f"## {_('pages.personnalisation.save_restore_config')}")
|
st.markdown(f"## {_('pages.personnalisation.save_restore_config')}")
|
||||||
if st.button(str(_("pages.personnalisation.export_config"))):
|
if st.button(str(_("pages.personnalisation.export_config")), icon=":material/save:"):
|
||||||
nodes = [n for n, d in G.nodes(data=True) if d.get("personnalisation") == "oui"]
|
nodes = [n for n, d in G.nodes(data=True) if d.get("personnalisation") == "oui"]
|
||||||
edges = [(u, v) for u, v in G.edges() if u in nodes]
|
edges = [(u, v) for u, v in G.edges() if u in nodes]
|
||||||
conf = {"nodes": nodes, "edges": edges}
|
conf = {"nodes": nodes, "edges": edges}
|
||||||
@ -13,7 +13,8 @@ def importer_exporter_graph(G):
|
|||||||
label=str(_("pages.personnalisation.download_json")),
|
label=str(_("pages.personnalisation.download_json")),
|
||||||
data=json_str,
|
data=json_str,
|
||||||
file_name="config_personnalisation.json",
|
file_name="config_personnalisation.json",
|
||||||
mime="application/json"
|
mime="application/json",
|
||||||
|
icon=":material/save:"
|
||||||
)
|
)
|
||||||
|
|
||||||
uploaded = st.file_uploader(str(_("pages.personnalisation.import_config")), type=["json"])
|
uploaded = st.file_uploader(str(_("pages.personnalisation.import_config")), type=["json"])
|
||||||
@ -37,7 +38,7 @@ def importer_exporter_graph(G):
|
|||||||
key="restaurer_selection"
|
key="restaurer_selection"
|
||||||
)
|
)
|
||||||
|
|
||||||
if st.button(str(_("pages.personnalisation.restore_selected")), type="primary"):
|
if st.button(str(_("pages.personnalisation.restore_selected")), type="primary", icon=":material/history:"):
|
||||||
for node in sel_nodes:
|
for node in sel_nodes:
|
||||||
if not G.has_node(node):
|
if not G.has_node(node):
|
||||||
G.add_node(node, niveau="0", personnalisation="oui", label=node)
|
G.add_node(node, niveau="0", personnalisation="oui", label=node)
|
||||||
|
|||||||
@ -5,11 +5,11 @@ from utils.translations import _
|
|||||||
from .ajout import ajouter_produit
|
from .ajout import ajouter_produit
|
||||||
from .modification import modifier_produit
|
from .modification import modifier_produit
|
||||||
from .import_export import importer_exporter_graph
|
from .import_export import importer_exporter_graph
|
||||||
|
from utils.widgets import html_expander
|
||||||
|
|
||||||
def interface_personnalisation(G):
|
def interface_personnalisation(G):
|
||||||
st.markdown(f"# {str(_('pages.personnalisation.title'))}")
|
st.markdown(f"# {str(_('pages.personnalisation.title'))}")
|
||||||
with st.expander(str(_("pages.personnalisation.help")), expanded=False):
|
html_expander(f"{str(_('pages.personnalisation.help'))}", content="\n".join(_("pages.personnalisation.help_content")), open_by_default=False, details_class="details_introduction")
|
||||||
st.markdown("\n".join(_("pages.personnalisation.help_content")))
|
|
||||||
st.markdown("---")
|
st.markdown("---")
|
||||||
|
|
||||||
G = ajouter_produit(G)
|
G = ajouter_produit(G)
|
||||||
|
|||||||
2
app/plan_d_action/__init__.py
Normal file
2
app/plan_d_action/__init__.py
Normal file
@ -0,0 +1,2 @@
|
|||||||
|
# __init__.py – app/fiches
|
||||||
|
from .interface import interface_plan_d_action
|
||||||
120
app/plan_d_action/interface.py
Normal file
120
app/plan_d_action/interface.py
Normal file
@ -0,0 +1,120 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
"""
|
||||||
|
Script pour générer un rapport factorisé des vulnérabilités critiques
|
||||||
|
suivant la structure définie dans Remarques.md.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import streamlit as st
|
||||||
|
import uuid
|
||||||
|
from utils.translations import _
|
||||||
|
from utils.widgets import html_expander
|
||||||
|
from networkx.drawing.nx_agraph import write_dot
|
||||||
|
|
||||||
|
from batch_ia import (
|
||||||
|
load_config,
|
||||||
|
write_report,
|
||||||
|
parse_graphs,
|
||||||
|
extract_data_from_graph,
|
||||||
|
calculate_vulnerabilities,
|
||||||
|
generate_report,
|
||||||
|
)
|
||||||
|
|
||||||
|
from app.plan_d_action.utils.data.plan_d_action import initialiser_interface
|
||||||
|
|
||||||
|
from app.plan_d_action.utils.interface.parser import preparer_graphe
|
||||||
|
from app.plan_d_action.utils.interface.niveau_utils import extraire_niveaux
|
||||||
|
from app.plan_d_action.utils.interface.selection import (
|
||||||
|
selectionner_minerais,
|
||||||
|
selectionner_noeuds,
|
||||||
|
extraire_chemins_selon_criteres
|
||||||
|
)
|
||||||
|
from app.plan_d_action.utils.interface.export import (
|
||||||
|
exporter_graphe_filtre,
|
||||||
|
extraire_liens_filtres
|
||||||
|
)
|
||||||
|
from app.plan_d_action.utils.interface.visualization import remplacer_par_badge
|
||||||
|
from app.plan_d_action.utils.interface.config import (
|
||||||
|
niveau_labels,
|
||||||
|
JOBS
|
||||||
|
)
|
||||||
|
|
||||||
|
inverse_niveau_labels = {v: k for k, v in niveau_labels.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def interface_plan_d_action(G_temp):
|
||||||
|
|
||||||
|
if "sel_prod" not in st.session_state:
|
||||||
|
st.session_state.sel_prod = None
|
||||||
|
if "sel_comp" not in st.session_state:
|
||||||
|
st.session_state.sel_comp = None
|
||||||
|
if "sel_miner" not in st.session_state:
|
||||||
|
st.session_state.sel_miner = None
|
||||||
|
|
||||||
|
if "plan_d_action" not in st.session_state:
|
||||||
|
st.session_state["plan_d_action"] = 0
|
||||||
|
if "g_md_done" not in st.session_state:
|
||||||
|
st.session_state["g_md_done"] = False
|
||||||
|
if "uuid" not in st.session_state:
|
||||||
|
st.session_state["uuid"] = str(uuid.uuid4())[:8]
|
||||||
|
st.session_state["G_dot"] = f"{JOBS}/{st.session_state["uuid"]}.dot"
|
||||||
|
st.session_state["G_md"] = f"{JOBS}/{st.session_state["uuid"]}.md"
|
||||||
|
|
||||||
|
if st.session_state["plan_d_action"] == 0:
|
||||||
|
st.markdown(f"# {str(_('pages.plan_d_action.title'))}")
|
||||||
|
html_expander(f"{str(_('pages.plan_d_action.help'))}", content="\n".join(_("pages.plan_d_action.help_content")), open_by_default=False, details_class="details_introduction")
|
||||||
|
# Préparation du graphe
|
||||||
|
G_temp, niveaux_temp = preparer_graphe(G_temp)
|
||||||
|
|
||||||
|
# Sélection des niveaux
|
||||||
|
niveau_depart = 0
|
||||||
|
niveau_arrivee = 99
|
||||||
|
# Sélection fine des noeuds
|
||||||
|
noeuds_depart, noeuds_arrivee = selectionner_noeuds(G_temp, niveaux_temp, niveau_depart)
|
||||||
|
# Sélection des minerais si nécessaire
|
||||||
|
if noeuds_depart:
|
||||||
|
minerais = selectionner_minerais(G_temp, noeuds_depart)
|
||||||
|
# Étape 1 : Extraction des niveaux des nœuds
|
||||||
|
niveaux = extraire_niveaux(G_temp)
|
||||||
|
# Étape 2 : Extraction des chemins selon les critères
|
||||||
|
chemins = extraire_chemins_selon_criteres(G_temp, niveaux, niveau_depart, noeuds_depart, noeuds_arrivee, minerais)
|
||||||
|
niveaux_speciaux = [1000, 1001, 1002, 1010, 1011, 1012]
|
||||||
|
# Extraction des liens sans filtrage
|
||||||
|
liens_chemins = extraire_liens_filtres(chemins, niveaux, niveau_depart, niveau_arrivee, niveaux_speciaux)
|
||||||
|
else:
|
||||||
|
liens_chemins = None
|
||||||
|
|
||||||
|
if liens_chemins:
|
||||||
|
G_final = exporter_graphe_filtre(G_temp, liens_chemins)
|
||||||
|
st.session_state["G_final"] = G_final
|
||||||
|
# formulaire ou sélection
|
||||||
|
if st.button(str(_("pages.plan_d_action.submit_request")), icon=":material/send:"):
|
||||||
|
# On déclenche la suite — mais on NE traite rien maintenant
|
||||||
|
st.session_state["plan_d_action"] = 1
|
||||||
|
st.rerun() # force la réexécution immédiatement avec état mis à jour
|
||||||
|
|
||||||
|
elif st.session_state["plan_d_action"] == 1:
|
||||||
|
st.markdown("")
|
||||||
|
# Traitement lourd une seule fois
|
||||||
|
if not st.session_state["g_md_done"]:
|
||||||
|
write_dot(st.session_state["G_final"], st.session_state["G_dot"])
|
||||||
|
config = load_config()
|
||||||
|
graph, ref_graph = parse_graphs(st.session_state["G_dot"])
|
||||||
|
data = extract_data_from_graph(graph, ref_graph)
|
||||||
|
results = calculate_vulnerabilities(data, config)
|
||||||
|
report, file_names = generate_report(data, results, config)
|
||||||
|
write_report(remplacer_par_badge(report), st.session_state["G_md"])
|
||||||
|
st.session_state["g_md_done"] = True # pour ne pas re-traiter à chaque affichage
|
||||||
|
|
||||||
|
# Affichage de l’interface Streamlit
|
||||||
|
initialiser_interface(st.session_state["G_md"])
|
||||||
|
if (st.button("Réinitialiser", icon=":material/refresh:")):
|
||||||
|
st.session_state["plan_d_action"] = 0
|
||||||
|
st.session_state["g_md_done"] = False
|
||||||
|
st.session_state.sel_prod = None
|
||||||
|
st.session_state.sel_comp = None
|
||||||
|
st.session_state.sel_miner = None
|
||||||
|
for f in JOBS.glob(f"*{st.session_state["uuid"]}*"):
|
||||||
|
if f.is_file():
|
||||||
|
f.unlink()
|
||||||
|
st.rerun()
|
||||||
8
app/plan_d_action/utils/data/__init__.py
Normal file
8
app/plan_d_action/utils/data/__init__.py
Normal file
@ -0,0 +1,8 @@
|
|||||||
|
from .config import (
|
||||||
|
PRECONISATIONS,
|
||||||
|
INDICATEURS
|
||||||
|
)
|
||||||
|
from .data_utils import(
|
||||||
|
colorer_couleurs,
|
||||||
|
set_vulnerability
|
||||||
|
)
|
||||||
165
app/plan_d_action/utils/data/config.py
Normal file
165
app/plan_d_action/utils/data/config.py
Normal file
@ -0,0 +1,165 @@
|
|||||||
|
PRECONISATIONS = {
|
||||||
|
'Facile': [
|
||||||
|
"Constituer des stocks stratégiques.",
|
||||||
|
"Surveiller activement les signaux géopolitiques.",
|
||||||
|
"Renforcer la surveillance des régions critiques."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Diversifier progressivement les fournisseurs.",
|
||||||
|
"Favoriser la modularité des produits.",
|
||||||
|
"Augmenter progressivement les taux de recyclage."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Investir fortement en R&D pour la substitution.",
|
||||||
|
"Développer des technologies alternatives robustes.",
|
||||||
|
"Établir des partenariats stratégiques locaux solides."
|
||||||
|
],
|
||||||
|
'Extraction': {
|
||||||
|
'Facile': [
|
||||||
|
"Constituer des stocks « in-country » (site minier / port) pour 30 jours.",
|
||||||
|
"Activer un moniteur de prix spot quotidien.",
|
||||||
|
"Lancer une veille ESG locale (manifestations, météo extrême)."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Négocier des contrats « take-or-pay » avec au moins 2 exploitants distincts.",
|
||||||
|
"Mettre en place un audit semestriel des pratiques de sécurité/logistique des mines.",
|
||||||
|
"Financer en co-investissement un entrepôt portuaire multi-produits."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Participer au capital d’un producteur émergent hors zone de concentration.",
|
||||||
|
"Obtenir des droits d’« off-take » de 5 ans sur 20 % de la production d’une mine alternative.",
|
||||||
|
"Soutenir (CAPEX) l’ouverture d’une nouvelle voie ferroviaire ou portuaire sécurisée."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
'Traitement': {
|
||||||
|
'Facile': [
|
||||||
|
"Sécuriser un stock tampon sur site (90 jours).",
|
||||||
|
"Faire certifier la traçabilité chimique du concentré."
|
||||||
|
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Valider un second affineur dans une région politiquement stable.",
|
||||||
|
"Imposer des clauses « force-majeure » limitant l’arrêt total à 48 h.",
|
||||||
|
"Explorer les possibilités de recyclage et d'économie circulaire"
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Co-développer un site de raffinage dans une zone « friend-shore ».",
|
||||||
|
"Financer un procédé de purification à rendement plus élevé (réduit la dépendance au minerai primaire).",
|
||||||
|
"Constituer des réserves stratégiques pour les périodes de tension"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
'Fabrication': {
|
||||||
|
'Facile': [
|
||||||
|
"Mettre un seuil minimal de sécurité (45 jours) sur le composant critique en usine SMT.",
|
||||||
|
"Suivre hebdomadairement la capacité libre des fondeurs/EMS.",
|
||||||
|
"Maintenir une veille technologique sur les évolutions du marché"
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Dual-sourcer le composant critique intégrant un minerai critique (au moins 30 % chez un second fondeur).",
|
||||||
|
"Déployer le « design-for-substitution » : même PCB compatible avec le composant concerné.",
|
||||||
|
"Optimiser les processus d'approvisionnement existants"
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Lancer un programme R&D de substitution ou d'alternative budgeté sur 3 ans.",
|
||||||
|
"Contractualiser un accord exclusif avec un fondeur hors zone rouge pour 25 % des volumes."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
'Assemblage': {
|
||||||
|
'Facile': [
|
||||||
|
"Allonger la rotation des stocks de produits finis (en aval) pour amortir un retard de 2 semaines.",
|
||||||
|
"Mettre en place un plan de re-déploiement du personnel sur d’autres lignes en cas de rupture composant."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Avoir un site d’assemblage secondaire (low-volume) dans une région verte, testé tous les 6 mois.",
|
||||||
|
"Segmenter les nomenclatures : version « premium » avec composant haut de gamme, version « fallback » avec composant moins critique."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Investir dans une plateforme d’assemblage flexible (robots modulaires) capable de basculer vers un composant de substitution en < 72 h.",
|
||||||
|
"Signer un accord gouvernemental pour un soutien logistique prioritaire (corridor aérien dédié) en cas de crise géopolitique.",
|
||||||
|
"Mettre en place des contrats à long terme avec des clauses de garantie d'approvisionnement"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
INDICATEURS = {
|
||||||
|
'Facile': [
|
||||||
|
"Suivi régulier de la stabilité géopolitique (ISG).",
|
||||||
|
"Durée réelle d'utilisation du matériel.",
|
||||||
|
"Niveau des stocks stratégiques disponibles."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Taux de diversification des fournisseurs par région.",
|
||||||
|
"Évolution trimestrielle de la concurrence intersectorielle (IVC).",
|
||||||
|
"Taux annuel de recyclage des composants critiques."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Budget annuel investi dans la recherche technologique.",
|
||||||
|
"Nombre de brevets déposés pour des substituts.",
|
||||||
|
"Progrès réel en matière de substitution technologique (ICS)."
|
||||||
|
],
|
||||||
|
'Extraction': {
|
||||||
|
'Facile': [
|
||||||
|
"Jours de stock portuaire (objectif ≥ 30).",
|
||||||
|
"Indice ISG moyen pondéré des pays extracteurs (alerte ≥ 60).",
|
||||||
|
"Volatilité hebdo du prix spot (écart-type %)."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Part du 2ᵉ fournisseur dans le volume total (objectif ≥ 20 %).",
|
||||||
|
"Délai moyen d’obtention des permis d’export."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Capacité annuelle d’une mine alternative financée (% du besoin interne).",
|
||||||
|
"Progrès physique de l’infrastructure logistique (Km de voie, % achevé)."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
'Traitement': {
|
||||||
|
'Facile': [
|
||||||
|
"Couverture stock tampon (jours).",
|
||||||
|
"Certificats de traçabilité obtenus (% lots)."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Nombre d’affineurs validés (objectif ≥ 2).",
|
||||||
|
"Taux de rendement global du procédé (%)."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Part de production refinée hors zone rouge (%).",
|
||||||
|
"Capex cumulé investi dans de nouveaux procédés (M€)."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
'Fabrication': {
|
||||||
|
'Facile': [
|
||||||
|
"Stock de composants critiques (jours).",
|
||||||
|
"Capacité libre des EMS (%) rapportée chaque vendredi."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Part du second fondeur dans la production du composant audio (%).",
|
||||||
|
"Nombre de PCB « design-for-substitution » validés."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Dépenses R&D substituts (€) vs budget.",
|
||||||
|
"TRI attendu sur les investisseurs fondeurs alternatifs."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
'Assemblage': {
|
||||||
|
'Facile': [
|
||||||
|
"Jours de produits finis en entrepôt.",
|
||||||
|
"Temps de retouche ligne en cas de rupture (heures)."
|
||||||
|
],
|
||||||
|
'Modérée': [
|
||||||
|
"Volume annuel produit sur le site de secours (%).",
|
||||||
|
"Temps de requalification d’une ligne vers la version « fallback »."
|
||||||
|
],
|
||||||
|
'Difficile': [
|
||||||
|
"Taux d’automatisation reconfigurable (% machines modulaires).",
|
||||||
|
"Nb d’heures du corridor aérien prioritaire utilisé vs capacité."
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
poids_operation = {
|
||||||
|
'Extraction': 1,
|
||||||
|
'Traitement': 1.5,
|
||||||
|
'Assemblage': 1.5,
|
||||||
|
'Fabrication': 2,
|
||||||
|
'Substitution': 2
|
||||||
|
}
|
||||||
192
app/plan_d_action/utils/data/data_processing.py
Normal file
192
app/plan_d_action/utils/data/data_processing.py
Normal file
@ -0,0 +1,192 @@
|
|||||||
|
import re
|
||||||
|
|
||||||
|
def parse_chains_md(filepath: str) -> tuple[dict, dict, dict, list, dict, dict]:
|
||||||
|
re_start_section = re.compile(r"^##\s*Chaînes\s+avec\s+risque\s+critique", re.IGNORECASE)
|
||||||
|
re_other_h2 = re.compile(r"^##\s+(?!(Chaînes\s+avec\s+risque\s+critique))")
|
||||||
|
re_chain_heading = re.compile(r"^###\s*(.+)\s*→\s*(.+)\s*→\s*(.+)$")
|
||||||
|
re_phase = re.compile(r"^\*\s*(Assemblage|Fabrication|Minerai|Extraction|Traitement)", re.IGNORECASE)
|
||||||
|
re_IHH = re.compile(r"IHH\s*[:]\s*([0-9]+(?:\.[0-9]+)?)")
|
||||||
|
re_ISG = re.compile(r"ISG\s*combiné\s*[:]\s*([0-9]+(?:\.[0-9]+)?)|ISG\s*[:]\s*([0-9]+(?:\.[0-9]+)?)", re.IGNORECASE)
|
||||||
|
re_ICS = re.compile(r"ICS\s*moyen\s*[:]\s*([0-9]+(?:\.[0-9]+)?)", re.IGNORECASE)
|
||||||
|
re_IVC = re.compile(r"IVC\s*[:]\s*([0-9]+(?:\.[0-9]+)?)", re.IGNORECASE)
|
||||||
|
|
||||||
|
produits, composants, mineraux, chains = {}, {}, {}, []
|
||||||
|
descriptions = {}
|
||||||
|
details_sections = {}
|
||||||
|
current_chain = None
|
||||||
|
current_phase = None
|
||||||
|
current_section = None
|
||||||
|
in_section = False
|
||||||
|
|
||||||
|
with open(filepath, encoding="utf-8") as f:
|
||||||
|
for raw_line in f:
|
||||||
|
line = raw_line.strip()
|
||||||
|
if not in_section:
|
||||||
|
if re_start_section.match(line):
|
||||||
|
in_section = True
|
||||||
|
continue
|
||||||
|
if re_other_h2.match(line):
|
||||||
|
break
|
||||||
|
m_chain = re_chain_heading.match(line)
|
||||||
|
if m_chain:
|
||||||
|
prod, comp, miner = map(str.strip, m_chain.groups())
|
||||||
|
produits.setdefault(prod, {"IHH_Assemblage": None, "ISG_Assemblage": None})
|
||||||
|
composants.setdefault(comp, {"IHH_Fabrication": None, "ISG_Fabrication": None})
|
||||||
|
mineraux.setdefault(miner, {
|
||||||
|
"ICS": None, "IVC": None,
|
||||||
|
"IHH_Extraction": None, "ISG_Extraction": None,
|
||||||
|
"IHH_Traitement": None, "ISG_Traitement": None
|
||||||
|
})
|
||||||
|
chains.append({"produit": prod, "composant": comp, "minerai": miner})
|
||||||
|
current_chain = {"prod": prod, "comp": comp, "miner": miner}
|
||||||
|
current_phase = None
|
||||||
|
current_section = f"{prod} → {comp} → {miner}"
|
||||||
|
descriptions[current_section] = ""
|
||||||
|
continue
|
||||||
|
if current_chain is None:
|
||||||
|
continue
|
||||||
|
m_phase = re_phase.match(line)
|
||||||
|
if m_phase:
|
||||||
|
current_phase = m_phase.group(1).capitalize()
|
||||||
|
continue
|
||||||
|
if current_phase:
|
||||||
|
p = current_chain
|
||||||
|
if current_phase == "Assemblage":
|
||||||
|
if (m := re_IHH.search(line)):
|
||||||
|
produits[p["prod"]]["IHH_Assemblage"] = float(m.group(1))
|
||||||
|
continue
|
||||||
|
if (m := re_ISG.search(line)):
|
||||||
|
raw = m.group(1) or m.group(2)
|
||||||
|
produits[p["prod"]]["ISG_Assemblage"] = float(raw)
|
||||||
|
continue
|
||||||
|
if current_phase == "Fabrication":
|
||||||
|
if (m := re_IHH.search(line)):
|
||||||
|
composants[p["comp"]]["IHH_Fabrication"] = float(m.group(1))
|
||||||
|
continue
|
||||||
|
if (m := re_ISG.search(line)):
|
||||||
|
raw = m.group(1) or m.group(2)
|
||||||
|
composants[p["comp"]]["ISG_Fabrication"] = float(raw)
|
||||||
|
continue
|
||||||
|
if current_phase == "Minerai":
|
||||||
|
if (m := re_ICS.search(line)):
|
||||||
|
mineraux[p["miner"]]["ICS"] = float(m.group(1))
|
||||||
|
continue
|
||||||
|
if (m := re_IVC.search(line)):
|
||||||
|
mineraux[p["miner"]]["IVC"] = float(m.group(1))
|
||||||
|
continue
|
||||||
|
if current_phase == "Extraction":
|
||||||
|
if (m := re_IHH.search(line)):
|
||||||
|
mineraux[p["miner"]]["IHH_Extraction"] = float(m.group(1))
|
||||||
|
continue
|
||||||
|
if (m := re_ISG.search(line)):
|
||||||
|
raw = m.group(1) or m.group(2)
|
||||||
|
mineraux[p["miner"]]["ISG_Extraction"] = float(raw)
|
||||||
|
continue
|
||||||
|
if current_phase == "Traitement":
|
||||||
|
if (m := re_IHH.search(line)):
|
||||||
|
mineraux[p["miner"]]["IHH_Traitement"] = float(m.group(1))
|
||||||
|
continue
|
||||||
|
if (m := re_ISG.search(line)):
|
||||||
|
raw = m.group(1) or m.group(2)
|
||||||
|
mineraux[p["miner"]]["ISG_Traitement"] = float(raw)
|
||||||
|
continue
|
||||||
|
else:
|
||||||
|
if current_section:
|
||||||
|
descriptions[current_section] += raw_line
|
||||||
|
|
||||||
|
# Parse detailed sections from the complete file
|
||||||
|
with open(filepath, encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Extract sections using regex patterns
|
||||||
|
lines = content.split('\n')
|
||||||
|
|
||||||
|
# Find section boundaries
|
||||||
|
operations_start = None
|
||||||
|
minerais_start = None
|
||||||
|
|
||||||
|
for i, line in enumerate(lines):
|
||||||
|
if line.strip() == "## Détails des opérations":
|
||||||
|
operations_start = i
|
||||||
|
elif line.strip() == "## Détails des minerais":
|
||||||
|
minerais_start = i
|
||||||
|
|
||||||
|
if operations_start is not None:
|
||||||
|
# Parse operations section (assemblage and fabrication)
|
||||||
|
operations_end = minerais_start if minerais_start else len(lines)
|
||||||
|
operations_lines = lines[operations_start:operations_end]
|
||||||
|
|
||||||
|
current_section_name = None
|
||||||
|
current_content = []
|
||||||
|
|
||||||
|
for line in operations_lines:
|
||||||
|
if line.startswith("### ") and " et " in line:
|
||||||
|
# Save previous section
|
||||||
|
if current_section_name and current_content:
|
||||||
|
details_sections[current_section_name] = '\n'.join(current_content)
|
||||||
|
|
||||||
|
# Start new section
|
||||||
|
section_title = line.replace("### ", "").strip()
|
||||||
|
if " et Assemblage" in section_title:
|
||||||
|
product_name = section_title.replace(" et Assemblage", "").strip()
|
||||||
|
current_section_name = f"{product_name}_assemblage"
|
||||||
|
elif " et Fabrication" in section_title:
|
||||||
|
component_name = section_title.replace(" et Fabrication", "").strip()
|
||||||
|
current_section_name = f"{component_name}_fabrication"
|
||||||
|
current_content = []
|
||||||
|
elif current_section_name:
|
||||||
|
current_content.append(line)
|
||||||
|
|
||||||
|
# Save last section
|
||||||
|
if current_section_name and current_content:
|
||||||
|
details_sections[current_section_name] = '\n'.join(current_content)
|
||||||
|
|
||||||
|
if minerais_start is not None:
|
||||||
|
# Parse minerais section
|
||||||
|
minerais_lines = lines[minerais_start:]
|
||||||
|
|
||||||
|
current_minerai = None
|
||||||
|
current_section_type = "general"
|
||||||
|
current_content = []
|
||||||
|
|
||||||
|
for line in minerais_lines:
|
||||||
|
if line.startswith("### ") and "→" not in line and " et " not in line:
|
||||||
|
# Save previous section
|
||||||
|
if current_minerai and current_content:
|
||||||
|
details_sections[f"{current_minerai}_{current_section_type}"] = '\n'.join(current_content)
|
||||||
|
|
||||||
|
# Start new minerai
|
||||||
|
current_minerai = line.replace("### ", "").strip()
|
||||||
|
current_section_type = "general"
|
||||||
|
current_content = []
|
||||||
|
|
||||||
|
elif line.startswith("#### Extraction"):
|
||||||
|
# Save previous section
|
||||||
|
if current_minerai and current_content:
|
||||||
|
details_sections[f"{current_minerai}_{current_section_type}"] = '\n'.join(current_content)
|
||||||
|
|
||||||
|
current_section_type = "extraction"
|
||||||
|
current_content = []
|
||||||
|
|
||||||
|
elif line.startswith("#### Traitement"):
|
||||||
|
# Save previous section
|
||||||
|
if current_minerai and current_content:
|
||||||
|
details_sections[f"{current_minerai}_{current_section_type}"] = '\n'.join(current_content)
|
||||||
|
|
||||||
|
current_section_type = "traitement"
|
||||||
|
current_content = []
|
||||||
|
|
||||||
|
elif line.startswith("## ") and current_minerai:
|
||||||
|
# End of minerais section
|
||||||
|
if current_content:
|
||||||
|
details_sections[f"{current_minerai}_{current_section_type}"] = '\n'.join(current_content)
|
||||||
|
break
|
||||||
|
|
||||||
|
elif current_minerai:
|
||||||
|
current_content.append(line)
|
||||||
|
|
||||||
|
# Save last section
|
||||||
|
if current_minerai and current_content:
|
||||||
|
details_sections[f"{current_minerai}_{current_section_type}"] = '\n'.join(current_content)
|
||||||
|
|
||||||
|
return produits, composants, mineraux, chains, descriptions, details_sections
|
||||||
67
app/plan_d_action/utils/data/data_utils.py
Normal file
67
app/plan_d_action/utils/data/data_utils.py
Normal file
@ -0,0 +1,67 @@
|
|||||||
|
import yaml
|
||||||
|
import streamlit as st
|
||||||
|
|
||||||
|
def get_seuil(seuils_dict, key):
|
||||||
|
try:
|
||||||
|
if key in seuils_dict:
|
||||||
|
data = seuils_dict[key]
|
||||||
|
for niveau in ["rouge", "orange", "vert"]:
|
||||||
|
if niveau in data:
|
||||||
|
seuil = data[niveau]
|
||||||
|
if "min" in seuil and seuil["min"] is not None:
|
||||||
|
return seuil["min"]
|
||||||
|
if "max" in seuil and seuil["max"] is not None:
|
||||||
|
return seuil["max"]
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
return None
|
||||||
|
|
||||||
|
def set_vulnerability(v1, v2, t1, t2, seuils):
|
||||||
|
v1_poids = 1
|
||||||
|
v1_couleur = "Vert"
|
||||||
|
if v1 > seuils[t1]["rouge"]["min"]:
|
||||||
|
v1_poids = 3
|
||||||
|
v1_couleur = "Rouge"
|
||||||
|
elif v1 > seuils[t1]["vert"]["max"]:
|
||||||
|
v1_poids = 2
|
||||||
|
v1_couleur = "Orange"
|
||||||
|
|
||||||
|
v2_poids = 1
|
||||||
|
v2_couleur = "Vert"
|
||||||
|
if v2 > seuils[t2]["rouge"]["min"]:
|
||||||
|
v2_poids = 3
|
||||||
|
v2_couleur = "Rouge"
|
||||||
|
elif v2 > seuils[t2]["vert"]["max"]:
|
||||||
|
v2_poids = 2
|
||||||
|
v2_couleur = "Orange"
|
||||||
|
|
||||||
|
poids = v1_poids * v2_poids
|
||||||
|
couleur = "Rouge"
|
||||||
|
if poids <= 2:
|
||||||
|
couleur = "Vert"
|
||||||
|
elif poids <= 4:
|
||||||
|
couleur = "Orange"
|
||||||
|
|
||||||
|
return poids, couleur, v1_couleur, v2_couleur
|
||||||
|
|
||||||
|
def colorer_couleurs(la_couleur):
|
||||||
|
t = la_couleur.lower()
|
||||||
|
if t == "rouge" or t == "difficile":
|
||||||
|
return f":red-badge[{la_couleur}]"
|
||||||
|
if t == "orange" or t == "modérée":
|
||||||
|
return f":orange-badge[{la_couleur}]"
|
||||||
|
if t == "vert" or t == "facile":
|
||||||
|
return f":green-badge[{la_couleur}]"
|
||||||
|
return la_couleur
|
||||||
|
|
||||||
|
def initialiser_seuils(config_path):
|
||||||
|
seuils = {}
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(config_path, "r", encoding="utf-8") as f:
|
||||||
|
config = yaml.safe_load(f)
|
||||||
|
seuils = config.get("seuils", seuils)
|
||||||
|
except FileNotFoundError:
|
||||||
|
st.warning(f"Fichier de configuration {config_path} non trouvé.")
|
||||||
|
|
||||||
|
return seuils
|
||||||
167
app/plan_d_action/utils/data/pda_interface.py
Normal file
167
app/plan_d_action/utils/data/pda_interface.py
Normal file
@ -0,0 +1,167 @@
|
|||||||
|
import streamlit as st
|
||||||
|
|
||||||
|
def afficher_bloc_ihh_isg(titre, ihh, isg, details_content=""):
|
||||||
|
st.markdown(f"### {titre}")
|
||||||
|
|
||||||
|
if not details_content:
|
||||||
|
st.markdown("Données non disponibles")
|
||||||
|
return
|
||||||
|
|
||||||
|
lines = details_content.split('\n')
|
||||||
|
|
||||||
|
# 1. Afficher vulnérabilité combinée en premier
|
||||||
|
if "#### Vulnérabilité combinée IHH-ISG" in details_content:
|
||||||
|
conteneur, = st.columns([1], gap="small", border=True)
|
||||||
|
with conteneur:
|
||||||
|
st.markdown("#### Vulnérabilité combinée IHH-ISG")
|
||||||
|
afficher_section_texte(lines, "#### Vulnérabilité combinée IHH-ISG", "###")
|
||||||
|
|
||||||
|
# 2. Afficher ISG des pays impliqués
|
||||||
|
if "##### ISG des pays impliqués" in details_content:
|
||||||
|
print(details_content)
|
||||||
|
st.markdown("#### ISG des pays impliqués")
|
||||||
|
afficher_section_avec_tableau(lines, "##### ISG des pays impliqués")
|
||||||
|
|
||||||
|
# Afficher le résumé ISG combiné
|
||||||
|
for line in lines:
|
||||||
|
if "**ISG combiné:" in line:
|
||||||
|
st.markdown(line)
|
||||||
|
break
|
||||||
|
|
||||||
|
# 3. Afficher la section IHH complète
|
||||||
|
if "#### Indice de Herfindahl-Hirschmann" in details_content:
|
||||||
|
st.markdown("#### Indice de Herfindahl-Hirschmann")
|
||||||
|
|
||||||
|
# Tableau de résumé IHH
|
||||||
|
afficher_section_avec_tableau(lines, "#### Indice de Herfindahl-Hirschmann")
|
||||||
|
|
||||||
|
# IHH par entreprise
|
||||||
|
if "##### IHH par entreprise (acteurs)" in details_content:
|
||||||
|
st.markdown("##### IHH par entreprise (acteurs)")
|
||||||
|
afficher_section_texte(lines, "##### IHH par entreprise (acteurs)", "##### IHH par pays")
|
||||||
|
|
||||||
|
# IHH par pays
|
||||||
|
if "##### IHH par pays" in details_content:
|
||||||
|
st.markdown("##### IHH par pays")
|
||||||
|
afficher_section_texte(lines, "##### IHH par pays", "##### En résumé")
|
||||||
|
|
||||||
|
# En résumé
|
||||||
|
if "##### En résumé" in details_content:
|
||||||
|
st.markdown("##### En résumé")
|
||||||
|
afficher_section_texte(lines, "##### En résumé", "####")
|
||||||
|
|
||||||
|
def afficher_section_avec_tableau(lines, section_start, section_end=None):
|
||||||
|
"""Affiche une section contenant un tableau"""
|
||||||
|
in_section = False
|
||||||
|
table_lines = []
|
||||||
|
|
||||||
|
for line in lines:
|
||||||
|
if section_start in line:
|
||||||
|
in_section = True
|
||||||
|
continue
|
||||||
|
elif in_section and section_end and section_end in line:
|
||||||
|
break
|
||||||
|
elif in_section and line.startswith('#') and section_start not in line:
|
||||||
|
break
|
||||||
|
elif in_section:
|
||||||
|
if line.strip().startswith('|'):
|
||||||
|
table_lines.append(line)
|
||||||
|
elif table_lines and not line.strip().startswith('|'):
|
||||||
|
# Fin du tableau
|
||||||
|
break
|
||||||
|
|
||||||
|
if table_lines:
|
||||||
|
st.markdown('\n'.join(table_lines))
|
||||||
|
|
||||||
|
def afficher_section_texte(lines, section_start, section_end_marker=None):
|
||||||
|
"""Affiche le texte d'une section sans les tableaux"""
|
||||||
|
in_section = False
|
||||||
|
|
||||||
|
for line in lines:
|
||||||
|
if section_start in line:
|
||||||
|
in_section = True
|
||||||
|
continue
|
||||||
|
elif in_section and section_end_marker and line.startswith(section_end_marker):
|
||||||
|
break
|
||||||
|
elif in_section and line.startswith('#') and section_start not in line:
|
||||||
|
break
|
||||||
|
elif in_section and line.strip() and not line.strip().startswith('|'):
|
||||||
|
st.markdown(line)
|
||||||
|
|
||||||
|
def afficher_description(titre, description):
|
||||||
|
st.markdown(f"## {titre}")
|
||||||
|
conteneur, = st.columns([1], gap="small", border=True)
|
||||||
|
with conteneur:
|
||||||
|
if description:
|
||||||
|
lines = description.split('\n')
|
||||||
|
description_lines = []
|
||||||
|
|
||||||
|
# Extraire le premier paragraphe descriptif
|
||||||
|
for line in lines:
|
||||||
|
line = line.strip()
|
||||||
|
if not line:
|
||||||
|
if description_lines: # Si on a déjà du contenu, une ligne vide termine le paragraphe
|
||||||
|
break
|
||||||
|
continue
|
||||||
|
# Arrêter aux titres de sections ou tableaux
|
||||||
|
if (line.startswith('####') or
|
||||||
|
line.startswith('|') or
|
||||||
|
line.startswith('**Unité')):
|
||||||
|
break
|
||||||
|
description_lines.append(line)
|
||||||
|
|
||||||
|
if description_lines:
|
||||||
|
# Rejoindre les lignes en un seul paragraphe
|
||||||
|
full_description = ' '.join(description_lines)
|
||||||
|
st.markdown(full_description)
|
||||||
|
else:
|
||||||
|
st.markdown("Description non disponible")
|
||||||
|
else:
|
||||||
|
st.markdown("Description non disponible")
|
||||||
|
|
||||||
|
def afficher_caracteristiques_minerai(minerai, mineraux_data, details_content=""):
|
||||||
|
st.markdown("### Caractéristiques générales")
|
||||||
|
|
||||||
|
if not details_content:
|
||||||
|
st.markdown("Données non disponibles")
|
||||||
|
return
|
||||||
|
|
||||||
|
lines = details_content.split('\n')
|
||||||
|
|
||||||
|
# 3. Afficher la vulnérabilité combinée ICS-IVC en dernier
|
||||||
|
if "#### Vulnérabilité combinée ICS-IVC" in details_content:
|
||||||
|
conteneur, = st.columns([1], gap="small", border=True)
|
||||||
|
with conteneur:
|
||||||
|
st.markdown("#### Vulnérabilité combinée ICS-IVC")
|
||||||
|
afficher_section_texte(lines, "#### Vulnérabilité combinée ICS-IVC", "####")
|
||||||
|
|
||||||
|
# 1. Afficher la section ICS complète
|
||||||
|
if "#### ICS" in details_content:
|
||||||
|
st.markdown("#### ICS")
|
||||||
|
|
||||||
|
# Afficher le premier tableau ICS (avec toutes les colonnes)
|
||||||
|
afficher_section_avec_tableau(lines, "#### ICS", "##### Valeurs d'ICS par composant")
|
||||||
|
|
||||||
|
# Afficher la sous-section "Valeurs d'ICS par composant"
|
||||||
|
if "##### Valeurs d'ICS par composant" in details_content:
|
||||||
|
st.markdown("##### Valeurs d'ICS par composant")
|
||||||
|
afficher_section_avec_tableau(lines, "##### Valeurs d'ICS par composant", "**ICS moyen")
|
||||||
|
|
||||||
|
# Afficher le résumé ICS moyen
|
||||||
|
for line in lines:
|
||||||
|
if "**ICS moyen" in line:
|
||||||
|
st.markdown(line)
|
||||||
|
break
|
||||||
|
|
||||||
|
# 2. Afficher la section IVC complète
|
||||||
|
if "#### IVC" in details_content:
|
||||||
|
st.markdown("#### IVC")
|
||||||
|
|
||||||
|
# Afficher la valeur IVC principale
|
||||||
|
for line in lines:
|
||||||
|
if "**IVC:" in line:
|
||||||
|
st.markdown(line)
|
||||||
|
break
|
||||||
|
|
||||||
|
# Afficher tous les détails de la section IVC
|
||||||
|
afficher_section_texte(lines, "#### IVC", "#### Vulnérabilité combinée ICS-IVC")
|
||||||
350
app/plan_d_action/utils/data/plan_d_action.py
Normal file
350
app/plan_d_action/utils/data/plan_d_action.py
Normal file
@ -0,0 +1,350 @@
|
|||||||
|
import streamlit as st
|
||||||
|
import matplotlib.pyplot as plt
|
||||||
|
|
||||||
|
from app.plan_d_action.utils.data.config import (
|
||||||
|
PRECONISATIONS,
|
||||||
|
INDICATEURS,
|
||||||
|
poids_operation
|
||||||
|
)
|
||||||
|
from app.plan_d_action.utils.data.data_processing import parse_chains_md
|
||||||
|
from app.plan_d_action.utils.data.data_utils import (
|
||||||
|
set_vulnerability,
|
||||||
|
colorer_couleurs
|
||||||
|
)
|
||||||
|
from app.plan_d_action.utils.data.pda_interface import (
|
||||||
|
afficher_bloc_ihh_isg,
|
||||||
|
afficher_description,
|
||||||
|
afficher_caracteristiques_minerai
|
||||||
|
)
|
||||||
|
from app.plan_d_action.utils.data.data_utils import initialiser_seuils
|
||||||
|
|
||||||
|
def calcul_poids_chaine(poids_A, poids_F, poids_T, poids_E, poids_M):
|
||||||
|
poids_total = (\
|
||||||
|
poids_A * poids_operation["Assemblage"] + \
|
||||||
|
poids_F * poids_operation["Fabrication"] + \
|
||||||
|
poids_T * poids_operation["Traitement"] + \
|
||||||
|
poids_E * poids_operation["Extraction"] + \
|
||||||
|
poids_M * poids_operation["Substitution"] \
|
||||||
|
) / sum(poids_operation.values())
|
||||||
|
|
||||||
|
if poids_total < 3:
|
||||||
|
criticite_chaine = "Modérée"
|
||||||
|
niveau_criticite = {"Facile"}
|
||||||
|
elif poids_total < 6:
|
||||||
|
criticite_chaine = "Élevée"
|
||||||
|
niveau_criticite = {"Facile", "Modérée"}
|
||||||
|
else:
|
||||||
|
criticite_chaine = "Critique"
|
||||||
|
niveau_criticite = {"Facile", "Modérée", "Difficile"}
|
||||||
|
|
||||||
|
return criticite_chaine, niveau_criticite, poids_total
|
||||||
|
|
||||||
|
def analyser_chaines(chaines, produits, composants, mineraux, seuils, top_n=None):
|
||||||
|
resultats = []
|
||||||
|
|
||||||
|
for chaine in chaines:
|
||||||
|
sel_prod = chaine["produit"]
|
||||||
|
sel_comp = chaine["composant"]
|
||||||
|
sel_miner = chaine["minerai"]
|
||||||
|
|
||||||
|
poids_A, *_ = set_vulnerability(produits[sel_prod]["IHH_Assemblage"], produits[sel_prod]["ISG_Assemblage"], "IHH", "ISG", seuils)
|
||||||
|
poids_F, *_ = set_vulnerability(composants[sel_comp]["IHH_Fabrication"], composants[sel_comp]["ISG_Fabrication"], "IHH", "ISG", seuils)
|
||||||
|
poids_T, *_ = set_vulnerability(mineraux[sel_miner]["IHH_Traitement"], mineraux[sel_miner]["ISG_Traitement"], "IHH", "ISG", seuils)
|
||||||
|
poids_E, *_ = set_vulnerability(mineraux[sel_miner]["IHH_Extraction"], mineraux[sel_miner]["ISG_Extraction"], "IHH", "ISG", seuils)
|
||||||
|
poids_M, *_ = set_vulnerability(mineraux[sel_miner]["ICS"], mineraux[sel_miner]["IVC"], "ICS", "IVC", seuils)
|
||||||
|
|
||||||
|
criticite_chaine, niveau_criticite, poids_total = calcul_poids_chaine(
|
||||||
|
poids_A, poids_F, poids_T, poids_E, poids_M
|
||||||
|
)
|
||||||
|
|
||||||
|
resultats.append({
|
||||||
|
"chaine": chaine,
|
||||||
|
"criticite_chaine": criticite_chaine,
|
||||||
|
"niveau_criticite": niveau_criticite,
|
||||||
|
"poids_total": poids_total
|
||||||
|
})
|
||||||
|
|
||||||
|
# Tri décroissant
|
||||||
|
resultats.sort(key=lambda x: x["poids_total"], reverse=True)
|
||||||
|
|
||||||
|
# Si top_n n'est pas spécifié, tout est retourné
|
||||||
|
if top_n is None or top_n >= len(resultats):
|
||||||
|
return resultats
|
||||||
|
|
||||||
|
# Déterminer le seuil de coupure
|
||||||
|
seuil_poids = resultats[top_n - 1]["poids_total"]
|
||||||
|
|
||||||
|
# Inclure tous ceux dont le poids est égal au seuil
|
||||||
|
top_resultats = [r for r in resultats if r["poids_total"] >= seuil_poids]
|
||||||
|
|
||||||
|
return top_resultats
|
||||||
|
|
||||||
|
def tableau_de_bord(chains, produits, composants, mineraux, seuils):
|
||||||
|
col_left, col_right = st.columns([2, 3], gap="small", border=True)
|
||||||
|
with col_left:
|
||||||
|
st.markdown("**<u>Panneau de sélection</u>**", unsafe_allow_html=True)
|
||||||
|
|
||||||
|
produits_disponibles = sorted({c["produit"] for c in chains})
|
||||||
|
sel_prod = st.selectbox("Produit", produits_disponibles, index=produits_disponibles.index(st.session_state.sel_prod) if st.session_state.sel_prod else 0)
|
||||||
|
|
||||||
|
composants_dispo = sorted({c["composant"] for c in chains if c["produit"] == sel_prod})
|
||||||
|
sel_comp = st.selectbox("Composant", composants_dispo, index=composants_dispo.index(st.session_state.sel_comp) if st.session_state.sel_comp else 0)
|
||||||
|
|
||||||
|
mineraux_dispo = sorted({c["minerai"] for c in chains if c["produit"] == sel_prod and c["composant"] == sel_comp})
|
||||||
|
sel_miner = st.selectbox("Minerai", mineraux_dispo, index=mineraux_dispo.index(st.session_state.sel_miner) if st.session_state.sel_miner else 0)
|
||||||
|
with col_right:
|
||||||
|
top_chains = analyser_chaines(chains, produits, composants, mineraux, seuils, top_n=5)
|
||||||
|
st.markdown("**<u>Top chaînes critiques pour sélection rapide</u>**", unsafe_allow_html=True)
|
||||||
|
for i, entry in enumerate(top_chains):
|
||||||
|
ch = entry["chaine"]
|
||||||
|
poids = entry["poids_total"]
|
||||||
|
criticite = entry["criticite_chaine"]
|
||||||
|
if st.button(f"**{ch['produit']} <-> {ch['composant']} <-> {ch['minerai']}** : {poids:.2f} → {criticite}", key=f"select_{i}"):
|
||||||
|
st.session_state.sel_prod = ch["produit"]
|
||||||
|
st.session_state.sel_comp = ch["composant"]
|
||||||
|
st.session_state.sel_miner = ch["minerai"]
|
||||||
|
st.rerun()
|
||||||
|
|
||||||
|
c1, c2 = st.columns([3, 2], gap="small", border=True, vertical_alignment='center')
|
||||||
|
with c1:
|
||||||
|
st.markdown("**<u>Synthèse des criticités</u>**", unsafe_allow_html=True)
|
||||||
|
poids_A, couleur_A, couleur_A_ihh, couleur_A_isg = set_vulnerability(produits[sel_prod]["IHH_Assemblage"], produits[sel_prod]["ISG_Assemblage"], "IHH", "ISG", seuils)
|
||||||
|
poids_F, couleur_F, couleur_F_ihh, couleur_F_isg = set_vulnerability(composants[sel_comp]["IHH_Fabrication"], composants[sel_comp]["ISG_Fabrication"], "IHH", "ISG", seuils)
|
||||||
|
poids_T, couleur_T, couleur_T_ihh, couleur_T_isg = set_vulnerability(mineraux[sel_miner]["IHH_Traitement"], mineraux[sel_miner]["ISG_Traitement"], "IHH", "ISG", seuils)
|
||||||
|
poids_E, couleur_E, couleur_E_ihh, couleur_E_isg = set_vulnerability(mineraux[sel_miner]["IHH_Extraction"], mineraux[sel_miner]["ISG_Extraction"], "IHH", "ISG", seuils)
|
||||||
|
poids_M, couleur_M, couleur_M_ics, couleur_M_ivc = set_vulnerability(mineraux[sel_miner]["ICS"], mineraux[sel_miner]["IVC"], "ICS", "IVC", seuils)
|
||||||
|
|
||||||
|
st.markdown(f"* **{sel_prod} - Assemblage** : {colorer_couleurs(couleur_A)} ({poids_A})")
|
||||||
|
st.markdown(f"* **{sel_comp} - Fabrication** : {colorer_couleurs(couleur_F)} ({poids_F})")
|
||||||
|
st.markdown(f"* **{sel_miner} - Traitement** : {colorer_couleurs(couleur_T)} ({poids_T})")
|
||||||
|
st.markdown(f"* **{sel_miner} - Extraction** : {colorer_couleurs(couleur_E)} ({poids_E})")
|
||||||
|
st.markdown(f"* **{sel_miner} - Minerai** : {colorer_couleurs(couleur_M)} ({poids_M})")
|
||||||
|
|
||||||
|
criticite_chaine, niveau_criticite, poids_total = calcul_poids_chaine(poids_A, poids_F, poids_T, poids_E, poids_M)
|
||||||
|
with c2:
|
||||||
|
st.error(f"**Criticité globale : {criticite_chaine} ({poids_total})**")
|
||||||
|
|
||||||
|
return (
|
||||||
|
sel_prod, sel_comp, sel_miner, niveau_criticite,
|
||||||
|
couleur_A, poids_A, couleur_F, poids_F, couleur_T, poids_T, couleur_E, poids_E, couleur_M, poids_M,
|
||||||
|
couleur_A_ihh, couleur_A_isg, couleur_F_ihh, couleur_F_isg, couleur_T_ihh, couleur_T_isg,couleur_E_ihh, couleur_E_isg, couleur_M_ics, couleur_M_ivc
|
||||||
|
)
|
||||||
|
|
||||||
|
def afficher_criticites(produits, composants, mineraux, sel_prod, sel_comp, sel_miner, seuils):
|
||||||
|
with st.expander("Vue d’ensemble des criticités", expanded=True):
|
||||||
|
st.markdown("## Vue d’ensemble des criticités", unsafe_allow_html=True)
|
||||||
|
|
||||||
|
col_left, col_right = st.columns([1, 1], gap="small", border=True)
|
||||||
|
|
||||||
|
with col_left:
|
||||||
|
fig1, ax1 = plt.subplots(figsize=(2, 2.4))
|
||||||
|
ax1.scatter([produits[sel_prod]["ISG_Assemblage"]], [produits[sel_prod]["IHH_Assemblage"]], label="Assemblage", s=5)
|
||||||
|
ax1.scatter([composants[sel_comp]["ISG_Fabrication"]], [composants[sel_comp]["IHH_Fabrication"]], label="Fabrication", s=5)
|
||||||
|
ax1.scatter([mineraux[sel_miner]["ISG_Extraction"]], [mineraux[sel_miner]["IHH_Extraction"]], label="Extraction", s=5)
|
||||||
|
ax1.scatter([mineraux[sel_miner]["ISG_Traitement"]], [mineraux[sel_miner]["IHH_Traitement"]], label="Traitement", s=5)
|
||||||
|
|
||||||
|
# Seuils ISG (vertical)
|
||||||
|
ax1.axvline(seuils["ISG"]["vert"]["max"], linestyle='--', color='green', alpha=0.7, linewidth=0.5) # Seuil vert-orange
|
||||||
|
ax1.axvline(seuils["ISG"]["rouge"]["min"], linestyle='--', color='red', alpha=0.7, linewidth=0.5) # Seuil orange-rouge
|
||||||
|
|
||||||
|
# Seuils IHH (horizontal)
|
||||||
|
ax1.axhline(seuils["IHH"]["vert"]["max"], linestyle='--', color='green', alpha=0.7, linewidth=0.5) # Seuil vert-orange
|
||||||
|
ax1.axhline(seuils["IHH"]["rouge"]["min"], linestyle='--', color='red', alpha=0.7, linewidth=0.5) # Seuil orange-rouge
|
||||||
|
|
||||||
|
ax1.set_xlim(0, 100)
|
||||||
|
ax1.set_ylim(0, 100)
|
||||||
|
ax1.set_xlabel("ISG", fontsize=4)
|
||||||
|
ax1.set_ylabel("IHH", fontsize=4)
|
||||||
|
ax1.tick_params(axis='both', which='major', labelsize=4)
|
||||||
|
ax1.legend(bbox_to_anchor=(0.5, -0.25), loc='upper center', fontsize=4)
|
||||||
|
plt.tight_layout()
|
||||||
|
st.pyplot(fig1)
|
||||||
|
|
||||||
|
with col_right:
|
||||||
|
fig2, ax2 = plt.subplots(figsize=(2, 2.1))
|
||||||
|
ax2.scatter([mineraux[sel_miner]["IVC"]], [mineraux[sel_miner]["ICS"]], color='green', s=5, label=sel_miner)
|
||||||
|
|
||||||
|
# Seuils IVC (vertical)
|
||||||
|
ax2.axvline(seuils["IVC"]["vert"]["max"], linestyle='--', color='green', alpha=0.7, linewidth=0.5) # Seuil vert-orange
|
||||||
|
ax2.axvline(seuils["IVC"]["rouge"]["min"], linestyle='--', color='red', alpha=0.7, linewidth=0.5) # Seuil orange-rouge
|
||||||
|
|
||||||
|
# Seuils ICS (horizontal)
|
||||||
|
ax2.axhline(seuils["ICS"]["vert"]["max"], linestyle='--', color='green', alpha=0.7, linewidth=0.5) # Seuil vert-orange
|
||||||
|
ax2.axhline(seuils["ICS"]["rouge"]["min"], linestyle='--', color='red', alpha=0.7, linewidth=0.5) # Seuil orange-rouge
|
||||||
|
|
||||||
|
ax2.set_xlim(0, max(100, mineraux[sel_miner]["IVC"]))
|
||||||
|
ax2.set_ylim(0, 1)
|
||||||
|
ax2.set_xlabel("IVC", fontsize=4)
|
||||||
|
ax2.set_ylabel("ICS", fontsize=4)
|
||||||
|
ax2.tick_params(axis='both', which='major', labelsize=4)
|
||||||
|
ax2.legend(bbox_to_anchor=(0.5, -0.25), loc='upper center', fontsize=4)
|
||||||
|
plt.tight_layout()
|
||||||
|
st.pyplot(fig2)
|
||||||
|
|
||||||
|
st.markdown(f"""
|
||||||
|
Les lignes pointillées en {colorer_couleurs("vert")} ou {colorer_couleurs("rouge")} représentent les seuils des indices concernés.\n
|
||||||
|
Les indices ISG (stabilité géopolitique) et IVC (concurrence intersectorielle) influent sur la probabilité de survenance d'un risque.\n
|
||||||
|
Les indices IHH (concentration géographique) et ICS (capacité de substitution) influent sur le niveau d'impact d'un risque.\n
|
||||||
|
Une opération se trouvant au-dessus des deux seuils a donc une forte probabilité d'être impactée avec un niveau élevé sur l'incapacité à continuer la production.
|
||||||
|
""")
|
||||||
|
|
||||||
|
def afficher_explications_et_details(
|
||||||
|
couleur_A, poids_A, couleur_F, poids_F, couleur_T, poids_T, couleur_E, poids_E, couleur_M, poids_M,
|
||||||
|
produits, composants, mineraux, sel_prod, sel_comp, sel_miner,
|
||||||
|
couleur_A_ihh, couleur_A_isg, couleur_F_ihh, couleur_F_isg, couleur_T_ihh, couleur_T_isg,couleur_E_ihh, couleur_E_isg, couleur_M_ics, couleur_M_ivc):
|
||||||
|
with st.expander("Explications et détails", expanded = True):
|
||||||
|
from collections import Counter
|
||||||
|
couleurs = [couleur_A, couleur_F, couleur_T, couleur_E, couleur_M]
|
||||||
|
compte = Counter(couleurs)
|
||||||
|
nb_rouge = compte["Rouge"]
|
||||||
|
nb_orange = compte["Orange"]
|
||||||
|
nb_vert = compte["Vert"]
|
||||||
|
|
||||||
|
st.markdown(f"""
|
||||||
|
Pour cette chaîne :blue-background[**{sel_prod} <-> {sel_comp} <-> {sel_miner}**], avec {nb_rouge} criticité(s) de niveau {colorer_couleurs("Rouge")}, {nb_orange} {colorer_couleurs("Orange")} et {nb_vert} {colorer_couleurs("Vert")}, les indices individuels par opération sont :
|
||||||
|
|
||||||
|
* **{sel_prod} - Assemblage** : {colorer_couleurs(couleur_A)} ({poids_A})
|
||||||
|
* IHH = {produits[sel_prod]["IHH_Assemblage"]} ({colorer_couleurs(couleur_A_ihh)}) <-> ISG = {produits[sel_prod]["ISG_Assemblage"]} ({colorer_couleurs(couleur_A_isg)})
|
||||||
|
* pondération de l'Assemblage dans le calcul de la criticité globale : 1,5
|
||||||
|
* se référer à **{sel_prod} et Assemblage** plus bas pour le détail complet
|
||||||
|
* **{sel_comp} - Fabrication** : {colorer_couleurs(couleur_F)} ({poids_F})
|
||||||
|
* IHH = {composants[sel_comp]["IHH_Fabrication"]} ({colorer_couleurs(couleur_F_ihh)}) <-> ISG = {composants[sel_comp]["ISG_Fabrication"]} ({colorer_couleurs(couleur_F_isg)})
|
||||||
|
* pondération de la Fabrication dans le calcul de la criticité globale : 2
|
||||||
|
* se référer à **{sel_comp} et Fabrication** plus bas pour le détail complet
|
||||||
|
* **{sel_miner} - Traitement** : {colorer_couleurs(couleur_A)} ({poids_A})
|
||||||
|
* IHH = {mineraux[sel_miner]["IHH_Traitement"]} ({colorer_couleurs(couleur_T_ihh)}) <-> ISG = {mineraux[sel_miner]["ISG_Traitement"]} ({colorer_couleurs(couleur_T_isg)})
|
||||||
|
* pondération du Traitement dans le calcul de la criticité globale : 1,5
|
||||||
|
* se référer à **{sel_miner} — Vue globale** plus bas pour le détail complet de l'ensemble du minerai
|
||||||
|
* **{sel_miner} - Extraction** : {colorer_couleurs(couleur_E)} ({poids_E})
|
||||||
|
* IHH = {mineraux[sel_miner]["IHH_Extraction"]} ({colorer_couleurs(couleur_E_ihh)}) <-> ISG = {mineraux[sel_miner]["ISG_Extraction"]} ({colorer_couleurs(couleur_E_isg)})
|
||||||
|
* pondération de l'Extraction dans le calcul de la criticité globale : 1
|
||||||
|
* **{sel_miner} - Minerai** : {colorer_couleurs(couleur_M)} ({poids_M})
|
||||||
|
* ICS = {mineraux[sel_miner]["ICS"]} ({colorer_couleurs(couleur_M_ics)}) <-> IVC = {mineraux[sel_miner]["IVC"]} ({colorer_couleurs(couleur_M_ivc)})
|
||||||
|
* pondération de la Substitution dans le calcul de la criticité globale : 2
|
||||||
|
""")
|
||||||
|
|
||||||
|
def afficher_preconisations_et_indicateurs_generiques(niveau_criticite, poids_A, poids_F, poids_T, poids_E, poids_M):
|
||||||
|
with st.expander("Préconisations et indicateurs génériques"):
|
||||||
|
col_left, col_right = st.columns([1, 1], gap="small", border=True)
|
||||||
|
with col_left:
|
||||||
|
st.markdown("### Préconisations :\n\n")
|
||||||
|
st.markdown("Mise en œuvre : \n")
|
||||||
|
for niveau, contenu in PRECONISATIONS.items():
|
||||||
|
if niveau in niveau_criticite:
|
||||||
|
contenu_md = f"* {colorer_couleurs(niveau)}\n"
|
||||||
|
for p in PRECONISATIONS[niveau]:
|
||||||
|
contenu_md += f" - {p}\n"
|
||||||
|
st.markdown(contenu_md)
|
||||||
|
with col_right:
|
||||||
|
st.markdown("### Indicateurs :\n\n")
|
||||||
|
st.markdown("Mise en œuvre : \n")
|
||||||
|
for niveau, contenu in INDICATEURS.items():
|
||||||
|
if niveau in niveau_criticite:
|
||||||
|
contenu_md = f"* {colorer_couleurs(niveau)}\n"
|
||||||
|
for p in INDICATEURS[niveau]:
|
||||||
|
contenu_md += f" - {p}\n"
|
||||||
|
st.markdown(contenu_md)
|
||||||
|
|
||||||
|
def afficher_preconisations_et_indicateurs_specifiques(sel_prod, sel_comp, sel_miner, niveau_criticite_operation):
|
||||||
|
for operation in ["Assemblage", "Fabrication", "Traitement", "Extraction"]:
|
||||||
|
if operation == "Assemblage":
|
||||||
|
item = sel_prod
|
||||||
|
elif operation == "Fabrication":
|
||||||
|
item = sel_comp
|
||||||
|
else:
|
||||||
|
item = sel_miner
|
||||||
|
with st.expander(f"Préconisations et indicateurs spécifiques - {operation}"):
|
||||||
|
st.markdown(f"### {operation} -> :blue-background[{item}]")
|
||||||
|
col_left, col_right = st.columns([1, 1], gap="small", border=True)
|
||||||
|
with col_left:
|
||||||
|
st.markdown("#### Préconisations :\n\n")
|
||||||
|
st.markdown("Mise en œuvre : \n")
|
||||||
|
for niveau, contenu in PRECONISATIONS[operation].items():
|
||||||
|
if niveau in niveau_criticite_operation[operation]:
|
||||||
|
contenu_md = f"* {colorer_couleurs(niveau)}\n"
|
||||||
|
for p in PRECONISATIONS[operation][niveau]:
|
||||||
|
contenu_md += f" - {p}\n"
|
||||||
|
st.markdown(contenu_md)
|
||||||
|
with col_right:
|
||||||
|
st.markdown("#### Indicateurs :\n\n")
|
||||||
|
st.markdown("Mise en œuvre : \n")
|
||||||
|
for niveau, contenu in INDICATEURS[operation].items():
|
||||||
|
if niveau in niveau_criticite_operation[operation]:
|
||||||
|
contenu_md = f"* {colorer_couleurs(niveau)}\n"
|
||||||
|
for p in INDICATEURS[operation][niveau]:
|
||||||
|
contenu_md += f" - {p}\n"
|
||||||
|
st.markdown(contenu_md)
|
||||||
|
|
||||||
|
def afficher_preconisations_et_indicateurs(niveau_criticite, sel_prod, sel_comp, sel_miner, poids_A, poids_F, poids_T, poids_E, poids_M):
|
||||||
|
st.markdown("## Préconisations et indicateurs")
|
||||||
|
|
||||||
|
afficher_preconisations_et_indicateurs_generiques(niveau_criticite, poids_A, poids_F, poids_T, poids_E, poids_M)
|
||||||
|
|
||||||
|
def affectation_poids(poids_operation):
|
||||||
|
if poids_operation < 3:
|
||||||
|
niveau_criticite = {"Facile"}
|
||||||
|
elif poids_operation < 6:
|
||||||
|
niveau_criticite = {"Facile", "Modérée"}
|
||||||
|
else:
|
||||||
|
niveau_criticite = {"Facile", "Modérée", "Difficile"}
|
||||||
|
return niveau_criticite
|
||||||
|
|
||||||
|
niveau_criticite_operation = {}
|
||||||
|
niveau_criticite_operation["Assemblage"] = affectation_poids(poids_A)
|
||||||
|
niveau_criticite_operation["Fabrication"] = affectation_poids(poids_F)
|
||||||
|
niveau_criticite_operation["Traitement"] = affectation_poids(poids_T)
|
||||||
|
niveau_criticite_operation["Extraction"] = affectation_poids(poids_E)
|
||||||
|
|
||||||
|
afficher_preconisations_et_indicateurs_specifiques(sel_prod, sel_comp, sel_miner, niveau_criticite_operation)
|
||||||
|
|
||||||
|
def afficher_details_operations(produits, composants, mineraux, sel_prod, sel_comp, sel_miner, details_sections):
|
||||||
|
st.markdown("## Détails des opérations")
|
||||||
|
|
||||||
|
with st.expander(f"{sel_prod} et Assemblage"):
|
||||||
|
assemblage_details = details_sections.get(f"{sel_prod}_assemblage", "")
|
||||||
|
|
||||||
|
afficher_description(f"{sel_prod} et Assemblage", assemblage_details)
|
||||||
|
afficher_bloc_ihh_isg("Assemblage", produits[sel_prod]["IHH_Assemblage"], produits[sel_prod]["ISG_Assemblage"], assemblage_details)
|
||||||
|
|
||||||
|
with st.expander(f"{sel_comp} et Fabrication"):
|
||||||
|
fabrication_details = details_sections.get(f"{sel_comp}_fabrication", "")
|
||||||
|
afficher_description(f"{sel_comp} et Fabrication", fabrication_details)
|
||||||
|
afficher_bloc_ihh_isg("Fabrication", composants[sel_comp]["IHH_Fabrication"], composants[sel_comp]["ISG_Fabrication"], fabrication_details)
|
||||||
|
|
||||||
|
with st.expander(f"{sel_miner} — Vue globale"):
|
||||||
|
minerai_general = details_sections.get(f"{sel_miner}_general", "")
|
||||||
|
afficher_description(f"{sel_miner} — Vue globale", minerai_general)
|
||||||
|
|
||||||
|
extraction_details = details_sections.get(f"{sel_miner}_extraction", "")
|
||||||
|
afficher_bloc_ihh_isg("Extraction", mineraux[sel_miner]["IHH_Extraction"], mineraux[sel_miner]["ISG_Extraction"], extraction_details)
|
||||||
|
|
||||||
|
traitement_details = details_sections.get(f"{sel_miner}_traitement", "").removesuffix("\n---\n")
|
||||||
|
afficher_bloc_ihh_isg("Traitement", mineraux[sel_miner]["IHH_Traitement"], mineraux[sel_miner]["ISG_Traitement"], traitement_details)
|
||||||
|
|
||||||
|
afficher_caracteristiques_minerai(sel_miner, mineraux[sel_miner], minerai_general)
|
||||||
|
|
||||||
|
def initialiser_interface(filepath: str, config_path: str = "assets/config.yaml"):
|
||||||
|
|
||||||
|
produits, composants, mineraux, chains, descriptions, details_sections = parse_chains_md(filepath)
|
||||||
|
|
||||||
|
if not chains:
|
||||||
|
st.warning("Aucune chaîne critique trouvée dans le fichier.")
|
||||||
|
return
|
||||||
|
|
||||||
|
seuils = initialiser_seuils(config_path)
|
||||||
|
|
||||||
|
sel_prod, sel_comp, sel_miner, niveau_criticite, \
|
||||||
|
couleur_A, poids_A, couleur_F, poids_F, couleur_T, poids_T, couleur_E, poids_E, couleur_M, poids_M, \
|
||||||
|
couleur_A_ihh, couleur_A_isg, couleur_F_ihh, couleur_F_isg, couleur_T_ihh, couleur_T_isg,couleur_E_ihh, couleur_E_isg, couleur_M_ics, couleur_M_ivc \
|
||||||
|
= tableau_de_bord(chains, produits, composants, mineraux, seuils)
|
||||||
|
|
||||||
|
afficher_criticites(produits, composants, mineraux, sel_prod, sel_comp, sel_miner, seuils)
|
||||||
|
|
||||||
|
afficher_explications_et_details(
|
||||||
|
couleur_A, poids_A, couleur_F, poids_F, couleur_T, poids_T, couleur_E, poids_E, couleur_M, poids_M,
|
||||||
|
produits, composants, mineraux, sel_prod, sel_comp, sel_miner,
|
||||||
|
couleur_A_ihh, couleur_A_isg, couleur_F_ihh, couleur_F_isg, couleur_T_ihh, couleur_T_isg,couleur_E_ihh, couleur_E_isg, couleur_M_ics, couleur_M_ivc)
|
||||||
|
|
||||||
|
afficher_preconisations_et_indicateurs(niveau_criticite, sel_prod, sel_comp, sel_miner, poids_A, poids_F, poids_T, poids_E, poids_M)
|
||||||
|
|
||||||
|
afficher_details_operations(produits, composants, mineraux, sel_prod, sel_comp, sel_miner, details_sections)
|
||||||
17
app/plan_d_action/utils/interface/__init__.py
Normal file
17
app/plan_d_action/utils/interface/__init__.py
Normal file
@ -0,0 +1,17 @@
|
|||||||
|
from .parser import preparer_graphe
|
||||||
|
from .niveau_utils import extraire_niveaux
|
||||||
|
from .selection import (
|
||||||
|
selectionner_minerais,
|
||||||
|
selectionner_noeuds,
|
||||||
|
extraire_chemins_selon_criteres
|
||||||
|
)
|
||||||
|
from .export import (
|
||||||
|
exporter_graphe_filtre,
|
||||||
|
extraire_liens_filtres
|
||||||
|
)
|
||||||
|
from .visualization import remplacer_par_badge
|
||||||
|
from .config import (
|
||||||
|
niveau_labels,
|
||||||
|
JOBS,
|
||||||
|
CORRESPONDANCE_COULEURS
|
||||||
|
)
|
||||||
27
app/plan_d_action/utils/interface/config.py
Normal file
27
app/plan_d_action/utils/interface/config.py
Normal file
@ -0,0 +1,27 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
CORRESPONDANCE_COULEURS = {
|
||||||
|
"Rouge": "red",
|
||||||
|
"Orange": "orange",
|
||||||
|
"Vert": "green",
|
||||||
|
"FAIBLE": "green",
|
||||||
|
"MODÉRÉE": "orange",
|
||||||
|
"ÉLEVÉE à CRITIQUE": "red"
|
||||||
|
}
|
||||||
|
|
||||||
|
niveau_labels = {
|
||||||
|
0: "Produit final",
|
||||||
|
1: "Composant",
|
||||||
|
2: "Minerai",
|
||||||
|
10: "Opération",
|
||||||
|
11: "Pays d'opération",
|
||||||
|
12: "Acteur d'opération",
|
||||||
|
99: "Pays géographique"
|
||||||
|
}
|
||||||
|
|
||||||
|
# Répertoire courant du script
|
||||||
|
CURRENT_DIR = Path(__file__).resolve().parent.parent
|
||||||
|
|
||||||
|
# Répertoire "jobs" dans app/plan_d_action
|
||||||
|
JOBS = CURRENT_DIR / "jobs"
|
||||||
|
JOBS.mkdir(exist_ok=True)
|
||||||
33
app/plan_d_action/utils/interface/export.py
Normal file
33
app/plan_d_action/utils/interface/export.py
Normal file
@ -0,0 +1,33 @@
|
|||||||
|
import networkx as nx
|
||||||
|
|
||||||
|
def exporter_graphe_filtre(G, liens_chemins):
|
||||||
|
"""Gère l'export du graphe filtré au format DOT"""
|
||||||
|
|
||||||
|
G_export = nx.DiGraph()
|
||||||
|
for u, v in liens_chemins:
|
||||||
|
G_export.add_node(u, **G.nodes[u])
|
||||||
|
G_export.add_node(v, **G.nodes[v])
|
||||||
|
data = G.get_edge_data(u, v)
|
||||||
|
if isinstance(data, dict) and all(isinstance(k, int) for k in data):
|
||||||
|
G_export.add_edge(u, v, **data[0])
|
||||||
|
elif isinstance(data, dict):
|
||||||
|
G_export.add_edge(u, v, **data)
|
||||||
|
else:
|
||||||
|
G_export.add_edge(u, v)
|
||||||
|
|
||||||
|
return(G_export)
|
||||||
|
|
||||||
|
def extraire_liens_filtres(chemins, niveaux, niveau_depart, niveau_arrivee, niveaux_speciaux):
|
||||||
|
"""Extrait les liens des chemins en respectant les niveaux"""
|
||||||
|
liens = set()
|
||||||
|
for chemin in chemins:
|
||||||
|
for i in range(len(chemin) - 1):
|
||||||
|
u, v = chemin[i], chemin[i + 1]
|
||||||
|
niveau_u = niveaux.get(u, 999)
|
||||||
|
niveau_v = niveaux.get(v, 999)
|
||||||
|
if (
|
||||||
|
(niveau_depart <= niveau_u <= niveau_arrivee or niveau_u in niveaux_speciaux)
|
||||||
|
and (niveau_depart <= niveau_v <= niveau_arrivee or niveau_v in niveaux_speciaux)
|
||||||
|
):
|
||||||
|
liens.add((u, v))
|
||||||
|
return liens
|
||||||
8
app/plan_d_action/utils/interface/niveau_utils.py
Normal file
8
app/plan_d_action/utils/interface/niveau_utils.py
Normal file
@ -0,0 +1,8 @@
|
|||||||
|
def extraire_niveaux(G):
|
||||||
|
"""Extrait les niveaux des nœuds du graphe"""
|
||||||
|
niveaux = {}
|
||||||
|
for node, attrs in G.nodes(data=True):
|
||||||
|
niveau_str = attrs.get("niveau")
|
||||||
|
if niveau_str:
|
||||||
|
niveaux[node] = int(str(niveau_str).strip('"'))
|
||||||
|
return niveaux
|
||||||
11
app/plan_d_action/utils/interface/parser.py
Normal file
11
app/plan_d_action/utils/interface/parser.py
Normal file
@ -0,0 +1,11 @@
|
|||||||
|
def preparer_graphe(G):
|
||||||
|
"""Nettoie et prépare le graphe pour l'analyse."""
|
||||||
|
niveaux_temp = {
|
||||||
|
node: int(str(attrs.get("niveau")).strip('"'))
|
||||||
|
for node, attrs in G.nodes(data=True)
|
||||||
|
if attrs.get("niveau") and str(attrs.get("niveau")).strip('"').isdigit()
|
||||||
|
}
|
||||||
|
G.remove_nodes_from([n for n in G.nodes() if n not in niveaux_temp])
|
||||||
|
G.remove_nodes_from(
|
||||||
|
[n for n in G.nodes() if niveaux_temp.get(n) == 10 and 'Reserves' in n])
|
||||||
|
return G, niveaux_temp
|
||||||
74
app/plan_d_action/utils/interface/selection.py
Normal file
74
app/plan_d_action/utils/interface/selection.py
Normal file
@ -0,0 +1,74 @@
|
|||||||
|
import streamlit as st
|
||||||
|
import networkx as nx
|
||||||
|
from utils.translations import _
|
||||||
|
|
||||||
|
from utils.graph_utils import (
|
||||||
|
extraire_chemins_depuis,
|
||||||
|
extraire_chemins_vers
|
||||||
|
)
|
||||||
|
|
||||||
|
def selectionner_minerais(G, noeuds_depart):
|
||||||
|
"""Interface pour sélectionner les minerais si nécessaire."""
|
||||||
|
minerais_selection = None
|
||||||
|
|
||||||
|
st.markdown(f"## {str(_('pages.plan_d_action.select_minerals'))}")
|
||||||
|
|
||||||
|
# Étape 1 : récupérer tous les nœuds descendants depuis les produits finaux
|
||||||
|
descendants = set()
|
||||||
|
for start in noeuds_depart:
|
||||||
|
descendants.update(nx.descendants(G, start)) # tous les successeurs (récursifs)
|
||||||
|
|
||||||
|
# Étape 2 : ne garder que les nœuds de niveau 2 parmi les descendants
|
||||||
|
minerais_nodes = sorted([
|
||||||
|
n for n in descendants
|
||||||
|
if G.nodes[n].get("niveau") and int(str(G.nodes[n].get("niveau")).strip('"')) == 2
|
||||||
|
])
|
||||||
|
|
||||||
|
minerais_selection = st.multiselect(
|
||||||
|
str(_("pages.plan_d_action.filter_by_minerals")),
|
||||||
|
minerais_nodes,
|
||||||
|
key="analyse_minerais"
|
||||||
|
)
|
||||||
|
|
||||||
|
return minerais_selection
|
||||||
|
|
||||||
|
|
||||||
|
def selectionner_noeuds(G, niveaux_temp, niveau_depart):
|
||||||
|
"""Interface pour sélectionner les nœuds spécifiques de départ et d'arrivée."""
|
||||||
|
st.markdown("---")
|
||||||
|
st.markdown(f"## {str(_('pages.plan_d_action.fine_selection'))}")
|
||||||
|
|
||||||
|
depart_nodes = [n for n in G.nodes() if niveaux_temp.get(n) == niveau_depart]
|
||||||
|
noeuds_arrivee = [n for n in G.nodes() if niveaux_temp.get(n) == 99]
|
||||||
|
|
||||||
|
noeuds_depart = st.multiselect(str(_("pages.plan_d_action.filter_start_nodes")),
|
||||||
|
sorted(depart_nodes),
|
||||||
|
key="analyse_noeuds_depart")
|
||||||
|
|
||||||
|
noeuds_depart = noeuds_depart if noeuds_depart else None
|
||||||
|
|
||||||
|
return noeuds_depart, noeuds_arrivee
|
||||||
|
|
||||||
|
def extraire_chemins_selon_criteres(G, niveaux, niveau_depart, noeuds_depart, noeuds_arrivee, minerais):
|
||||||
|
"""Extrait les chemins selon les critères spécifiés"""
|
||||||
|
chemins = []
|
||||||
|
if noeuds_depart and noeuds_arrivee:
|
||||||
|
for nd in noeuds_depart:
|
||||||
|
for na in noeuds_arrivee:
|
||||||
|
tous_chemins = extraire_chemins_depuis(G, nd)
|
||||||
|
chemins.extend([chemin for chemin in tous_chemins if na in chemin])
|
||||||
|
elif noeuds_depart:
|
||||||
|
for nd in noeuds_depart:
|
||||||
|
chemins.extend(extraire_chemins_depuis(G, nd))
|
||||||
|
elif noeuds_arrivee:
|
||||||
|
for na in noeuds_arrivee:
|
||||||
|
chemins.extend(extraire_chemins_vers(G, na, niveau_depart))
|
||||||
|
else:
|
||||||
|
sources_depart = [n for n in G.nodes() if niveaux.get(n) == niveau_depart]
|
||||||
|
for nd in sources_depart:
|
||||||
|
chemins.extend(extraire_chemins_depuis(G, nd))
|
||||||
|
|
||||||
|
if minerais:
|
||||||
|
chemins = [chemin for chemin in chemins if any(n in minerais for n in chemin)]
|
||||||
|
|
||||||
|
return chemins
|
||||||
11
app/plan_d_action/utils/interface/visualization.py
Normal file
11
app/plan_d_action/utils/interface/visualization.py
Normal file
@ -0,0 +1,11 @@
|
|||||||
|
import re
|
||||||
|
from app.plan_d_action.utils.interface.config import CORRESPONDANCE_COULEURS
|
||||||
|
|
||||||
|
def remplacer_par_badge(markdown_text, correspondance=CORRESPONDANCE_COULEURS):
|
||||||
|
# Échappe les mots à remplacer s'ils contiennent des accents ou espaces
|
||||||
|
for mot, couleur in correspondance.items():
|
||||||
|
# Utilise des bords de mots (\b) pour éviter les remplacements partiels
|
||||||
|
pattern = r'\b' + re.escape(mot) + r'\b'
|
||||||
|
remplacement = f":{couleur}-badge[{mot}]"
|
||||||
|
markdown_text = re.sub(pattern, remplacement, markdown_text)
|
||||||
|
return markdown_text
|
||||||
11344
app/plan_d_action/utils/jobs/59479ea2.md
Normal file
11344
app/plan_d_action/utils/jobs/59479ea2.md
Normal file
File diff suppressed because it is too large
Load Diff
@ -55,8 +55,8 @@ def afficher_graphique_altair(df):
|
|||||||
base = alt.Chart(df_cat).encode(
|
base = alt.Chart(df_cat).encode(
|
||||||
x=alt.X('ihh_pays:Q', title=str(_("pages.visualisations.axis_titles.ihh_countries"))),
|
x=alt.X('ihh_pays:Q', title=str(_("pages.visualisations.axis_titles.ihh_countries"))),
|
||||||
y=alt.Y('ihh_acteurs:Q', title=str(_("pages.visualisations.axis_titles.ihh_actors"))),
|
y=alt.Y('ihh_acteurs:Q', title=str(_("pages.visualisations.axis_titles.ihh_actors"))),
|
||||||
size=alt.Size('criticite_cat:Q', scale=alt.Scale(domain=[1, 2, 3], range=[50, 500, 1000]), legend=None),
|
size=alt.Size('ics_cat:Q', scale=alt.Scale(domain=[1, 2, 3], range=[50, 500, 1000]), legend=None),
|
||||||
color=alt.Color('criticite_cat:N', scale=alt.Scale(domain=[1, 2, 3], range=['darkgreen', 'orange', 'darkred']))
|
color=alt.Color('ics_cat:N', scale=alt.Scale(domain=[1, 2, 3], range=['darkgreen', 'orange', 'darkred']))
|
||||||
)
|
)
|
||||||
|
|
||||||
points = base.mark_circle(opacity=0.6)
|
points = base.mark_circle(opacity=0.6)
|
||||||
@ -162,7 +162,7 @@ def creer_graphes(donnees):
|
|||||||
st.error(f"{str(_('errors.graph_creation_error'))} {e}")
|
st.error(f"{str(_('errors.graph_creation_error'))} {e}")
|
||||||
|
|
||||||
|
|
||||||
def lancer_visualisation_ihh_criticite(graph):
|
def lancer_visualisation_ihh_ics(graph):
|
||||||
try:
|
try:
|
||||||
import networkx as nx
|
import networkx as nx
|
||||||
from utils.graph_utils import recuperer_donnees
|
from utils.graph_utils import recuperer_donnees
|
||||||
|
|||||||
@ -1,25 +1,25 @@
|
|||||||
import streamlit as st
|
import streamlit as st
|
||||||
|
from utils.widgets import html_expander
|
||||||
from utils.translations import _
|
from utils.translations import _
|
||||||
|
|
||||||
from .graphes import (
|
from .graphes import (
|
||||||
lancer_visualisation_ihh_criticite,
|
lancer_visualisation_ihh_ics,
|
||||||
lancer_visualisation_ihh_ivc
|
lancer_visualisation_ihh_ivc
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def interface_visualisations(G_temp, G_temp_ivc):
|
def interface_visualisations(G_temp, G_temp_ivc):
|
||||||
st.markdown(f"# {str(_('pages.visualisations.title'))}")
|
st.markdown(f"# {str(_('pages.visualisations.title'))}")
|
||||||
with st.expander(str(_("pages.visualisations.help")), expanded=False):
|
html_expander(f"{str(_('pages.visualisations.help'))}", content="\n".join(_("pages.visualisations.help_content")), open_by_default=False, details_class="details_introduction")
|
||||||
st.markdown("\n".join(_("pages.visualisations.help_content")))
|
|
||||||
st.markdown("---")
|
st.markdown("---")
|
||||||
|
|
||||||
st.markdown(f"""## {str(_("pages.visualisations.ihh_criticality"))}
|
st.markdown(f"""## {str(_("pages.visualisations.ihh_criticality"))}
|
||||||
|
|
||||||
{str(_("pages.visualisations.ihh_criticality_desc"))}
|
{str(_("pages.visualisations.ihh_criticality_desc"))}
|
||||||
""")
|
""")
|
||||||
if st.button(str(_("buttons.run")), key="btn_ihh_criticite"):
|
if st.button(str(_("buttons.run")), key="btn_ihh_ics", icon=":material/bubble_chart:"):
|
||||||
try:
|
try:
|
||||||
lancer_visualisation_ihh_criticite(G_temp)
|
lancer_visualisation_ihh_ics(G_temp)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
st.error(f"{str(_('errors.ihh_criticality_error'))} {e}")
|
st.error(f"{str(_('errors.ihh_criticality_error'))} {e}")
|
||||||
|
|
||||||
@ -28,7 +28,7 @@ def interface_visualisations(G_temp, G_temp_ivc):
|
|||||||
{str(_("pages.visualisations.ihh_ivc_desc"))}
|
{str(_("pages.visualisations.ihh_ivc_desc"))}
|
||||||
""")
|
""")
|
||||||
|
|
||||||
if st.button(str(_("buttons.run")), key="btn_ihh_ivc"):
|
if st.button(str(_("buttons.run")), key="btn_ihh_ivc", icon=":material/bubble_chart:"):
|
||||||
try:
|
try:
|
||||||
lancer_visualisation_ihh_ivc(G_temp_ivc)
|
lancer_visualisation_ihh_ivc(G_temp_ivc)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|||||||
@ -2,9 +2,12 @@
|
|||||||
|
|
||||||
## Styles
|
## Styles
|
||||||
|
|
||||||
Le fichier **styles.css** a été construit pour agir sur le styme produit par Streamlit ou pour décorer des éléments construits par fabnum.py
|
Le fichier **base.css** a été construit pour agir sur le style produit par Streamlit ou pour décorer des éléments construits par fabnum.py, indépendamment du thème choisi
|
||||||
|
Les deux fichiers **theme-light.css** et **theme-dark.css** contiennent les variables utilisées par base.css pour afficher les couleurs.
|
||||||
|
|
||||||
Il sera important de regarder s'il est possible d'interagir avec le css de Streamlit sans passer par des déclarations !important
|
Streamlit utilise le theme par défaut du système du poste de travail de l'internaute. Afin de maîtriser complètement le thème avec base.css, il est important que la configuration côté serveur de Streamlit soit forcée au thème light dans le fichier .streamlit/config.toml :
|
||||||
|
[theme]
|
||||||
|
base = "light"
|
||||||
|
|
||||||
## Icone
|
## Icone
|
||||||
|
|
||||||
|
|||||||
@ -1,11 +1,11 @@
|
|||||||
version: 1.1
|
version: 1.1
|
||||||
date: 2025-05-06
|
date: 2025-05-27
|
||||||
|
|
||||||
seuils:
|
seuils:
|
||||||
IVC: # Indice de vulnérabilité concurrentielle
|
IVC: # Indice de vulnérabilité concurrentielle
|
||||||
vert: { max: 5 }
|
vert: { max: 15 }
|
||||||
orange: { min: 5, max: 15 }
|
orange: { min: 15, max: 60 }
|
||||||
rouge: { min: 15 }
|
rouge: { min: 60 }
|
||||||
|
|
||||||
IHH: # Index Herfindahl-Hirschman
|
IHH: # Index Herfindahl-Hirschman
|
||||||
vert: { max: 15 }
|
vert: { max: 15 }
|
||||||
|
|||||||
@ -1,276 +1,331 @@
|
|||||||
{
|
{
|
||||||
"app": {
|
"app": {
|
||||||
"title": "Fabnum – Chain Analysis",
|
"title": "Fabnum – Chain Analysis",
|
||||||
"description": "Ecosystem exploration and vulnerability identification.",
|
"description": "Ecosystem exploration and vulnerability identification.",
|
||||||
"dev_mode": "You are in the development environment."
|
"dev_mode": "You are in the development environment."
|
||||||
},
|
|
||||||
"header": {
|
|
||||||
"title": "FabNum - Digital Manufacturing Chain",
|
|
||||||
"subtitle": "Ecosystem exploration and vulnerability identification."
|
|
||||||
},
|
|
||||||
"footer": {
|
|
||||||
"copyright": "Fabnum © 2025",
|
|
||||||
"contact": "Contact",
|
|
||||||
"license": "License",
|
|
||||||
"license_text": "CC BY-NC-ND",
|
|
||||||
"eco_note": "🌱 CO₂ calculations via",
|
|
||||||
"eco_provider": "The Green Web Foundation",
|
|
||||||
"powered_by": "🚀 Powered by",
|
|
||||||
"powered_by_name": "Streamlit"
|
|
||||||
},
|
|
||||||
"sidebar": {
|
|
||||||
"menu": "Main Menu",
|
|
||||||
"navigation": "Main Navigation",
|
|
||||||
"theme": "Theme",
|
|
||||||
"theme_light": "Light",
|
|
||||||
"theme_dark": "Dark",
|
|
||||||
"theme_instructions_only": "Theme changes can only be made from the Instructions tab.",
|
|
||||||
"impact": "Environmental Impact",
|
|
||||||
"loading": "Loading..."
|
|
||||||
},
|
|
||||||
"auth": {
|
|
||||||
"title": "Authentication",
|
|
||||||
"username": "Username_token",
|
|
||||||
"token": "Gitea Personal Access Token",
|
|
||||||
"login": "Login",
|
|
||||||
"logout": "Logout",
|
|
||||||
"logged_as": "Logged in as",
|
|
||||||
"error": "❌ Access denied.",
|
|
||||||
"gitea_error": "❌ Unable to verify user with Gitea.",
|
|
||||||
"success": "Successfully logged out."
|
|
||||||
},
|
|
||||||
"navigation": {
|
|
||||||
"instructions": "Instructions",
|
|
||||||
"personnalisation": "Customization",
|
|
||||||
"analyse": "Analysis",
|
|
||||||
"visualisations": "Visualizations",
|
|
||||||
"fiches": "Cards"
|
|
||||||
},
|
|
||||||
"pages": {
|
|
||||||
"instructions": {
|
|
||||||
"title": "Instructions"
|
|
||||||
},
|
},
|
||||||
"personnalisation": {
|
"header": {
|
||||||
"title": "Final Product Customization",
|
"title": "FabNum - Digital Manufacturing Chain",
|
||||||
"help": "How to use this tab?",
|
"subtitle": "Ecosystem exploration and vulnerability identification."
|
||||||
"help_content": [
|
|
||||||
"1. Click on \"Add a final product\" to create a new product",
|
|
||||||
"2. Give your product a name",
|
|
||||||
"3. Select an appropriate assembly operation (if relevant)",
|
|
||||||
"4. Choose the components that make up your product from the list provided",
|
|
||||||
"5. Save your configuration for future reuse",
|
|
||||||
"6. You will be able to modify or delete your custom products later"
|
|
||||||
],
|
|
||||||
"add_new_product": "Add a new final product",
|
|
||||||
"new_product_name": "New product name (unique)",
|
|
||||||
"assembly_operation": "Assembly operation (optional)",
|
|
||||||
"none": "-- None --",
|
|
||||||
"components_to_link": "Components to link",
|
|
||||||
"create_product": "Create product",
|
|
||||||
"added": "added",
|
|
||||||
"modify_product": "Modify an added final product",
|
|
||||||
"products_to_modify": "Products to modify",
|
|
||||||
"delete": "Delete",
|
|
||||||
"linked_assembly_operation": "Linked assembly operation",
|
|
||||||
"components_linked_to": "Components linked to",
|
|
||||||
"update": "Update",
|
|
||||||
"updated": "updated",
|
|
||||||
"deleted": "deleted",
|
|
||||||
"save_restore_config": "Save or restore configuration",
|
|
||||||
"export_config": "Export configuration",
|
|
||||||
"download_json": "Download (JSON)",
|
|
||||||
"import_config": "Import a JSON configuration (max 100 KB)",
|
|
||||||
"file_too_large": "File too large (max 100 KB).",
|
|
||||||
"no_products_found": "No products found in the file.",
|
|
||||||
"select_products_to_restore": "Select products to restore",
|
|
||||||
"products_to_restore": "Products to restore",
|
|
||||||
"restore_selected": "Restore selected items",
|
|
||||||
"config_restored": "Partial configuration successfully restored.",
|
|
||||||
"import_error": "Import error:"
|
|
||||||
},
|
},
|
||||||
"analyse": {
|
"footer": {
|
||||||
"title": "Graph Analysis",
|
"copyright": "Fabnum © 2025",
|
||||||
"help": "How to use this tab?",
|
"contact": "Contact",
|
||||||
"help_content": [
|
"license": "License",
|
||||||
"1. Select the starting level (final product, component, or mineral)",
|
"license_text": "CC BY-NC-ND",
|
||||||
"2. Choose the desired destination level",
|
"eco_note": "🌱 CO₂ calculations via",
|
||||||
"3. Refine your selection by specifying either one or more specific minerals to target or specific items at each level (optional)",
|
"eco_provider": "The Green Web Foundation",
|
||||||
"4. Define the analysis criteria by selecting the relevant vulnerability indices",
|
"powered_by": "🚀 Powered by",
|
||||||
"5. Choose the index combination mode (AND/OR) according to your analysis needs",
|
"powered_by_name": "Streamlit"
|
||||||
"6. Explore the generated graph using zoom and panning controls; you can switch to full screen mode for the graph"
|
|
||||||
],
|
|
||||||
"selection_nodes": "Selection of start and end nodes",
|
|
||||||
"select_level": "-- Select a level --",
|
|
||||||
"start_level": "Start level",
|
|
||||||
"end_level": "End level",
|
|
||||||
"select_minerals": "Select one or more minerals",
|
|
||||||
"filter_by_minerals": "Filter by minerals (optional)",
|
|
||||||
"fine_selection": "Fine selection of items",
|
|
||||||
"filter_start_nodes": "Filter by start nodes (optional)",
|
|
||||||
"filter_end_nodes": "Filter by end nodes (optional)",
|
|
||||||
"vulnerability_filters": "Selection of filters to identify vulnerabilities",
|
|
||||||
"filter_ics": "Filter paths containing at least one critical mineral for a component (ICS > 66%)",
|
|
||||||
"filter_ivc": "Filter paths containing at least one critical mineral in relation to sectoral competition (IVC > 30)",
|
|
||||||
"filter_ihh": "Filter paths containing at least one critical operation in relation to geographical or industrial concentration (IHH countries or actors > 25)",
|
|
||||||
"apply_ihh_filter": "Apply IHH filter on:",
|
|
||||||
"countries": "Countries",
|
|
||||||
"actors": "Actors",
|
|
||||||
"filter_isg": "Filter paths containing an unstable country (ISG ≥ 60)",
|
|
||||||
"filter_logic": "Filter logic",
|
|
||||||
"or": "OR",
|
|
||||||
"and": "AND",
|
|
||||||
"run_analysis": "Run analysis",
|
|
||||||
"sankey": {
|
|
||||||
"no_paths": "No paths found for the specified criteria.",
|
|
||||||
"no_matching_paths": "No paths match the criteria.",
|
|
||||||
"filtered_hierarchy": "Hierarchy filtered by levels and nodes",
|
|
||||||
"download_dot": "Download filtered DOT file",
|
|
||||||
"relation": "Relation"
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
"visualisations": {
|
"sidebar": {
|
||||||
"title": "Visualizations",
|
"menu": "Main Menu",
|
||||||
"help": "How to use this tab?",
|
"navigation": "Main Navigation",
|
||||||
"help_content": [
|
"theme": "Theme",
|
||||||
"1. Explore the graphs presenting the Herfindahl-Hirschmann Index (IHH)",
|
"theme_light": "Light",
|
||||||
"2. Analyze its relationship with the average criticality of minerals or their Competitive Vulnerability Index (IVC)",
|
"theme_dark": "Dark",
|
||||||
"3. Zoom in on the graphs to better discover the information",
|
"theme_instructions_only": "Theme changes can only be made from the Instructions tab.",
|
||||||
"",
|
"impact": "Environmental Impact",
|
||||||
"It is important to remember that the IHH has two thresholds:",
|
"loading": "Loading..."
|
||||||
"* below 15, concentration is considered to be low",
|
|
||||||
"* above 25, it is considered to be high",
|
|
||||||
"",
|
|
||||||
"Thus, the higher a point is positioned in the top right of the graphs, the higher the risks.",
|
|
||||||
"The graphs present 2 horizontal and vertical lines to mark these thresholds."
|
|
||||||
],
|
|
||||||
"ihh_criticality": "Herfindahl-Hirschmann Index - IHH vs Criticality",
|
|
||||||
"ihh_criticality_desc": "The size of the points indicates the substitutability criticality of the mineral.",
|
|
||||||
"ihh_ivc": "Herfindahl-Hirschmann Index - IHH vs IVC",
|
|
||||||
"ihh_ivc_desc": "The size of the points indicates the competitive criticality of the mineral.",
|
|
||||||
"launch": "Launch",
|
|
||||||
"no_data": "No data to display.",
|
|
||||||
"categories": {
|
|
||||||
"assembly": "Assembly",
|
|
||||||
"manufacturing": "Manufacturing",
|
|
||||||
"processing": "Processing",
|
|
||||||
"extraction": "Extraction"
|
|
||||||
},
|
|
||||||
"axis_titles": {
|
|
||||||
"ihh_countries": "IHH Countries (%)",
|
|
||||||
"ihh_actors": "IHH Actors (%)",
|
|
||||||
"ihh_extraction": "IHH Extraction (%)",
|
|
||||||
"ihh_reserves": "IHH Reserves (%)"
|
|
||||||
},
|
|
||||||
"chart_titles": {
|
|
||||||
"concentration_criticality": "Concentration and Criticality – {0}",
|
|
||||||
"concentration_resources": "Concentration of Critical Resources vs IVC Vulnerability"
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
"fiches": {
|
"auth": {
|
||||||
"title": "Card Discovery",
|
"title": "Authentication",
|
||||||
"help": "How to use this tab?",
|
"username": "Username_token",
|
||||||
"help_content": [
|
"token": "Gitea Personal Access Token",
|
||||||
"1. Browse the list of available cards by category",
|
"login": "Login",
|
||||||
"2. Select a card to display its full content",
|
"logout": "Logout",
|
||||||
"3. Consult detailed data, graphs, and additional analyses",
|
"logged_as": "Logged in as",
|
||||||
"4. Use this information to deepen your understanding of the identified vulnerabilities",
|
"error": "❌ Access denied.",
|
||||||
"",
|
"gitea_error": "❌ Unable to verify user with Gitea.",
|
||||||
"The categories are as follows:",
|
"success": "Successfully logged out."
|
||||||
"* Assembly: operation of assembling final products from components",
|
},
|
||||||
"* Related: various operations necessary to manufacture digital technology, but not directly entering its composition",
|
"navigation": {
|
||||||
"* Criticalities: indices used to identify and evaluate vulnerabilities",
|
"instructions": "Instructions",
|
||||||
"* Manufacturing: operation of manufacturing components from minerals",
|
"personnalisation": "Customization",
|
||||||
"* Mineral: description and operations of extraction and processing of minerals"
|
"analyse": "Analysis",
|
||||||
],
|
"ia_nalyse": "AI'nalysis",
|
||||||
"no_files": "No cards available at the moment.",
|
"plan_d_action": "Actions plan",
|
||||||
"choose_category": "Choose a card category",
|
"visualisations": "Visualizations",
|
||||||
"select_folder": "-- Select a folder --",
|
"fiches": "Cards"
|
||||||
"choose_file": "Choose a card",
|
},
|
||||||
"select_file": "-- Select a card --",
|
"pages": {
|
||||||
"loading_error": "Error loading the card:",
|
"instructions": {
|
||||||
"download_pdf": "Download this card as PDF",
|
"title": "Instructions"
|
||||||
"pdf_unavailable": "The PDF file for this card is not available.",
|
|
||||||
"ticket_management": "Ticket management for this card",
|
|
||||||
"tickets": {
|
|
||||||
"create_new": "Create a new ticket linked to this card",
|
|
||||||
"model_load_error": "Unable to load the ticket template.",
|
|
||||||
"contribution_type": "Contribution type",
|
|
||||||
"specify": "Specify",
|
|
||||||
"other": "Other",
|
|
||||||
"concerned_card": "Concerned card",
|
|
||||||
"subject": "Subject of the proposal",
|
|
||||||
"preview": "Preview ticket",
|
|
||||||
"cancel": "Cancel",
|
|
||||||
"preview_title": "Ticket preview",
|
|
||||||
"summary": "Summary",
|
|
||||||
"title": "Title",
|
|
||||||
"labels": "Labels",
|
|
||||||
"confirm": "Confirm ticket creation",
|
|
||||||
"created": "Ticket created and form cleared.",
|
|
||||||
"model_error": "Template loading error:",
|
|
||||||
"no_linked_tickets": "No tickets linked to this card.",
|
|
||||||
"associated_tickets": "Tickets associated with this card",
|
|
||||||
"moderation_notice": "ticket(s) awaiting moderation are not displayed.",
|
|
||||||
"status": {
|
|
||||||
"awaiting": "Awaiting processing",
|
|
||||||
"in_progress": "In progress",
|
|
||||||
"completed": "Completed",
|
|
||||||
"rejected": "Rejected",
|
|
||||||
"others": "Others"
|
|
||||||
},
|
},
|
||||||
"no_title": "No title",
|
"personnalisation": {
|
||||||
"unknown": "unknown",
|
"title": "Final Product Customization",
|
||||||
"subject_label": "Subject",
|
"help": "How to use this tab?",
|
||||||
"no_labels": "none",
|
"help_content": [
|
||||||
"comments": "Comment(s):",
|
"1. Click on \"Add a final product\" to create a new product",
|
||||||
"no_comments": "No comments.",
|
"2. Give your product a name",
|
||||||
"comment_error": "Error retrieving comments:",
|
"3. Select an appropriate assembly operation (if relevant)",
|
||||||
"opened_by": "Opened by",
|
"4. Choose the components that make up your product from the list provided",
|
||||||
"on_date": "on",
|
"5. Save your configuration for future reuse",
|
||||||
"updated": "UPDATED"
|
"6. You will be able to modify or delete your custom products later"
|
||||||
}
|
],
|
||||||
|
"add_new_product": "Add a new final product",
|
||||||
|
"new_product_name": "New product name (unique)",
|
||||||
|
"assembly_operation": "Assembly operation (optional)",
|
||||||
|
"none": "-- None --",
|
||||||
|
"components_to_link": "Components to link",
|
||||||
|
"create_product": "Create product",
|
||||||
|
"added": "added",
|
||||||
|
"modify_product": "Modify an added final product",
|
||||||
|
"products_to_modify": "Products to modify",
|
||||||
|
"delete": "Delete",
|
||||||
|
"linked_assembly_operation": "Linked assembly operation",
|
||||||
|
"components_linked_to": "Components linked to",
|
||||||
|
"update": "Update",
|
||||||
|
"updated": "updated",
|
||||||
|
"deleted": "deleted",
|
||||||
|
"save_restore_config": "Save or restore configuration",
|
||||||
|
"export_config": "Export configuration",
|
||||||
|
"download_json": "Download (JSON)",
|
||||||
|
"import_config": "Import a JSON configuration (max 100 KB)",
|
||||||
|
"file_too_large": "File too large (max 100 KB).",
|
||||||
|
"no_products_found": "No products found in the file.",
|
||||||
|
"select_products_to_restore": "Select products to restore",
|
||||||
|
"products_to_restore": "Products to restore",
|
||||||
|
"restore_selected": "Restore selected items",
|
||||||
|
"config_restored": "Partial configuration successfully restored.",
|
||||||
|
"import_error": "Import error:"
|
||||||
|
},
|
||||||
|
"analyse": {
|
||||||
|
"title": "Graph Analysis",
|
||||||
|
"help": "How to use this tab?",
|
||||||
|
"help_content": [
|
||||||
|
"1. Select the starting level (final product, component, or mineral)",
|
||||||
|
"2. Choose the desired destination level",
|
||||||
|
"3. Refine your selection by specifying either one or more specific minerals to target or specific items at each level (optional)",
|
||||||
|
"4. Define the analysis criteria by selecting the relevant vulnerability indices",
|
||||||
|
"5. Choose the index combination mode (AND/OR) according to your analysis needs",
|
||||||
|
"6. Explore the generated graph using zoom and panning controls; you can switch to full screen mode for the graph"
|
||||||
|
],
|
||||||
|
"selection_nodes": "Selection of start and end nodes",
|
||||||
|
"select_level": "-- Select a level --",
|
||||||
|
"start_level": "Start level",
|
||||||
|
"end_level": "End level",
|
||||||
|
"select_minerals": "Select one or more minerals",
|
||||||
|
"filter_by_minerals": "Filter by minerals (optional)",
|
||||||
|
"fine_selection": "Fine selection of items",
|
||||||
|
"filter_start_nodes": "Filter by start nodes (optional)",
|
||||||
|
"filter_end_nodes": "Filter by end nodes (optional)",
|
||||||
|
"vulnerability_filters": "Selection of filters to identify vulnerabilities",
|
||||||
|
"filter_ics": "Filter paths containing at least one critical mineral for a component (ICS > 66%)",
|
||||||
|
"filter_ivc": "Filter paths containing at least one critical mineral in relation to sectoral competition (IVC > 30)",
|
||||||
|
"filter_ihh": "Filter paths containing at least one critical operation in relation to geographical or industrial concentration (IHH countries or actors > 25)",
|
||||||
|
"apply_ihh_filter": "Apply IHH filter on:",
|
||||||
|
"countries": "Countries",
|
||||||
|
"actors": "Actors",
|
||||||
|
"filter_isg": "Filter paths containing an unstable country (ISG ≥ 60)",
|
||||||
|
"filter_logic": "Filter logic",
|
||||||
|
"or": "OR",
|
||||||
|
"and": "AND",
|
||||||
|
"run_analysis": "Run analysis",
|
||||||
|
"sankey": {
|
||||||
|
"no_paths": "No paths found for the specified criteria.",
|
||||||
|
"no_matching_paths": "No paths match the criteria.",
|
||||||
|
"filtered_hierarchy": "Hierarchy filtered by levels and nodes",
|
||||||
|
"download_dot": "Download filtered DOT file",
|
||||||
|
"relation": "Relation"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ia_nalyse": {
|
||||||
|
"title": "Graph Analysis by AI",
|
||||||
|
"help": "How to use this tab?",
|
||||||
|
"help_content": [
|
||||||
|
"The graph covers all levels, from end products to geographic countries.\n",
|
||||||
|
"1. You can select minerals through which the paths go.",
|
||||||
|
"2. You can choose end products to perform an analysis tailored to your context.\n",
|
||||||
|
"These two selections are optional, but strongly recommended for a more relevant analysis.",
|
||||||
|
"The analysis is carried out using a private AI on a minimalist server. The result is therefore not immediate (approximately 30 minutes) and you will be notified of the progress."
|
||||||
|
],
|
||||||
|
"select_minerals": "Select one or more minerals",
|
||||||
|
"filter_by_minerals": "Filter by minerals (optional, but highly recommended)",
|
||||||
|
"fine_selection": "End product selection",
|
||||||
|
"filter_start_nodes": "Filter by start nodes (optional, but recommended)",
|
||||||
|
"run_analysis": "Run analysis",
|
||||||
|
"confirm_download": "Confirm download",
|
||||||
|
"submit_request": "Submit your request",
|
||||||
|
"empty_graph": "The graph is empty. Please make another selection."
|
||||||
|
},
|
||||||
|
"plan_d_action": {
|
||||||
|
"title": "Graph analysis for action",
|
||||||
|
"help": "How to use this tab?",
|
||||||
|
"help_content": [
|
||||||
|
"The graph covers all levels, from end products to geographic countries.\n",
|
||||||
|
"1. You can select minerals through which the paths go.",
|
||||||
|
"2. You can choose end products to perform an analysis tailored to your context.\n",
|
||||||
|
"These two selections are optional, but strongly recommended for a more relevant analysis.",
|
||||||
|
"The recommendations for actions and indicator monitoring are generic. They must therefore be adapted to the context.",
|
||||||
|
"The proposed actions or indicators depend on the type of organization concerned and can be applied directly or required of digital providers."
|
||||||
|
],
|
||||||
|
"select_minerals": "Select one or more minerals",
|
||||||
|
"filter_by_minerals": "Filter by minerals (optional, but highly recommended)",
|
||||||
|
"fine_selection": "End product selection",
|
||||||
|
"filter_start_nodes": "Filter by start nodes (optional, but recommended)",
|
||||||
|
"run_analysis": "Run analysis",
|
||||||
|
"confirm_download": "Confirm download",
|
||||||
|
"submit_request": "Submit your request",
|
||||||
|
"empty_graph": "The graph is empty. Please make another selection."
|
||||||
|
},
|
||||||
|
"visualisations": {
|
||||||
|
"title": "Visualizations",
|
||||||
|
"help": "How to use this tab?",
|
||||||
|
"help_content": [
|
||||||
|
"1. Explore the graphs presenting the Herfindahl-Hirschmann Index (IHH)",
|
||||||
|
"2. Analyze its relationship with the average criticality of minerals or their Competitive Vulnerability Index (IVC)",
|
||||||
|
"3. Zoom in on the graphs to better discover the information",
|
||||||
|
"",
|
||||||
|
"It is important to remember that the IHH has two thresholds:",
|
||||||
|
"* below 15, concentration is considered to be low",
|
||||||
|
"* above 25, it is considered to be high",
|
||||||
|
"",
|
||||||
|
"Thus, the higher a point is positioned in the top right of the graphs, the higher the risks.",
|
||||||
|
"The graphs present 2 horizontal and vertical lines to mark these thresholds."
|
||||||
|
],
|
||||||
|
"ihh_criticality": "Herfindahl-Hirschmann Index - IHH vs Criticality",
|
||||||
|
"ihh_criticality_desc": "The size of the points indicates the substitutability criticality of the mineral.",
|
||||||
|
"ihh_ivc": "Herfindahl-Hirschmann Index - IHH vs IVC",
|
||||||
|
"ihh_ivc_desc": "The size of the points indicates the competitive criticality of the mineral.",
|
||||||
|
"launch": "Launch",
|
||||||
|
"no_data": "No data to display.",
|
||||||
|
"categories": {
|
||||||
|
"assembly": "Assembly",
|
||||||
|
"manufacturing": "Manufacturing",
|
||||||
|
"processing": "Processing",
|
||||||
|
"extraction": "Extraction"
|
||||||
|
},
|
||||||
|
"axis_titles": {
|
||||||
|
"ihh_countries": "IHH Countries (%)",
|
||||||
|
"ihh_actors": "IHH Actors (%)",
|
||||||
|
"ihh_extraction": "IHH Extraction (%)",
|
||||||
|
"ihh_reserves": "IHH Reserves (%)"
|
||||||
|
},
|
||||||
|
"chart_titles": {
|
||||||
|
"concentration_criticality": "Concentration and Criticality – {0}",
|
||||||
|
"concentration_resources": "Concentration of Critical Resources vs IVC Vulnerability"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"fiches": {
|
||||||
|
"title": "Card Discovery",
|
||||||
|
"help": "How to use this tab?",
|
||||||
|
"help_content": [
|
||||||
|
"1. Browse the list of available cards by category",
|
||||||
|
"2. Select a card to display its full content",
|
||||||
|
"3. Consult detailed data, graphs, and additional analyses",
|
||||||
|
"4. Use this information to deepen your understanding of the identified vulnerabilities",
|
||||||
|
"",
|
||||||
|
"The categories are as follows:",
|
||||||
|
"* Assembly: operation of assembling final products from components",
|
||||||
|
"* Related: various operations necessary to manufacture digital technology, but not directly entering its composition",
|
||||||
|
"* Criticalities: indices used to identify and evaluate vulnerabilities",
|
||||||
|
"* Manufacturing: operation of manufacturing components from minerals",
|
||||||
|
"* Mineral: description and operations of extraction and processing of minerals"
|
||||||
|
],
|
||||||
|
"no_files": "No cards available at the moment.",
|
||||||
|
"choose_category": "Choose a card category",
|
||||||
|
"select_folder": "-- Select a folder --",
|
||||||
|
"choose_file": "Choose a card",
|
||||||
|
"select_file": "-- Select a card --",
|
||||||
|
"loading_error": "Error loading the card:",
|
||||||
|
"download_pdf": "Download this card as PDF",
|
||||||
|
"pdf_unavailable": "The PDF file for this card is not available.",
|
||||||
|
"ticket_management": "Ticket management for this card",
|
||||||
|
"tickets": {
|
||||||
|
"create_new": "Create a new ticket linked to this card",
|
||||||
|
"model_load_error": "Unable to load the ticket template.",
|
||||||
|
"contribution_type": "Contribution type",
|
||||||
|
"specify": "Specify",
|
||||||
|
"other": "Other",
|
||||||
|
"concerned_card": "Concerned card",
|
||||||
|
"subject": "Subject of the proposal",
|
||||||
|
"preview": "Preview ticket",
|
||||||
|
"cancel": "Cancel",
|
||||||
|
"preview_title": "Ticket preview",
|
||||||
|
"summary": "Summary",
|
||||||
|
"title": "Title",
|
||||||
|
"labels": "Labels",
|
||||||
|
"confirm": "Confirm ticket creation",
|
||||||
|
"created": "Ticket created and form cleared.",
|
||||||
|
"model_error": "Template loading error:",
|
||||||
|
"no_linked_tickets": "No tickets linked to this card.",
|
||||||
|
"associated_tickets": "Tickets associated with this card",
|
||||||
|
"moderation_notice": "ticket(s) awaiting moderation are not displayed.",
|
||||||
|
"status": {
|
||||||
|
"awaiting": "Awaiting processing",
|
||||||
|
"in_progress": "In progress",
|
||||||
|
"completed": "Completed",
|
||||||
|
"rejected": "Rejected",
|
||||||
|
"others": "Others"
|
||||||
|
},
|
||||||
|
"no_title": "No title",
|
||||||
|
"unknown": "unknown",
|
||||||
|
"subject_label": "Subject",
|
||||||
|
"no_labels": "none",
|
||||||
|
"comments": "Comment(s):",
|
||||||
|
"no_comments": "No comments.",
|
||||||
|
"comment_error": "Error retrieving comments:",
|
||||||
|
"opened_by": "Opened by",
|
||||||
|
"on_date": "on",
|
||||||
|
"updated": "UPDATED",
|
||||||
|
"continue": "Continuer",
|
||||||
|
"created_success": "Ticket created and placed in moderation",
|
||||||
|
"created_error": "Ticket creation failed. Please try later",
|
||||||
|
"see_ticket": "See ticket"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"node_levels": {
|
||||||
|
"0": "Final product",
|
||||||
|
"1": "Component",
|
||||||
|
"2": "Mineral",
|
||||||
|
"10": "Operation",
|
||||||
|
"11": "Operation country",
|
||||||
|
"12": "Operation actor",
|
||||||
|
"99": "Geographic country"
|
||||||
|
},
|
||||||
|
"errors": {
|
||||||
|
"log_read_error": "Log reading error:",
|
||||||
|
"graph_preview_error": "Graph preview error:",
|
||||||
|
"graph_creation_error": "Error creating the graph:",
|
||||||
|
"ihh_criticality_error": "Error in IHH vs Criticality visualization:",
|
||||||
|
"ihh_ivc_error": "Error in IHH vs IVC visualization:",
|
||||||
|
"comment_fetch_error": "Error retrieving comments:",
|
||||||
|
"template_load_error": "Template loading error:",
|
||||||
|
"import_error": "Import error:"
|
||||||
|
},
|
||||||
|
"buttons": {
|
||||||
|
"download": "Download",
|
||||||
|
"run": "Run",
|
||||||
|
"save": "Save",
|
||||||
|
"cancel": "Cancel",
|
||||||
|
"confirm": "Confirm",
|
||||||
|
"filter": "Filter",
|
||||||
|
"search": "Search",
|
||||||
|
"create": "Create",
|
||||||
|
"update": "Update",
|
||||||
|
"delete": "Delete",
|
||||||
|
"preview": "Preview",
|
||||||
|
"export": "Export",
|
||||||
|
"import": "Import",
|
||||||
|
"restore": "Restore",
|
||||||
|
"refresh": "Refresh",
|
||||||
|
"browse_files": "Browse files"
|
||||||
|
},
|
||||||
|
"ui": {
|
||||||
|
"file_uploader": {
|
||||||
|
"drag_drop_here": "Drag and drop file here",
|
||||||
|
"size_limit": "100 KB limit per file • JSON"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"batch": {
|
||||||
|
"in_queue": "In queue",
|
||||||
|
"in_progress": "Analysis in progress",
|
||||||
|
"failure": "Error",
|
||||||
|
"unknwon_error": "unknown error",
|
||||||
|
"no_task": "No task wainting or in progress",
|
||||||
|
"complete": "Analysis complete. Download the result in zip format, which contains the detailed report and analysis.",
|
||||||
|
"step": "Step"
|
||||||
}
|
}
|
||||||
},
|
|
||||||
"node_levels": {
|
|
||||||
"0": "Final product",
|
|
||||||
"1": "Component",
|
|
||||||
"2": "Mineral",
|
|
||||||
"10": "Operation",
|
|
||||||
"11": "Operation country",
|
|
||||||
"12": "Operation actor",
|
|
||||||
"99": "Geographic country"
|
|
||||||
},
|
|
||||||
"errors": {
|
|
||||||
"log_read_error": "Log reading error:",
|
|
||||||
"graph_preview_error": "Graph preview error:",
|
|
||||||
"graph_creation_error": "Error creating the graph:",
|
|
||||||
"ihh_criticality_error": "Error in IHH vs Criticality visualization:",
|
|
||||||
"ihh_ivc_error": "Error in IHH vs IVC visualization:",
|
|
||||||
"comment_fetch_error": "Error retrieving comments:",
|
|
||||||
"template_load_error": "Template loading error:",
|
|
||||||
"import_error": "Import error:"
|
|
||||||
},
|
|
||||||
"buttons": {
|
|
||||||
"download": "Download",
|
|
||||||
"run": "Run",
|
|
||||||
"save": "Save",
|
|
||||||
"cancel": "Cancel",
|
|
||||||
"confirm": "Confirm",
|
|
||||||
"filter": "Filter",
|
|
||||||
"search": "Search",
|
|
||||||
"create": "Create",
|
|
||||||
"update": "Update",
|
|
||||||
"delete": "Delete",
|
|
||||||
"preview": "Preview",
|
|
||||||
"export": "Export",
|
|
||||||
"import": "Import",
|
|
||||||
"restore": "Restore",
|
|
||||||
"browse_files": "Browse files"
|
|
||||||
},
|
|
||||||
"ui": {
|
|
||||||
"file_uploader": {
|
|
||||||
"drag_drop_here": "Drag and drop file here",
|
|
||||||
"size_limit": "100 KB limit per file • JSON"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@ -1,276 +1,331 @@
|
|||||||
{
|
{
|
||||||
"app": {
|
"app": {
|
||||||
"title": "Fabnum – Analyse de chaîne",
|
"title": "Fabnum – Analyse de chaîne",
|
||||||
"description": "Parcours de l'écosystème et identification des vulnérabilités.",
|
"description": "Parcours de l'écosystème et identification des vulnérabilités.",
|
||||||
"dev_mode": "Vous êtes dans l'environnement de développement."
|
"dev_mode": "Vous êtes dans l'environnement de développement."
|
||||||
},
|
|
||||||
"header": {
|
|
||||||
"title": "FabNum - Chaîne de fabrication du numérique",
|
|
||||||
"subtitle": "Parcours de l'écosystème et identification des vulnérabilités."
|
|
||||||
},
|
|
||||||
"footer": {
|
|
||||||
"copyright": "Fabnum © 2025",
|
|
||||||
"contact": "Contact",
|
|
||||||
"license": "Licence",
|
|
||||||
"license_text": "CC BY-NC-ND",
|
|
||||||
"eco_note": "🌱 Calculs CO₂ via",
|
|
||||||
"eco_provider": "The Green Web Foundation",
|
|
||||||
"powered_by": "🚀 Propulsé par",
|
|
||||||
"powered_by_name": "Streamlit"
|
|
||||||
},
|
|
||||||
"sidebar": {
|
|
||||||
"menu": "Menu principal",
|
|
||||||
"navigation": "Navigation principale",
|
|
||||||
"theme": "Thème",
|
|
||||||
"theme_light": "Clair",
|
|
||||||
"theme_dark": "Sombre",
|
|
||||||
"theme_instructions_only": "Le changement de thème ne peut se faire que depuis l'onglet Instructions.",
|
|
||||||
"impact": "Impact environnemental",
|
|
||||||
"loading": "Chargement en cours…"
|
|
||||||
},
|
|
||||||
"auth": {
|
|
||||||
"title": "Authentification",
|
|
||||||
"username": "Identifiant_token",
|
|
||||||
"token": "Token d'accès personnel Gitea",
|
|
||||||
"login": "Se connecter",
|
|
||||||
"logout": "Se déconnecter",
|
|
||||||
"logged_as": "Connecté en tant que",
|
|
||||||
"error": "❌ Accès refusé.",
|
|
||||||
"gitea_error": "❌ Impossible de vérifier l'utilisateur auprès de Gitea.",
|
|
||||||
"success": "Déconnecté avec succès."
|
|
||||||
},
|
|
||||||
"navigation": {
|
|
||||||
"instructions": "Instructions",
|
|
||||||
"personnalisation": "Personnalisation",
|
|
||||||
"analyse": "Analyse",
|
|
||||||
"visualisations": "Visualisations",
|
|
||||||
"fiches": "Fiches"
|
|
||||||
},
|
|
||||||
"pages": {
|
|
||||||
"instructions": {
|
|
||||||
"title": "Instructions"
|
|
||||||
},
|
},
|
||||||
"personnalisation": {
|
"header": {
|
||||||
"title": "Personnalisation des produits finaux",
|
"title": "FabNum - Chaîne de fabrication du numérique",
|
||||||
"help": "Comment utiliser cet onglet ?",
|
"subtitle": "Parcours de l'écosystème et identification des vulnérabilités."
|
||||||
"help_content": [
|
|
||||||
"1. Cliquez sur « Ajouter un produit final » pour créer un nouveau produit",
|
|
||||||
"2. Donnez un nom à votre produit",
|
|
||||||
"3. Sélectionnez une opération d'assemblage appropriée (si pertinent)",
|
|
||||||
"4. Choisissez les composants qui constituent votre produit dans la liste proposée",
|
|
||||||
"5. Sauvegardez votre configuration pour une réutilisation future",
|
|
||||||
"6. Vous pourrez par la suite modifier ou supprimer vos produits personnalisés"
|
|
||||||
],
|
|
||||||
"add_new_product": "Ajouter un nouveau produit final",
|
|
||||||
"new_product_name": "Nom du nouveau produit (unique)",
|
|
||||||
"assembly_operation": "Opération d'assemblage (optionnelle)",
|
|
||||||
"none": "-- Aucune --",
|
|
||||||
"components_to_link": "Composants à lier",
|
|
||||||
"create_product": "Créer le produit",
|
|
||||||
"added": "ajouté",
|
|
||||||
"modify_product": "Modifier un produit final ajouté",
|
|
||||||
"products_to_modify": "Produits à modifier",
|
|
||||||
"delete": "Supprimer",
|
|
||||||
"linked_assembly_operation": "Opération d'assemblage liée",
|
|
||||||
"components_linked_to": "Composants liés à",
|
|
||||||
"update": "Mettre à jour",
|
|
||||||
"updated": "mis à jour",
|
|
||||||
"deleted": "supprimé",
|
|
||||||
"save_restore_config": "Sauvegarder ou restaurer la configuration",
|
|
||||||
"export_config": "Exporter configuration",
|
|
||||||
"download_json": "Télécharger (JSON)",
|
|
||||||
"import_config": "Importer une configuration JSON (max 100 Ko)",
|
|
||||||
"file_too_large": "Fichier trop volumineux (max 100 Ko).",
|
|
||||||
"no_products_found": "Aucun produit trouvé dans le fichier.",
|
|
||||||
"select_products_to_restore": "Sélection des produits à restaurer",
|
|
||||||
"products_to_restore": "Produits à restaurer",
|
|
||||||
"restore_selected": "Restaurer les éléments sélectionnés",
|
|
||||||
"config_restored": "Configuration partielle restaurée avec succès.",
|
|
||||||
"import_error": "Erreur d'import :"
|
|
||||||
},
|
},
|
||||||
"analyse": {
|
"footer": {
|
||||||
"title": "Analyse du graphe",
|
"copyright": "Fabnum © 2025",
|
||||||
"help": "Comment utiliser cet onglet ?",
|
"contact": "Contact",
|
||||||
"help_content": [
|
"license": "Licence",
|
||||||
"1. Sélectionnez le niveau de départ (produit final, composant ou minerai)",
|
"license_text": "CC BY-NC-ND",
|
||||||
"2. Choisissez le niveau d'arrivée souhaité",
|
"eco_note": "🌱 Calculs CO₂ via",
|
||||||
"3. Affinez votre sélection en spécifiant soit un ou des minerais à cibler spécifiquement ou des items précis à chaque niveau (optionnel)",
|
"eco_provider": "The Green Web Foundation",
|
||||||
"4. Définissez les critères d'analyse en sélectionnant les indices de vulnérabilité pertinents",
|
"powered_by": "🚀 Propulsé par",
|
||||||
"5. Choisissez le mode de combinaison des indices (ET/OU) selon votre besoin d'analyse",
|
"powered_by_name": "Streamlit"
|
||||||
"6. Explorez le graphique généré en utilisant les contrôles de zoom et de déplacement ; vous pouvez basculer en mode plein écran pour le graphe"
|
|
||||||
],
|
|
||||||
"selection_nodes": "Sélection des nœuds de départ et d'arrivée",
|
|
||||||
"select_level": "-- Sélectionner un niveau --",
|
|
||||||
"start_level": "Niveau de départ",
|
|
||||||
"end_level": "Niveau d'arrivée",
|
|
||||||
"select_minerals": "Sélectionner un ou plusieurs minerais",
|
|
||||||
"filter_by_minerals": "Filtrer par minerais (optionnel)",
|
|
||||||
"fine_selection": "Sélection fine des items",
|
|
||||||
"filter_start_nodes": "Filtrer par noeuds de départ (optionnel)",
|
|
||||||
"filter_end_nodes": "Filtrer par noeuds d'arrivée (optionnel)",
|
|
||||||
"vulnerability_filters": "Sélection des filtres pour identifier les vulnérabilités",
|
|
||||||
"filter_ics": "Filtrer les chemins contenant au moins minerai critique pour un composant (ICS > 66 %)",
|
|
||||||
"filter_ivc": "Filtrer les chemins contenant au moins un minerai critique par rapport à la concurrence sectorielle (IVC > 30)",
|
|
||||||
"filter_ihh": "Filtrer les chemins contenant au moins une opération critique par rapport à la concentration géographique ou industrielle (IHH pays ou acteurs > 25)",
|
|
||||||
"apply_ihh_filter": "Appliquer le filtre IHH sur :",
|
|
||||||
"countries": "Pays",
|
|
||||||
"actors": "Acteurs",
|
|
||||||
"filter_isg": "Filtrer les chemins contenant un pays instable (ISG ≥ 60)",
|
|
||||||
"filter_logic": "Logique de filtrage",
|
|
||||||
"or": "OU",
|
|
||||||
"and": "ET",
|
|
||||||
"run_analysis": "Lancer l'analyse",
|
|
||||||
"sankey": {
|
|
||||||
"no_paths": "Aucun chemin trouvé pour les critères spécifiés.",
|
|
||||||
"no_matching_paths": "Aucun chemin ne correspond aux critères.",
|
|
||||||
"filtered_hierarchy": "Hiérarchie filtrée par niveaux et noeuds",
|
|
||||||
"download_dot": "Télécharger le fichier DOT filtré",
|
|
||||||
"relation": "Relation"
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
"visualisations": {
|
"sidebar": {
|
||||||
"title": "Visualisations",
|
"menu": "Menu principal",
|
||||||
"help": "Comment utiliser cet onglet ?",
|
"navigation": "Navigation principale",
|
||||||
"help_content": [
|
"theme": "Thème",
|
||||||
"1. Explorez les graphiques présentant l'Indice de Herfindahl-Hirschmann (IHH)",
|
"theme_light": "Clair",
|
||||||
"2. Analysez sa relation avec la criticité moyenne des minerais ou leur Indice de Vulnérabilité Concurrentielle (IVC)",
|
"theme_dark": "Sombre",
|
||||||
"3. Zoomer dans les graphes pour mieux découvrir les informations",
|
"theme_instructions_only": "Le changement de thème ne peut se faire que depuis l'onglet Instructions.",
|
||||||
"",
|
"impact": "Impact environnemental",
|
||||||
"Il est important de se rappeler que l'IHH a deux seuils :",
|
"loading": "Chargement en cours…"
|
||||||
"* en-dessous de 15, la concentration est considérée comme étant faible",
|
|
||||||
"* au-dessus de 25, elle est considérée comme étant forte",
|
|
||||||
"",
|
|
||||||
"Ainsi plus le positionnement d'un point est en haut à droite des graphiques, plus les risques sont élevés.",
|
|
||||||
"Les graphiques présentent 2 droites horizontales et vetrticales pour matérialiser ces seuils."
|
|
||||||
],
|
|
||||||
"ihh_criticality": "Indice de Herfindahl-Hirschmann - IHH vs Criticité",
|
|
||||||
"ihh_criticality_desc": "La taille des points donne l'indication de la criticité de substituabilité du minerai.",
|
|
||||||
"ihh_ivc": "Indice de Herfindahl-Hirschmann - IHH vs IVC",
|
|
||||||
"ihh_ivc_desc": "La taille des points donne l'indication de la criticité concurrentielle du minerai.",
|
|
||||||
"launch": "Lancer",
|
|
||||||
"no_data": "Aucune donnée à visualiser.",
|
|
||||||
"categories": {
|
|
||||||
"assembly": "Assemblage",
|
|
||||||
"manufacturing": "Fabrication",
|
|
||||||
"processing": "Traitement",
|
|
||||||
"extraction": "Extraction"
|
|
||||||
},
|
|
||||||
"axis_titles": {
|
|
||||||
"ihh_countries": "IHH Pays (%)",
|
|
||||||
"ihh_actors": "IHH Acteurs (%)",
|
|
||||||
"ihh_extraction": "IHH Extraction (%)",
|
|
||||||
"ihh_reserves": "IHH Réserves (%)"
|
|
||||||
},
|
|
||||||
"chart_titles": {
|
|
||||||
"concentration_criticality": "Concentration et criticité – {0}",
|
|
||||||
"concentration_resources": "Concentration des ressources critiques vs vulnérabilité IVC"
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
"fiches": {
|
"auth": {
|
||||||
"title": "Découverte des fiches",
|
"title": "Authentification",
|
||||||
"help": "Comment utiliser cet onglet ?",
|
"username": "Identifiant_token",
|
||||||
"help_content": [
|
"token": "Token d'accès personnel Gitea",
|
||||||
"1. Parcourez la liste des fiches disponibles par catégorie",
|
"login": "Se connecter",
|
||||||
"2. Sélectionnez une fiche pour afficher son contenu complet",
|
"logout": "Se déconnecter",
|
||||||
"3. Consultez les données détaillées, graphiques et analyses supplémentaires",
|
"logged_as": "Connecté en tant que",
|
||||||
"4. Utilisez ces informations pour approfondir votre compréhension des vulnérabilités identifiées",
|
"error": "❌ Accès refusé.",
|
||||||
"",
|
"gitea_error": "❌ Impossible de vérifier l'utilisateur auprès de Gitea.",
|
||||||
"Les catégories sont les suivantes :",
|
"success": "Déconnecté avec succès."
|
||||||
"* Assemblage : opération d'assemblage des produits finaux à partir des composants",
|
},
|
||||||
"* Connexe : opérations diverses nécessaires pour fabriquer le numérique, mais n'entrant pas directement dans sa composition",
|
"navigation": {
|
||||||
"* Criticités : indices utilisés pour identifier et évaluer les vulnérabilités",
|
"instructions": "Instructions",
|
||||||
"* Fabrication : opération de fabrication des composants à partir de minerais",
|
"personnalisation": "Personnalisation",
|
||||||
"* Minerai : description et opérations d'extraction et de traitement des minerais"
|
"analyse": "Analyse",
|
||||||
],
|
"ia_nalyse": "IA'nalyse",
|
||||||
"no_files": "Aucune fiche disponible pour le moment.",
|
"plan_d_action": "Plan d'action",
|
||||||
"choose_category": "Choisissez une catégorie de fiches",
|
"visualisations": "Visualisations",
|
||||||
"select_folder": "-- Sélectionner un dossier --",
|
"fiches": "Fiches"
|
||||||
"choose_file": "Choisissez une fiche",
|
},
|
||||||
"select_file": "-- Sélectionner une fiche --",
|
"pages": {
|
||||||
"loading_error": "Erreur lors du chargement de la fiche :",
|
"instructions": {
|
||||||
"download_pdf": "Télécharger cette fiche en PDF",
|
"title": "Instructions"
|
||||||
"pdf_unavailable": "Le fichier PDF de cette fiche n'est pas disponible.",
|
|
||||||
"ticket_management": "Gestion des tickets pour cette fiche",
|
|
||||||
"tickets": {
|
|
||||||
"create_new": "Créer un nouveau ticket lié à cette fiche",
|
|
||||||
"model_load_error": "Impossible de charger le modèle de ticket.",
|
|
||||||
"contribution_type": "Type de contribution",
|
|
||||||
"specify": "Précisez",
|
|
||||||
"other": "Autre",
|
|
||||||
"concerned_card": "Fiche concernée",
|
|
||||||
"subject": "Sujet de la proposition",
|
|
||||||
"preview": "Prévisualiser le ticket",
|
|
||||||
"cancel": "Annuler",
|
|
||||||
"preview_title": "Prévisualisation du ticket",
|
|
||||||
"summary": "Résumé",
|
|
||||||
"title": "Titre",
|
|
||||||
"labels": "Labels",
|
|
||||||
"confirm": "Confirmer la création du ticket",
|
|
||||||
"created": "Ticket créé et formulaire vidé.",
|
|
||||||
"model_error": "Erreur chargement modèle :",
|
|
||||||
"no_linked_tickets": "Aucun ticket lié à cette fiche.",
|
|
||||||
"associated_tickets": "Tickets associés à cette fiche",
|
|
||||||
"moderation_notice": "ticket(s) en attente de modération ne sont pas affichés.",
|
|
||||||
"status": {
|
|
||||||
"awaiting": "En attente de traitement",
|
|
||||||
"in_progress": "En cours",
|
|
||||||
"completed": "Terminés",
|
|
||||||
"rejected": "Non retenus",
|
|
||||||
"others": "Autres"
|
|
||||||
},
|
},
|
||||||
"no_title": "Sans titre",
|
"personnalisation": {
|
||||||
"unknown": "inconnu",
|
"title": "Personnalisation des produits finaux",
|
||||||
"subject_label": "Sujet",
|
"help": "Comment utiliser cet onglet ?",
|
||||||
"no_labels": "aucun",
|
"help_content": [
|
||||||
"comments": "Commentaire(s) :",
|
"1. Cliquez sur « Ajouter un produit final » pour créer un nouveau produit",
|
||||||
"no_comments": "Aucun commentaire.",
|
"2. Donnez un nom à votre produit",
|
||||||
"comment_error": "Erreur lors de la récupération des commentaires :",
|
"3. Sélectionnez une opération d'assemblage appropriée (si pertinent)",
|
||||||
"opened_by": "Ouvert par",
|
"4. Choisissez les composants qui constituent votre produit dans la liste proposée",
|
||||||
"on_date": "le",
|
"5. Sauvegardez votre configuration pour une réutilisation future",
|
||||||
"updated": "MAJ"
|
"6. Vous pourrez par la suite modifier ou supprimer vos produits personnalisés"
|
||||||
}
|
],
|
||||||
|
"add_new_product": "Ajouter un nouveau produit final",
|
||||||
|
"new_product_name": "Nom du nouveau produit (unique)",
|
||||||
|
"assembly_operation": "Opération d'assemblage (optionnelle)",
|
||||||
|
"none": "-- Aucune --",
|
||||||
|
"components_to_link": "Composants à lier",
|
||||||
|
"create_product": "Créer le produit",
|
||||||
|
"added": "ajouté",
|
||||||
|
"modify_product": "Modifier un produit final ajouté",
|
||||||
|
"products_to_modify": "Produits à modifier",
|
||||||
|
"delete": "Supprimer",
|
||||||
|
"linked_assembly_operation": "Opération d'assemblage liée",
|
||||||
|
"components_linked_to": "Composants liés à",
|
||||||
|
"update": "Mettre à jour",
|
||||||
|
"updated": "mis à jour",
|
||||||
|
"deleted": "supprimé",
|
||||||
|
"save_restore_config": "Sauvegarder ou restaurer la configuration",
|
||||||
|
"export_config": "Exporter configuration",
|
||||||
|
"download_json": "Télécharger (JSON)",
|
||||||
|
"import_config": "Importer une configuration JSON (max 100 Ko)",
|
||||||
|
"file_too_large": "Fichier trop volumineux (max 100 Ko).",
|
||||||
|
"no_products_found": "Aucun produit trouvé dans le fichier.",
|
||||||
|
"select_products_to_restore": "Sélection des produits à restaurer",
|
||||||
|
"products_to_restore": "Produits à restaurer",
|
||||||
|
"restore_selected": "Restaurer les éléments sélectionnés",
|
||||||
|
"config_restored": "Configuration partielle restaurée avec succès.",
|
||||||
|
"import_error": "Erreur d'import :"
|
||||||
|
},
|
||||||
|
"analyse": {
|
||||||
|
"title": "Analyse du graphe",
|
||||||
|
"help": "Comment utiliser cet onglet ?",
|
||||||
|
"help_content": [
|
||||||
|
"1. Sélectionnez le niveau de départ (produit final, composant ou minerai)",
|
||||||
|
"2. Choisissez le niveau d'arrivée souhaité",
|
||||||
|
"3. Affinez votre sélection en spécifiant soit un ou des minerais à cibler spécifiquement ou des items précis à chaque niveau (optionnel)",
|
||||||
|
"4. Définissez les critères d'analyse en sélectionnant les indices de vulnérabilité pertinents",
|
||||||
|
"5. Choisissez le mode de combinaison des indices (ET/OU) selon votre besoin d'analyse",
|
||||||
|
"6. Explorez le graphique généré en utilisant les contrôles de zoom et de déplacement ; vous pouvez basculer en mode plein écran pour le graphe"
|
||||||
|
],
|
||||||
|
"selection_nodes": "Sélection des nœuds de départ et d'arrivée",
|
||||||
|
"select_level": "-- Sélectionner un niveau --",
|
||||||
|
"start_level": "Niveau de départ",
|
||||||
|
"end_level": "Niveau d'arrivée",
|
||||||
|
"select_minerals": "Sélectionner un ou plusieurs minerais",
|
||||||
|
"filter_by_minerals": "Filtrer par minerais (optionnel)",
|
||||||
|
"fine_selection": "Sélection fine des items",
|
||||||
|
"filter_start_nodes": "Filtrer par noeuds de départ (optionnel)",
|
||||||
|
"filter_end_nodes": "Filtrer par noeuds d'arrivée (optionnel)",
|
||||||
|
"vulnerability_filters": "Sélection des filtres pour identifier les vulnérabilités",
|
||||||
|
"filter_ics": "Filtrer les chemins contenant au moins minerai critique pour un composant (ICS > 66 %)",
|
||||||
|
"filter_ivc": "Filtrer les chemins contenant au moins un minerai critique par rapport à la concurrence sectorielle (IVC > 30)",
|
||||||
|
"filter_ihh": "Filtrer les chemins contenant au moins une opération critique par rapport à la concentration géographique ou industrielle (IHH pays ou acteurs > 25)",
|
||||||
|
"apply_ihh_filter": "Appliquer le filtre IHH sur :",
|
||||||
|
"countries": "Pays",
|
||||||
|
"actors": "Acteurs",
|
||||||
|
"filter_isg": "Filtrer les chemins contenant un pays instable (ISG ≥ 60)",
|
||||||
|
"filter_logic": "Logique de filtrage",
|
||||||
|
"or": "OU",
|
||||||
|
"and": "ET",
|
||||||
|
"run_analysis": "Lancer l'analyse",
|
||||||
|
"sankey": {
|
||||||
|
"no_paths": "Aucun chemin trouvé pour les critères spécifiés.",
|
||||||
|
"no_matching_paths": "Aucun chemin ne correspond aux critères.",
|
||||||
|
"filtered_hierarchy": "Hiérarchie filtrée par niveaux et noeuds",
|
||||||
|
"download_dot": "Télécharger le fichier DOT filtré",
|
||||||
|
"relation": "Relation"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ia_nalyse": {
|
||||||
|
"title": "Analyse du graphe par IA",
|
||||||
|
"help": "Comment utiliser cet onglet ?",
|
||||||
|
"help_content": [
|
||||||
|
"Le graphe intègre l'ensemble des niveaux, des produits finaux aux pays géographiques.\n",
|
||||||
|
"1. Vous pouvez sélectionner des minerais par lesquels passent les chemins.",
|
||||||
|
"2. Vous pouvez choisir des produits finaux pour faire une analyse adaptée à votre contexte.\n",
|
||||||
|
" Ces deux sélections sont optionnelles, mais fortement recommandées pour avoir une meilleure pertinence de l'analyse.",
|
||||||
|
"L'analyse se réalise à l'aide d'une IA privée, sur un serveur minimaliste. Le résultat n'est donc pas immédiat (ordre de grandeur : 30 minutes) et vous serez informé de l'avancement."
|
||||||
|
],
|
||||||
|
"select_minerals": "Sélectionner un ou plusieurs minerais",
|
||||||
|
"filter_by_minerals": "Filtrer par minerais (optionnel, mais recommandé)",
|
||||||
|
"fine_selection": "Sélection des produits finaux",
|
||||||
|
"filter_start_nodes": "Filtrer par noeuds de départ (optionnel, mais recommandé)",
|
||||||
|
"run_analysis": "Lancer l'analyse",
|
||||||
|
"confirm_download": "Confirmer le téléchargement",
|
||||||
|
"submit_request": "Soumettre votre demande",
|
||||||
|
"empty_graph": "Le graphe est vide. Veuillez faire une autre sélection."
|
||||||
|
},
|
||||||
|
"plan_d_action": {
|
||||||
|
"title": "Analyse du graphe pour action",
|
||||||
|
"help": "Comment utiliser cet onglet ?",
|
||||||
|
"help_content": [
|
||||||
|
"Le graphe intègre l'ensemble des niveaux, des produits finaux aux pays géographiques.\n",
|
||||||
|
"1. Vous pouvez sélectionner des minerais par lesquels passent les chemins.",
|
||||||
|
"2. Vous pouvez choisir des produits finaux pour faire une analyse adaptée à votre contexte.\n",
|
||||||
|
" Ces deux sélections sont optionnelles, mais fortement recommandées pour avoir une meilleure pertinence de l'analyse.",
|
||||||
|
"Les préconisations d'actions et de suivi d'indicateurs sont génériques. Elles doivent donc être adaptées au contexte.",
|
||||||
|
"Les actions ou les indicateurs proposés dépendent du type d'organisation concernée et peuvent être appliquées directement ou exigées des fournisseurs de numérique."
|
||||||
|
],
|
||||||
|
"select_minerals": "Sélectionner un ou plusieurs minerais",
|
||||||
|
"filter_by_minerals": "Filtrer par minerais (optionnel, mais recommandé)",
|
||||||
|
"fine_selection": "Sélection des produits finaux",
|
||||||
|
"filter_start_nodes": "Filtrer par noeuds de départ (optionnel, mais recommandé)",
|
||||||
|
"run_analysis": "Lancer l'analyse",
|
||||||
|
"confirm_download": "Confirmer le téléchargement",
|
||||||
|
"submit_request": "Soumettre votre demande",
|
||||||
|
"empty_graph": "Le graphe est vide. Veuillez faire une autre sélection."
|
||||||
|
},
|
||||||
|
"visualisations": {
|
||||||
|
"title": "Visualisations",
|
||||||
|
"help": "Comment utiliser cet onglet ?",
|
||||||
|
"help_content": [
|
||||||
|
"1. Explorez les graphiques présentant l'Indice de Herfindahl-Hirschmann (IHH)",
|
||||||
|
"2. Analysez sa relation avec la criticité moyenne des minerais ou leur Indice de Vulnérabilité Concurrentielle (IVC)",
|
||||||
|
"3. Zoomer dans les graphes pour mieux découvrir les informations",
|
||||||
|
"",
|
||||||
|
"Il est important de se rappeler que l'IHH a deux seuils :",
|
||||||
|
"* en-dessous de 15, la concentration est considérée comme étant faible",
|
||||||
|
"* au-dessus de 25, elle est considérée comme étant forte",
|
||||||
|
"",
|
||||||
|
"Ainsi plus le positionnement d'un point est en haut à droite des graphiques, plus les risques sont élevés.",
|
||||||
|
"Les graphiques présentent 2 droites horizontales et vetrticales pour matérialiser ces seuils."
|
||||||
|
],
|
||||||
|
"ihh_criticality": "Indice de Herfindahl-Hirschmann - IHH vs Criticité",
|
||||||
|
"ihh_criticality_desc": "La taille des points donne l'indication de la criticité de substituabilité du minerai.",
|
||||||
|
"ihh_ivc": "Indice de Herfindahl-Hirschmann - IHH vs IVC",
|
||||||
|
"ihh_ivc_desc": "La taille des points donne l'indication de la criticité concurrentielle du minerai.",
|
||||||
|
"launch": "Lancer",
|
||||||
|
"no_data": "Aucune donnée à visualiser.",
|
||||||
|
"categories": {
|
||||||
|
"assembly": "Assemblage",
|
||||||
|
"manufacturing": "Fabrication",
|
||||||
|
"processing": "Traitement",
|
||||||
|
"extraction": "Extraction"
|
||||||
|
},
|
||||||
|
"axis_titles": {
|
||||||
|
"ihh_countries": "IHH Pays (%)",
|
||||||
|
"ihh_actors": "IHH Acteurs (%)",
|
||||||
|
"ihh_extraction": "IHH Extraction (%)",
|
||||||
|
"ihh_reserves": "IHH Réserves (%)"
|
||||||
|
},
|
||||||
|
"chart_titles": {
|
||||||
|
"concentration_criticality": "Concentration et criticité – {0}",
|
||||||
|
"concentration_resources": "Concentration des ressources critiques vs vulnérabilité IVC"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"fiches": {
|
||||||
|
"title": "Découverte des fiches",
|
||||||
|
"help": "Comment utiliser cet onglet ?",
|
||||||
|
"help_content": [
|
||||||
|
"1. Parcourez la liste des fiches disponibles par catégorie",
|
||||||
|
"2. Sélectionnez une fiche pour afficher son contenu complet",
|
||||||
|
"3. Consultez les données détaillées, graphiques et analyses supplémentaires",
|
||||||
|
"4. Utilisez ces informations pour approfondir votre compréhension des vulnérabilités identifiées",
|
||||||
|
"",
|
||||||
|
"Les catégories sont les suivantes :",
|
||||||
|
"* Assemblage : opération d'assemblage des produits finaux à partir des composants",
|
||||||
|
"* Connexe : opérations diverses nécessaires pour fabriquer le numérique, mais n'entrant pas directement dans sa composition",
|
||||||
|
"* Criticités : indices utilisés pour identifier et évaluer les vulnérabilités",
|
||||||
|
"* Fabrication : opération de fabrication des composants à partir de minerais",
|
||||||
|
"* Minerai : description et opérations d'extraction et de traitement des minerais"
|
||||||
|
],
|
||||||
|
"no_files": "Aucune fiche disponible pour le moment.",
|
||||||
|
"choose_category": "Choisissez une catégorie de fiches",
|
||||||
|
"select_folder": "-- Sélectionner un dossier --",
|
||||||
|
"choose_file": "Choisissez une fiche",
|
||||||
|
"select_file": "-- Sélectionner une fiche --",
|
||||||
|
"loading_error": "Erreur lors du chargement de la fiche :",
|
||||||
|
"download_pdf": "Télécharger cette fiche en PDF",
|
||||||
|
"pdf_unavailable": "Le fichier PDF de cette fiche n'est pas disponible.",
|
||||||
|
"ticket_management": "Gestion des tickets pour cette fiche",
|
||||||
|
"tickets": {
|
||||||
|
"create_new": "Créer un nouveau ticket lié à cette fiche",
|
||||||
|
"model_load_error": "Impossible de charger le modèle de ticket.",
|
||||||
|
"contribution_type": "Type de contribution",
|
||||||
|
"specify": "Précisez",
|
||||||
|
"other": "Autre",
|
||||||
|
"concerned_card": "Fiche concernée",
|
||||||
|
"subject": "Sujet de la proposition",
|
||||||
|
"preview": "Prévisualiser le ticket",
|
||||||
|
"cancel": "Annuler",
|
||||||
|
"preview_title": "Prévisualisation du ticket",
|
||||||
|
"summary": "Résumé",
|
||||||
|
"title": "Titre",
|
||||||
|
"labels": "Labels",
|
||||||
|
"confirm": "Confirmer la création du ticket",
|
||||||
|
"created": "Ticket créé et formulaire vidé.",
|
||||||
|
"model_error": "Erreur chargement modèle :",
|
||||||
|
"no_linked_tickets": "Aucun ticket lié à cette fiche.",
|
||||||
|
"associated_tickets": "Tickets associés à cette fiche",
|
||||||
|
"moderation_notice": "ticket(s) en attente de modération ne sont pas affichés.",
|
||||||
|
"status": {
|
||||||
|
"awaiting": "En attente de traitement",
|
||||||
|
"in_progress": "En cours",
|
||||||
|
"completed": "Terminés",
|
||||||
|
"rejected": "Non retenus",
|
||||||
|
"others": "Autres"
|
||||||
|
},
|
||||||
|
"no_title": "Sans titre",
|
||||||
|
"unknown": "inconnu",
|
||||||
|
"subject_label": "Sujet",
|
||||||
|
"no_labels": "aucun",
|
||||||
|
"comments": "Commentaire(s) :",
|
||||||
|
"no_comments": "Aucun commentaire.",
|
||||||
|
"comment_error": "Erreur lors de la récupération des commentaires :",
|
||||||
|
"opened_by": "Ouvert par",
|
||||||
|
"on_date": "le",
|
||||||
|
"updated": "MAJ",
|
||||||
|
"continue": "Continuer",
|
||||||
|
"created_success": "Ticket créé et placé en modération",
|
||||||
|
"created_error": "Échec de la création du ticket. Veuillez réessayer plus tard",
|
||||||
|
"see_ticket": "Voir le ticket"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"node_levels": {
|
||||||
|
"0": "Produit final",
|
||||||
|
"1": "Composant",
|
||||||
|
"2": "Minerai",
|
||||||
|
"10": "Opération",
|
||||||
|
"11": "Pays d'opération",
|
||||||
|
"12": "Acteur d'opération",
|
||||||
|
"99": "Pays géographique"
|
||||||
|
},
|
||||||
|
"errors": {
|
||||||
|
"log_read_error": "Erreur lecture log:",
|
||||||
|
"graph_preview_error": "Erreur de prévisualisation du graphe :",
|
||||||
|
"graph_creation_error": "Erreur lors de la création du graphique :",
|
||||||
|
"ihh_criticality_error": "Erreur dans la visualisation IHH vs Criticité :",
|
||||||
|
"ihh_ivc_error": "Erreur dans la visualisation IHH vs IVC :",
|
||||||
|
"comment_fetch_error": "Erreur lors de la récupération des commentaires :",
|
||||||
|
"template_load_error": "Erreur chargement modèle :",
|
||||||
|
"import_error": "Erreur d'import :"
|
||||||
|
},
|
||||||
|
"buttons": {
|
||||||
|
"download": "Télécharger",
|
||||||
|
"run": "Lancer",
|
||||||
|
"save": "Enregistrer",
|
||||||
|
"cancel": "Annuler",
|
||||||
|
"confirm": "Confirmer",
|
||||||
|
"filter": "Filtrer",
|
||||||
|
"search": "Rechercher",
|
||||||
|
"create": "Créer",
|
||||||
|
"update": "Mettre à jour",
|
||||||
|
"delete": "Supprimer",
|
||||||
|
"preview": "Prévisualiser",
|
||||||
|
"export": "Exporter",
|
||||||
|
"import": "Importer",
|
||||||
|
"restore": "Restaurer",
|
||||||
|
"refresh": "Rafraîchir",
|
||||||
|
"browse_files": "Parcourir les fichiers"
|
||||||
|
},
|
||||||
|
"ui": {
|
||||||
|
"file_uploader": {
|
||||||
|
"drag_drop_here": "Glissez-déposez votre fichier ici",
|
||||||
|
"size_limit": "Limite 100 Ko par fichier • JSON"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"batch": {
|
||||||
|
"in_queue": "En attente",
|
||||||
|
"in_progress": "Analyse en cours",
|
||||||
|
"failure": "Échec",
|
||||||
|
"unknwon_error": "erreur inconnue",
|
||||||
|
"no_task": "Aucune tâche en attente ou en cours",
|
||||||
|
"complete": "Analyse terminée. Télécharger le résultat au format zip, qui contient le rapport détaillé et l'analyse.",
|
||||||
|
"step": "Étape"
|
||||||
}
|
}
|
||||||
},
|
|
||||||
"node_levels": {
|
|
||||||
"0": "Produit final",
|
|
||||||
"1": "Composant",
|
|
||||||
"2": "Minerai",
|
|
||||||
"10": "Opération",
|
|
||||||
"11": "Pays d'opération",
|
|
||||||
"12": "Acteur d'opération",
|
|
||||||
"99": "Pays géographique"
|
|
||||||
},
|
|
||||||
"errors": {
|
|
||||||
"log_read_error": "Erreur lecture log:",
|
|
||||||
"graph_preview_error": "Erreur de prévisualisation du graphe :",
|
|
||||||
"graph_creation_error": "Erreur lors de la création du graphique :",
|
|
||||||
"ihh_criticality_error": "Erreur dans la visualisation IHH vs Criticité :",
|
|
||||||
"ihh_ivc_error": "Erreur dans la visualisation IHH vs IVC :",
|
|
||||||
"comment_fetch_error": "Erreur lors de la récupération des commentaires :",
|
|
||||||
"template_load_error": "Erreur chargement modèle :",
|
|
||||||
"import_error": "Erreur d'import :"
|
|
||||||
},
|
|
||||||
"buttons": {
|
|
||||||
"download": "Télécharger",
|
|
||||||
"run": "Lancer",
|
|
||||||
"save": "Enregistrer",
|
|
||||||
"cancel": "Annuler",
|
|
||||||
"confirm": "Confirmer",
|
|
||||||
"filter": "Filtrer",
|
|
||||||
"search": "Rechercher",
|
|
||||||
"create": "Créer",
|
|
||||||
"update": "Mettre à jour",
|
|
||||||
"delete": "Supprimer",
|
|
||||||
"preview": "Prévisualiser",
|
|
||||||
"export": "Exporter",
|
|
||||||
"import": "Importer",
|
|
||||||
"restore": "Restaurer",
|
|
||||||
"browse_files": "Parcourir les fichiers"
|
|
||||||
},
|
|
||||||
"ui": {
|
|
||||||
"file_uploader": {
|
|
||||||
"drag_drop_here": "Glissez-déposez votre fichier ici",
|
|
||||||
"size_limit": "Limite 100 Ko par fichier • JSON"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@ -4,33 +4,33 @@
|
|||||||
1. Reset et base
|
1. Reset et base
|
||||||
========================================== */
|
========================================== */
|
||||||
.stAppHeader {
|
.stAppHeader {
|
||||||
visibility: hidden;
|
visibility: hidden;
|
||||||
}
|
}
|
||||||
|
|
||||||
body,
|
body,
|
||||||
html {
|
html {
|
||||||
font-family:
|
font-family:
|
||||||
-apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, Helvetica, Arial,
|
-apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, Helvetica, Arial,
|
||||||
sans-serif;
|
sans-serif;
|
||||||
}
|
}
|
||||||
|
|
||||||
body,
|
body,
|
||||||
.stApp,
|
.stApp,
|
||||||
.block-container {
|
.block-container {
|
||||||
background-color: var(--bg-color) !important;
|
background-color: var(--bg-color) !important;
|
||||||
color: var(--text-color) !important;
|
color: var(--text-color) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ==========================================
|
/* ==========================================
|
||||||
2. Layout et containers
|
2. Layout et containers
|
||||||
========================================== */
|
========================================== */
|
||||||
.block-container {
|
.block-container {
|
||||||
max-width: 1024px !important;
|
max-width: 1024px !important;
|
||||||
padding: 0 1rem 10rem;
|
padding: 0 1rem 10rem;
|
||||||
}
|
}
|
||||||
|
|
||||||
.stVerticalBlock {
|
.stVerticalBlock {
|
||||||
gap: 0.5rem !important;
|
gap: 0.5rem !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ==========================================
|
/* ==========================================
|
||||||
@ -42,97 +42,97 @@ body,
|
|||||||
.stDownloadButton > button,
|
.stDownloadButton > button,
|
||||||
.stFormSubmitButton > button,
|
.stFormSubmitButton > button,
|
||||||
.stSlider > div > div {
|
.stSlider > div > div {
|
||||||
background-color: darkgreen !important;
|
background-color: darkgreen !important;
|
||||||
color: white !important;
|
color: white !important;
|
||||||
border: 1px solid grey;
|
border: 1px solid grey;
|
||||||
}
|
}
|
||||||
|
|
||||||
.st-key-FormSubmitter-auth_form-Se-connecter {
|
.st-key-FormSubmitter-auth_form-Se-connecter {
|
||||||
margin-left: auto;
|
margin-left: auto;
|
||||||
margin-right: auto;
|
margin-right: auto;
|
||||||
}
|
}
|
||||||
|
|
||||||
section:not([data-testid="stSidebar"])
|
section:not([data-testid="stSidebar"])
|
||||||
button[data-testid="stBaseButton-primary"],
|
button[data-testid="stBaseButton-primary"],
|
||||||
section:not([data-testid="stSidebar"])
|
section:not([data-testid="stSidebar"])
|
||||||
button[data-testid="stBaseButton-secondary"] {
|
button[data-testid="stBaseButton-secondary"] {
|
||||||
color: white !important;
|
color: white !important;
|
||||||
background: darkgreen !important;
|
background: darkgreen !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
section:not([data-testid="stSidebar"])
|
section:not([data-testid="stSidebar"])
|
||||||
button[data-testid="stBaseButton-primary"]
|
button[data-testid="stBaseButton-primary"]
|
||||||
p,
|
p,
|
||||||
section:not([data-testid="stSidebar"])
|
section:not([data-testid="stSidebar"])
|
||||||
button[data-testid="stBaseButton-secondary"]
|
button[data-testid="stBaseButton-secondary"]
|
||||||
p {
|
p {
|
||||||
color: white !important;
|
color: white !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
.bouton-fictif {
|
.bouton-fictif {
|
||||||
display: inline-flex;
|
display: inline-flex;
|
||||||
-moz-box-align: center;
|
-moz-box-align: center;
|
||||||
align-items: center;
|
align-items: center;
|
||||||
-moz-box-pack: center;
|
-moz-box-pack: center;
|
||||||
justify-content: center;
|
justify-content: center;
|
||||||
padding: 0.25rem 0.75rem;
|
padding: 0.25rem 0.75rem;
|
||||||
border-radius: 0.5rem;
|
border-radius: 0.5rem;
|
||||||
min-height: 2.5rem;
|
min-height: 2.5rem;
|
||||||
margin-bottom: 20px;
|
margin-bottom: 20px;
|
||||||
line-height: 1;
|
line-height: 1;
|
||||||
text-transform: none;
|
text-transform: none;
|
||||||
font-size: x-large;
|
font-size: x-large;
|
||||||
font-family: inherit;
|
font-family: inherit;
|
||||||
user-select: none;
|
user-select: none;
|
||||||
border: 1px solid rgba(49, 51, 63, 0.2);
|
border: 1px solid rgba(49, 51, 63, 0.2);
|
||||||
background-color: darkgrey !important;
|
background-color: darkgrey !important;
|
||||||
color: darkgreen !important;
|
color: darkgreen !important;
|
||||||
font-weight: bold !important;
|
font-weight: bold !important;
|
||||||
width: 100%;
|
width: 100%;
|
||||||
letter-spacing: 0.2em;
|
letter-spacing: 0.2em;
|
||||||
}
|
}
|
||||||
|
|
||||||
button[data-testid="stBaseButton-headerNoPadding"] svg {
|
button[data-testid="stBaseButton-headerNoPadding"] svg {
|
||||||
fill: var(--text-color) !important;
|
fill: var(--text-color) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* --- 3.2 Onglets et radiogroup --- */
|
/* --- 3.2 Onglets et radiogroup --- */
|
||||||
div[role="radiogroup"] > label {
|
div[role="radiogroup"] > label {
|
||||||
padding: 0.5em 1em;
|
padding: 0.5em 1em;
|
||||||
border-radius: 0.4em;
|
border-radius: 0.4em;
|
||||||
margin-right: 0.5em;
|
margin-right: 0.5em;
|
||||||
cursor: pointer;
|
cursor: pointer;
|
||||||
border: 1px solid #fff;
|
border: 1px solid #fff;
|
||||||
}
|
}
|
||||||
|
|
||||||
div[role="radiogroup"] > label[data-selected="true"] {
|
div[role="radiogroup"] > label[data-selected="true"] {
|
||||||
font-weight: bold;
|
font-weight: bold;
|
||||||
border: 2px solid #145a1a;
|
border: 2px solid #145a1a;
|
||||||
}
|
}
|
||||||
|
|
||||||
section:not([data-testid="stSidebar"]) div[role="radiogroup"] > label p {
|
section:not([data-testid="stSidebar"]) div[role="radiogroup"] > label p {
|
||||||
background-color: var(--radio-bg) !important;
|
background-color: var(--radio-bg) !important;
|
||||||
color: var(--radio-text) !important;
|
color: var(--radio-text) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
section:not([data-testid="stSidebar"])
|
section:not([data-testid="stSidebar"])
|
||||||
div[role="radiogroup"]
|
div[role="radiogroup"]
|
||||||
> label[data-selected="true"] {
|
> label[data-selected="true"] {
|
||||||
background-color: var(--radio-selected-bg) !important;
|
background-color: var(--radio-selected-bg) !important;
|
||||||
color: var(--radio-selected-text) !important;
|
color: var(--radio-selected-text) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* --- 3.3 Champs de formulaire --- */
|
/* --- 3.3 Champs de formulaire --- */
|
||||||
div[data-baseweb="select"],
|
div[data-baseweb="select"],
|
||||||
section:not([data-testid="stSidebar"]) div[data-baseweb="base-input"],
|
section:not([data-testid="stSidebar"]) div[data-baseweb="base-input"],
|
||||||
section[data-testid="stFileUploaderDropzone"] {
|
section[data-testid="stFileUploaderDropzone"] {
|
||||||
border: 1px solid var(--input-border) !important;
|
border: 1px solid var(--input-border) !important;
|
||||||
border-radius: 5px;
|
border-radius: 5px;
|
||||||
padding: 4px;
|
padding: 4px;
|
||||||
}
|
}
|
||||||
|
|
||||||
small {
|
small {
|
||||||
display: none;
|
display: none;
|
||||||
}
|
}
|
||||||
|
|
||||||
section:not([data-testid="stSidebar"]) div[data-testid="stSelectbox"] p,
|
section:not([data-testid="stSidebar"]) div[data-testid="stSelectbox"] p,
|
||||||
@ -142,7 +142,7 @@ section:not([data-testid="stSidebar"]) div[data-testid="stCheckbox"] p,
|
|||||||
section:not([data-testid="stSidebar"]) div[data-testid="stTextInput"] p,
|
section:not([data-testid="stSidebar"]) div[data-testid="stTextInput"] p,
|
||||||
section:not([data-testid="stSidebar"]) div[data-testid="stTextArea"] p,
|
section:not([data-testid="stSidebar"]) div[data-testid="stTextArea"] p,
|
||||||
section:not([data-testid="stSidebar"]) div[data-testid="stAlertContentInfo"] p {
|
section:not([data-testid="stSidebar"]) div[data-testid="stAlertContentInfo"] p {
|
||||||
color: var(--text-color) !important;
|
color: var(--text-color) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ==========================================
|
/* ==========================================
|
||||||
@ -151,98 +151,98 @@ section:not([data-testid="stSidebar"]) div[data-testid="stAlertContentInfo"] p {
|
|||||||
|
|
||||||
/* --- 4.1 Header --- */
|
/* --- 4.1 Header --- */
|
||||||
.wide-header {
|
.wide-header {
|
||||||
width: 100vw;
|
width: 100vw;
|
||||||
margin-left: calc(-50vw + 50%);
|
margin-left: calc(-50vw + 50%);
|
||||||
box-shadow: 0 2px 4px rgba(0, 0, 0, 0.1);
|
box-shadow: 0 2px 4px rgba(0, 0, 0, 0.1);
|
||||||
border-bottom: 1px solid #ddd;
|
border-bottom: 1px solid #ddd;
|
||||||
text-align: center;
|
text-align: center;
|
||||||
padding-top: 1rem;
|
padding-top: 1rem;
|
||||||
margin-top: -1.25em;
|
margin-top: -1.25em;
|
||||||
background-color: var(--header-bg);
|
background-color: var(--header-bg);
|
||||||
}
|
}
|
||||||
|
|
||||||
.titre-header {
|
.titre-header {
|
||||||
font-size: 2rem !important;
|
font-size: 2rem !important;
|
||||||
font-weight: bolder !important;
|
font-weight: bolder !important;
|
||||||
color: var(--header-title);
|
color: var(--header-title);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* --- 4.2 Footer --- */
|
/* --- 4.2 Footer --- */
|
||||||
.wide-footer {
|
.wide-footer {
|
||||||
width: 100vw;
|
width: 100vw;
|
||||||
margin-left: calc(-50vw + 50%);
|
margin-left: calc(-50vw + 50%);
|
||||||
margin-top: 3rem;
|
margin-top: 3rem;
|
||||||
box-shadow: 0 2px 4px rgba(0, 0, 0, 0.1);
|
box-shadow: 0 2px 4px rgba(0, 0, 0, 0.1);
|
||||||
border-top: 1px solid #ddd;
|
border-top: 1px solid #ddd;
|
||||||
text-align: center;
|
text-align: center;
|
||||||
padding-top: 1rem;
|
padding-top: 1rem;
|
||||||
background-color: var(--footer-bg);
|
background-color: var(--footer-bg);
|
||||||
}
|
}
|
||||||
|
|
||||||
.info-footer {
|
.info-footer {
|
||||||
font-size: 1rem !important;
|
font-size: 1rem !important;
|
||||||
font-weight: 800;
|
font-weight: 800;
|
||||||
color: var(--footer-text);
|
color: var(--footer-text);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ==========================================
|
/* ==========================================
|
||||||
5. Sidebar
|
5. Sidebar
|
||||||
========================================== */
|
========================================== */
|
||||||
section[data-testid="stSidebar"] {
|
section[data-testid="stSidebar"] {
|
||||||
background-color: #ccc !important;
|
background-color: #ccc !important;
|
||||||
color: #111 !important;
|
color: #111 !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
section[data-testid="stSidebar"] .stButton > button {
|
section[data-testid="stSidebar"] .stButton > button {
|
||||||
background-color: darkgreen !important;
|
background-color: darkgreen !important;
|
||||||
color: white !important;
|
color: white !important;
|
||||||
font-weight: bold !important;
|
font-weight: bold !important;
|
||||||
border: 1px solid #ccc !important;
|
border: 1px solid #ccc !important;
|
||||||
width: 100%;
|
width: 100%;
|
||||||
}
|
}
|
||||||
|
|
||||||
section[data-testid="stSidebar"] .decorative-heading {
|
section[data-testid="stSidebar"] .decorative-heading {
|
||||||
font-size: 1.25rem;
|
font-size: 1.25rem;
|
||||||
font-weight: bold;
|
font-weight: bold;
|
||||||
margin-bottom: 0.5rem;
|
margin-bottom: 0.5rem;
|
||||||
text-align: center;
|
text-align: center;
|
||||||
color: #145a1a;
|
color: #145a1a;
|
||||||
}
|
}
|
||||||
|
|
||||||
section[data-testid="stSidebar"] div[role="radiogroup"] {
|
section[data-testid="stSidebar"] div[role="radiogroup"] {
|
||||||
justify-content: center !important;
|
justify-content: center !important;
|
||||||
display: flex !important;
|
display: flex !important;
|
||||||
gap: 1rem;
|
gap: 1rem;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ==========================================
|
/* ==========================================
|
||||||
6. Tables
|
6. Tables
|
||||||
========================================== */
|
========================================== */
|
||||||
table {
|
table {
|
||||||
border: 1px solid var(--table-border) !important;
|
border: 1px solid var(--table-border) !important;
|
||||||
border-collapse: collapse;
|
border-collapse: collapse;
|
||||||
width: 100%;
|
width: 100%;
|
||||||
margin-bottom: 1.5em;
|
margin-bottom: 1.5em;
|
||||||
}
|
}
|
||||||
|
|
||||||
th,
|
th,
|
||||||
td {
|
td {
|
||||||
border: 1px solid var(--table-border) !important;
|
border: 1px solid var(--table-border) !important;
|
||||||
padding: 8px;
|
padding: 8px;
|
||||||
text-align: left;
|
text-align: left;
|
||||||
}
|
}
|
||||||
|
|
||||||
caption {
|
caption {
|
||||||
caption-side: top;
|
caption-side: top;
|
||||||
font-weight: bold;
|
font-weight: bold;
|
||||||
padding: 0.5em;
|
padding: 0.5em;
|
||||||
text-align: left;
|
text-align: left;
|
||||||
caption-side: bottom;
|
caption-side: bottom;
|
||||||
text-align: center;
|
text-align: center;
|
||||||
}
|
}
|
||||||
|
|
||||||
table[role="table"] th[scope="col"] {
|
table[role="table"] th[scope="col"] {
|
||||||
background-color: var(--background-color);
|
background-color: var(--background-color);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ==========================================
|
/* ==========================================
|
||||||
@ -254,94 +254,99 @@ table[role="table"] th[scope="col"] {
|
|||||||
|
|
||||||
/* --- 7.2 Graphiques --- */
|
/* --- 7.2 Graphiques --- */
|
||||||
.stPlotlyChart text {
|
.stPlotlyChart text {
|
||||||
fill: var(--plot-text) !important;
|
fill: var(--plot-text) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
.stPlotlyChart text {
|
.stPlotlyChart text {
|
||||||
fill: black !important;
|
fill: black !important;
|
||||||
text-shadow: none !important;
|
text-shadow: none !important;
|
||||||
font-weight: bold !important;
|
font-weight: bold !important;
|
||||||
font-size: 14px !important;
|
font-size: 14px !important;
|
||||||
font-family: Verdana, sans-serif !important;
|
font-family: Verdana, sans-serif !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Cache complètement la section d'actions Vega */
|
/* Cache complètement la section d'actions Vega */
|
||||||
.vega-actions {
|
.vega-actions {
|
||||||
display: none !important;
|
display: none !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Et aussi le <details> parent, s'il faut tout masquer */
|
/* Et aussi le <details> parent, s'il faut tout masquer */
|
||||||
details[title="Click to view actions"] {
|
details[title="Click to view actions"] {
|
||||||
display: none !important;
|
display: none !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* --- 7.3 Détails et paragraphes --- */
|
/* --- 7.3 Détails et paragraphes --- */
|
||||||
details {
|
details {
|
||||||
border: 1px solid #ccc;
|
border: 1px solid #ccc;
|
||||||
border-radius: 6px;
|
border-radius: 6px;
|
||||||
padding: 0.5em;
|
padding: 0.5em;
|
||||||
margin-bottom: 0.5em;
|
margin-bottom: 0.5em;
|
||||||
background-color: var(--background-color);
|
background-color: var(--background-color);
|
||||||
border-color: var(--details-border) !important;
|
border-color: var(--details-border) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
section:not([data-testid="stSidebar"])
|
section:not([data-testid="stSidebar"])
|
||||||
div:not[data-testid="stElementContainer"]
|
div:not[data-testid="stElementContainer"]
|
||||||
p:not(#Authentification):not(#Theme) {
|
p:not(#Authentification):not(#Theme) {
|
||||||
color: var(--paragraph-color) !important;
|
color: var(--paragraph-color) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
section:not([data-testid="stSidebar"]) hr {
|
section:not([data-testid="stSidebar"]) hr {
|
||||||
background-color: var(--hr-color) !important;
|
background-color: var(--hr-color) !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* --- 7.4 Conteneurs de commentaires et tickets --- */
|
/* --- 7.4 Conteneurs de commentaires et tickets --- */
|
||||||
.conteneur_commentaire,
|
.conteneur_commentaire,
|
||||||
.conteneur_ticket {
|
.conteneur_ticket {
|
||||||
background: var(--background-color);
|
background: var(--background-color);
|
||||||
padding: 1em;
|
padding: 1em;
|
||||||
border-radius: 8px;
|
border-radius: 8px;
|
||||||
margin-bottom: 1em;
|
margin-bottom: 1em;
|
||||||
border: 1px solid #ccc;
|
border: 1px solid #ccc;
|
||||||
}
|
}
|
||||||
|
|
||||||
.commentaire_auteur,
|
.commentaire_auteur,
|
||||||
.ticket_auteur {
|
.ticket_auteur {
|
||||||
color: var(--text-color) !important;
|
color: var(--text-color) !important;
|
||||||
margin: 0;
|
margin: 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
.commentaire_contenu,
|
.commentaire_contenu,
|
||||||
.ticket_contenu {
|
.ticket_contenu {
|
||||||
color: var(--text-color) !important;
|
color: var(--text-color) !important;
|
||||||
margin: 0.5rem 0 0;
|
margin: 0.5rem 0 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* --- 7.5 Blocs mathématiques --- */
|
/* --- 7.5 Blocs mathématiques --- */
|
||||||
.math-block {
|
.math-block {
|
||||||
display: block;
|
display: block;
|
||||||
text-align: center;
|
text-align: center;
|
||||||
margin: 1em 0;
|
margin: 1em 0;
|
||||||
border: 1px solid var(--math-block-border);
|
border: 1px solid var(--math-block-border);
|
||||||
border-radius: 10px;
|
border-radius: 10px;
|
||||||
background: var(--math-block-bg);
|
background: var(--math-block-bg);
|
||||||
font-size: x-large;
|
font-size: x-large;
|
||||||
color: var(--text-color);
|
color: var(--text-color);
|
||||||
}
|
}
|
||||||
|
|
||||||
.math-block math {
|
.math-block math {
|
||||||
display: inline-block;
|
display: inline-block;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* ==========================================
|
/* ==========================================
|
||||||
8. Éléments spécifiques
|
8. Éléments spécifiques
|
||||||
========================================== */
|
========================================== */
|
||||||
div.stElementContainer.element-container.st-key-nom_utilisateur {
|
div.stElementContainer.element-container.st-key-nom_utilisateur {
|
||||||
display: none !important;
|
display: none !important;
|
||||||
}
|
}
|
||||||
|
|
||||||
.st-key-telecharger_fiche_pdf {
|
.st-key-telecharger_fiche_pdf {
|
||||||
margin-left: auto;
|
margin-left: auto;
|
||||||
margin-right: auto;
|
margin-right: auto;
|
||||||
margin-top: 1rem;
|
margin-top: 1rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.details_introduction {
|
||||||
|
margin-top: 0.5rem;
|
||||||
|
margin-bottom: 0.5rem;
|
||||||
}
|
}
|
||||||
|
|||||||
1
batch_ia/.gitignore
vendored
Normal file
1
batch_ia/.gitignore
vendored
Normal file
@ -0,0 +1 @@
|
|||||||
|
temp_sections/
|
||||||
27
batch_ia/__init__.py
Normal file
27
batch_ia/__init__.py
Normal file
@ -0,0 +1,27 @@
|
|||||||
|
# batch_ia/__init__.py
|
||||||
|
|
||||||
|
# config.py
|
||||||
|
from .utils.config import TEMPLATE_PATH, load_config
|
||||||
|
|
||||||
|
# files.py
|
||||||
|
from .utils.files import write_report
|
||||||
|
|
||||||
|
# graphs.py
|
||||||
|
from .utils.graphs import (
|
||||||
|
parse_graphs,
|
||||||
|
extract_data_from_graph,
|
||||||
|
calculate_vulnerabilities
|
||||||
|
)
|
||||||
|
|
||||||
|
# sections.py
|
||||||
|
from .utils.sections import generate_report
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"TEMPLATE_PATH",
|
||||||
|
"load_config",
|
||||||
|
"write_report",
|
||||||
|
"parse_graphs",
|
||||||
|
"extract_data_from_graph",
|
||||||
|
"calculate_vulnerabilities",
|
||||||
|
"generate_report",
|
||||||
|
]
|
||||||
75
batch_ia/analyse_ia.py
Normal file
75
batch_ia/analyse_ia.py
Normal file
@ -0,0 +1,75 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
"""
|
||||||
|
Script pour générer un rapport factorisé des vulnérabilités critiques
|
||||||
|
suivant la structure définie dans Remarques.md.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
import streamlit as st
|
||||||
|
|
||||||
|
from utils.config import (
|
||||||
|
TEMP_SECTIONS,
|
||||||
|
TEMPLATE_PATH, session_uuid,
|
||||||
|
load_config
|
||||||
|
)
|
||||||
|
|
||||||
|
from utils.files import (
|
||||||
|
write_report
|
||||||
|
)
|
||||||
|
|
||||||
|
from utils.graphs import (
|
||||||
|
parse_graphs,
|
||||||
|
extract_data_from_graph,
|
||||||
|
calculate_vulnerabilities
|
||||||
|
)
|
||||||
|
|
||||||
|
from utils.sections import (
|
||||||
|
generate_report
|
||||||
|
)
|
||||||
|
|
||||||
|
from utils.sections_utils import (
|
||||||
|
nettoyer_texte_fr
|
||||||
|
)
|
||||||
|
|
||||||
|
from utils.ia import (
|
||||||
|
ingest_document,
|
||||||
|
ia_analyse,
|
||||||
|
supprimer_fichiers,
|
||||||
|
generer_rapport_final
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main(dot_path, output_path):
|
||||||
|
"""Fonction principale du script."""
|
||||||
|
# Charger la configuration
|
||||||
|
config = load_config()
|
||||||
|
|
||||||
|
# Analyser les graphes
|
||||||
|
graph, ref_graph = parse_graphs(dot_path)
|
||||||
|
# Extraire les données
|
||||||
|
data = extract_data_from_graph(graph, ref_graph)
|
||||||
|
# Calculer les vulnérabilités
|
||||||
|
results = calculate_vulnerabilities(data, config)
|
||||||
|
if "step" not in st.session_state:
|
||||||
|
st.session_state["step"] = 1
|
||||||
|
# Générer le rapport
|
||||||
|
report, file_names = generate_report(data, results, config)
|
||||||
|
# Écrire le rapport
|
||||||
|
write_report(report, TEMPLATE_PATH)
|
||||||
|
ingest_document(TEMPLATE_PATH)
|
||||||
|
# Générer l'analyse par l'IA du rapport compler
|
||||||
|
analyse_finale = nettoyer_texte_fr(ia_analyse(file_names))
|
||||||
|
analyse_fichier = TEMP_SECTIONS / TEMPLATE_PATH.name.replace(".md", " - analyse.md")
|
||||||
|
write_report(analyse_finale, analyse_fichier)
|
||||||
|
|
||||||
|
if generer_rapport_final(TEMPLATE_PATH, analyse_fichier, output_path):
|
||||||
|
supprimer_fichiers(session_uuid)
|
||||||
|
else:
|
||||||
|
print("")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
dot_path = Path(sys.argv[1])
|
||||||
|
output_path = Path(sys.argv[2])
|
||||||
|
main(dot_path, output_path)
|
||||||
31
batch_ia/batch-fabnum-dev.service
Normal file
31
batch_ia/batch-fabnum-dev.service
Normal file
@ -0,0 +1,31 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Service batch IA pour utilisateur fabnum
|
||||||
|
After=network.target
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=simple
|
||||||
|
User=fabnum
|
||||||
|
WorkingDirectory=/home/fabnum/fabnum-dev/batch_ia
|
||||||
|
Environment=PYTHONPATH=/home/fabnum/fabnum-dev
|
||||||
|
ExecStart=/home/fabnum/fabnum-dev/venv/bin/python /home/fabnum/fabnum-dev/batch_ia/batch_runner.py
|
||||||
|
Restart=always
|
||||||
|
Nice=10
|
||||||
|
CPUSchedulingPolicy=batch
|
||||||
|
|
||||||
|
# Limites de ressources
|
||||||
|
CPUQuota=87.5% # ~14 cores sur 16
|
||||||
|
MemoryMax=12G # RAM maximale autorisée
|
||||||
|
TasksMax=1 # maximum 1 subprocess/thread simultané
|
||||||
|
|
||||||
|
# Sécurité renforcée
|
||||||
|
ProtectSystem=full
|
||||||
|
ReadWritePaths=/home/fabnum/fabnum-dev/batch_ia
|
||||||
|
|
||||||
|
# Journal propre
|
||||||
|
StandardOutput=journal
|
||||||
|
StandardError=journal
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.target
|
||||||
|
|
||||||
|
# semanage fcontext -a -t svirt_sandbox_file_t "/home/fabnum/fabnum-dev/batch_ia(/.*)?"
|
||||||
32
batch_ia/batch_runner.py
Normal file
32
batch_ia/batch_runner.py
Normal file
@ -0,0 +1,32 @@
|
|||||||
|
import time
|
||||||
|
import subprocess
|
||||||
|
from batch_utils import charger_status, sauvegarder_status, JOBS_DIR
|
||||||
|
|
||||||
|
|
||||||
|
while True:
|
||||||
|
status = charger_status()
|
||||||
|
jobs = [(login, data) for login, data in status.items() if data["status"] == "en attente"]
|
||||||
|
jobs = sorted(jobs, key=lambda x: x[1].get("timestamp", 0))
|
||||||
|
|
||||||
|
for i, (login, data) in enumerate(jobs):
|
||||||
|
status[login]["position"] = i
|
||||||
|
sauvegarder_status(status)
|
||||||
|
|
||||||
|
if jobs:
|
||||||
|
login, _ = jobs[0]
|
||||||
|
dot_file = JOBS_DIR / f"{login}.dot"
|
||||||
|
result_file = JOBS_DIR / f"{login}.zip"
|
||||||
|
|
||||||
|
status[login]["status"] = "en cours"
|
||||||
|
sauvegarder_status(status)
|
||||||
|
|
||||||
|
try:
|
||||||
|
subprocess.run(["python3", "analyse_ia.py", str(dot_file), str(result_file)], check=True)
|
||||||
|
status[login]["status"] = "terminé"
|
||||||
|
except Exception as e:
|
||||||
|
status[login]["status"] = "échoué"
|
||||||
|
status[login]["error"] = str(e)
|
||||||
|
|
||||||
|
sauvegarder_status(status)
|
||||||
|
|
||||||
|
time.sleep(60)
|
||||||
64
batch_ia/batch_utils.py
Normal file
64
batch_ia/batch_utils.py
Normal file
@ -0,0 +1,64 @@
|
|||||||
|
import json
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
from networkx.drawing.nx_agraph import write_dot
|
||||||
|
import streamlit as st
|
||||||
|
from utils.translations import _
|
||||||
|
|
||||||
|
BATCH_DIR = Path(__file__).resolve().parent
|
||||||
|
JOBS_DIR = BATCH_DIR / "jobs"
|
||||||
|
STATUS_FILE = BATCH_DIR / "status.json"
|
||||||
|
ANALYSE = " - analyse.md"
|
||||||
|
RAPPORT = " - rapport.md"
|
||||||
|
|
||||||
|
def charger_status():
|
||||||
|
if STATUS_FILE.exists():
|
||||||
|
return json.loads(STATUS_FILE.read_text())
|
||||||
|
return {}
|
||||||
|
|
||||||
|
def sauvegarder_status(data):
|
||||||
|
STATUS_FILE.write_text(json.dumps(data, indent=2))
|
||||||
|
|
||||||
|
def statut_utilisateur(login):
|
||||||
|
status = charger_status()
|
||||||
|
entry = status.get(login)
|
||||||
|
if not entry:
|
||||||
|
return {"statut": None, "position": None, "telechargement": None,
|
||||||
|
"message": f"{str(_('batch.no_task'))}."}
|
||||||
|
|
||||||
|
if entry["status"] == "en attente":
|
||||||
|
return {"statut": "en attente", "position": entry.get("position"),
|
||||||
|
"telechargement": None,
|
||||||
|
"message": f"{str(_('batch.in_queue'))} (position {entry.get('position', '?')})."}
|
||||||
|
|
||||||
|
if entry["status"] == "en cours":
|
||||||
|
if "step" not in st.session_state:
|
||||||
|
st.session_state["step"] = 1
|
||||||
|
return {"statut": "en cours", "position": 0,
|
||||||
|
"telechargement": None, "message": f"{str(_('batch.in_progress'))} ({str(_('batch.step'))} {st.session_state["step"]}/5)."}
|
||||||
|
|
||||||
|
if entry["status"] == "terminé":
|
||||||
|
result_file = JOBS_DIR / f"{login}.zip"
|
||||||
|
if result_file.exists():
|
||||||
|
return {"statut": "terminé", "position": None,
|
||||||
|
"telechargement": result_file.read_bytes(),
|
||||||
|
"message": f"{str(_('batch.complete'))}."}
|
||||||
|
|
||||||
|
if entry["status"] == "échoué":
|
||||||
|
return {"statut": "échoué", "position": None, "telechargement": None,
|
||||||
|
"message": f"{str(_('batch.failure'))} : {entry.get('error', {str(_('batch.unknown_error'))})}"}
|
||||||
|
|
||||||
|
def soumettre_batch(login, G):
|
||||||
|
if statut_utilisateur(login)["statut"]:
|
||||||
|
raise RuntimeError("Un batch est déjà en cours.")
|
||||||
|
write_dot(G, JOBS_DIR / f"{login}.dot")
|
||||||
|
status = charger_status()
|
||||||
|
status[login] = {"status": "en attente", "timestamp": time.time()}
|
||||||
|
sauvegarder_status(status)
|
||||||
|
|
||||||
|
def nettoyage_post_telechargement(login):
|
||||||
|
(JOBS_DIR / f"{login}.dot").unlink(missing_ok=True)
|
||||||
|
(JOBS_DIR / f"{login}.zip").unlink(missing_ok=True)
|
||||||
|
status = charger_status()
|
||||||
|
status.pop(login, None)
|
||||||
|
sauvegarder_status(status)
|
||||||
100
batch_ia/nettoyer_pgpt.py
Normal file
100
batch_ia/nettoyer_pgpt.py
Normal file
@ -0,0 +1,100 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Script de nettoyage pour PrivateGPT
|
||||||
|
|
||||||
|
Ce script permet de lister et supprimer les documents ingérés dans PrivateGPT.
|
||||||
|
Options:
|
||||||
|
- Lister tous les documents
|
||||||
|
- Supprimer des documents par préfixe (ex: "temp_section_")
|
||||||
|
- Supprimer des documents par motif
|
||||||
|
- Supprimer tous les documents
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import requests
|
||||||
|
import time
|
||||||
|
from typing import List, Dict, Any, Optional
|
||||||
|
|
||||||
|
# Configuration de l'API PrivateGPT
|
||||||
|
PGPT_URL = "http://127.0.0.1:8001"
|
||||||
|
API_URL = f"{PGPT_URL}/v1"
|
||||||
|
|
||||||
|
def list_documents() -> List[Dict[str, Any]]:
|
||||||
|
"""Liste tous les documents ingérés et renvoie la liste des métadonnées"""
|
||||||
|
try:
|
||||||
|
# Récupérer la liste des documents
|
||||||
|
response = requests.get(f"{API_URL}/ingest/list")
|
||||||
|
response.raise_for_status()
|
||||||
|
data = response.json()
|
||||||
|
|
||||||
|
# Format de réponse OpenAI
|
||||||
|
if "data" in data:
|
||||||
|
documents = data.get("data", [])
|
||||||
|
# Format alternatif
|
||||||
|
else:
|
||||||
|
documents = data.get("documents", [])
|
||||||
|
|
||||||
|
# Construire une liste normalisée des documents
|
||||||
|
normalized_docs = []
|
||||||
|
for doc in documents:
|
||||||
|
doc_id = doc.get("doc_id") or doc.get("id")
|
||||||
|
metadata = doc.get("doc_metadata", {})
|
||||||
|
filename = metadata.get("file_name") or metadata.get("filename", "Inconnu")
|
||||||
|
|
||||||
|
normalized_docs.append({
|
||||||
|
"id": doc_id,
|
||||||
|
"filename": filename,
|
||||||
|
"metadata": metadata
|
||||||
|
})
|
||||||
|
|
||||||
|
return normalized_docs
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Erreur lors de la récupération des documents: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def delete_document(doc_id: str) -> bool:
|
||||||
|
"""Supprime un document par son ID"""
|
||||||
|
try:
|
||||||
|
response = requests.delete(f"{API_URL}/ingest/{doc_id}")
|
||||||
|
if response.status_code == 200:
|
||||||
|
return True
|
||||||
|
else:
|
||||||
|
print(f"⚠️ Échec de la suppression de l'ID {doc_id}: Code {response.status_code}")
|
||||||
|
return False
|
||||||
|
except Exception as e:
|
||||||
|
print(f"❌ Erreur lors de la suppression de l'ID {doc_id}: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def delete_documents_by_criteria(pattern) -> int:
|
||||||
|
"""
|
||||||
|
Supprime des documents selon différents critères
|
||||||
|
Retourne le nombre de documents supprimés
|
||||||
|
"""
|
||||||
|
|
||||||
|
documents = list_documents()
|
||||||
|
|
||||||
|
if not documents or not pattern:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Comptage des suppressions réussies
|
||||||
|
success_count = 0
|
||||||
|
|
||||||
|
# Filtrer les documents à supprimer
|
||||||
|
docs_to_delete = []
|
||||||
|
try:
|
||||||
|
regex = re.compile(pattern)
|
||||||
|
docs_to_delete = [doc for doc in documents if regex.search(doc["filename"])]
|
||||||
|
except re.error as e:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Supprimer les documents
|
||||||
|
for doc in docs_to_delete:
|
||||||
|
delete_document(doc["id"])
|
||||||
|
# Petite pause pour éviter de surcharger l'API
|
||||||
|
time.sleep(0.1)
|
||||||
|
|
||||||
|
return success_count
|
||||||
1
batch_ia/status.json
Normal file
1
batch_ia/status.json
Normal file
@ -0,0 +1 @@
|
|||||||
|
{}
|
||||||
1
batch_ia/utils/__init__.py
Normal file
1
batch_ia/utils/__init__.py
Normal file
@ -0,0 +1 @@
|
|||||||
|
|
||||||
93
batch_ia/utils/config.py
Normal file
93
batch_ia/utils/config.py
Normal file
@ -0,0 +1,93 @@
|
|||||||
|
import os
|
||||||
|
import yaml
|
||||||
|
from pathlib import Path
|
||||||
|
import uuid
|
||||||
|
|
||||||
|
def init_uuid():
|
||||||
|
if not TEMP_SECTIONS.exists():
|
||||||
|
TEMP_SECTIONS.mkdir(parents=True)
|
||||||
|
session_uuid = str(uuid.uuid4())[:8] # Utiliser les 8 premiers caractères pour plus de concision
|
||||||
|
print(f"🔑 UUID de session généré: {session_uuid}")
|
||||||
|
return session_uuid
|
||||||
|
|
||||||
|
BASE_DIR = Path(__file__).resolve().parent.parent
|
||||||
|
CORPUS_DIR = BASE_DIR.parent / "Corpus"
|
||||||
|
THRESHOLDS_PATH = BASE_DIR.parent / "assets" / "config.yaml"
|
||||||
|
REFERENCE_GRAPH_PATH = BASE_DIR.parent / "schema.txt"
|
||||||
|
GRAPH_PATH = BASE_DIR.parent / "graphe.dot"
|
||||||
|
TEMP_SECTIONS = BASE_DIR / "temp_sections"
|
||||||
|
session_uuid = init_uuid()
|
||||||
|
TEMPLATE_PATH = TEMP_SECTIONS / f"rapport_final - {session_uuid}.md"
|
||||||
|
|
||||||
|
PGPT_URL = "http://127.0.0.1:8001"
|
||||||
|
API_URL = f"{PGPT_URL}/v1"
|
||||||
|
PROMPT_METHODOLOGIE = """
|
||||||
|
Le rapport à examiner a été établi à partir de la méthodologie suivante.
|
||||||
|
|
||||||
|
Le dispositif d’évaluation des risques proposé repose sur quatre indices clairement définis, chacun analysant un aspect spécifique des risques dans la chaîne d’approvisionnement numérique. L’indice IHH mesure la concentration géographique ou industrielle, permettant d’évaluer la dépendance vis-à-vis de certains acteurs ou régions. L’indice ISG indique la stabilité géopolitique des pays impliqués dans la chaîne de production, en intégrant des critères politiques, sociaux et climatiques. L’indice ICS quantifie la facilité ou la difficulté à remplacer ou substituer un élément spécifique dans la chaîne, évaluant ainsi les risques liés à la dépendance technologique et économique. Enfin, l’indice IVC examine la pression concurrentielle sur les ressources utilisées par le numérique, révélant ainsi le risque potentiel que ces ressources soient détournées vers d’autres secteurs industriels.
|
||||||
|
|
||||||
|
Ces indices se combinent judicieusement par paires pour une évaluation approfondie et pertinente des risques. La combinaison IHH-ISG permet d’associer la gravité d'un impact potentiel (IHH) à la probabilité de survenance d’un événement perturbateur (ISG), créant ainsi une matrice de vulnérabilité combinée utile pour identifier rapidement les points critiques dans la chaîne de production. La combinaison ICS-IVC fonctionne selon la même logique, mais se concentre spécifiquement sur les ressources minérales : l’ICS indique la gravité potentielle d'une rupture d'approvisionnement due à une faible substituabilité, tandis que l’IVC évalue la probabilité que les ressources soient captées par d'autres secteurs industriels concurrents. Ces combinaisons permettent d’obtenir une analyse précise et opérationnelle du niveau de risque global.
|
||||||
|
|
||||||
|
Les avantages de cette méthodologie résident dans son approche à la fois systématique et granulaire, adaptée à l'échelle décisionnelle d'un COMEX. Elle permet d’identifier avec précision les vulnérabilités majeures et leurs origines spécifiques, facilitant ainsi la prise de décision stratégique éclairée et proactive. En combinant des facteurs géopolitiques, industriels, technologiques et concurrentiels, ces indices offrent un suivi efficace de la chaîne de fabrication numérique, garantissant ainsi une gestion optimale des risques et la continuité opérationnelle à long terme.
|
||||||
|
"""
|
||||||
|
|
||||||
|
DICTIONNAIRE_CRITICITES = {
|
||||||
|
"IHH": {"vert": "Faible", "orange": "Modérée", "rouge": "Élevée"},
|
||||||
|
"ISG": {"vert": "Stable", "orange": "Intermédiaire", "rouge": "Instable"},
|
||||||
|
"ICS": {"vert": "Facile", "orange": "Moyenne", "rouge": "Difficile"},
|
||||||
|
"IVC": {"vert": "Faible", "orange": "Modérée", "rouge": "Forte"}
|
||||||
|
}
|
||||||
|
POIDS_COULEURS = {
|
||||||
|
"Vert": 1,
|
||||||
|
"Orange": 2,
|
||||||
|
"Rouge": 3
|
||||||
|
}
|
||||||
|
|
||||||
|
def load_config(thresholds_path=THRESHOLDS_PATH):
|
||||||
|
"""Charge la configuration depuis les fichiers YAML."""
|
||||||
|
config = {}
|
||||||
|
# Charger les seuils
|
||||||
|
if os.path.exists(thresholds_path):
|
||||||
|
with open(thresholds_path, 'r', encoding='utf-8') as f:
|
||||||
|
thresholds = yaml.safe_load(f)
|
||||||
|
config['thresholds'] = thresholds.get('seuils', {})
|
||||||
|
return config
|
||||||
|
|
||||||
|
def determine_threshold_color(value, index_type, thresholds):
|
||||||
|
"""
|
||||||
|
Détermine la couleur du seuil en fonction du type d'indice et de sa valeur.
|
||||||
|
Utilise les seuils de config.yaml si disponibles.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Récupérer les seuils pour cet indice
|
||||||
|
if index_type in thresholds:
|
||||||
|
index_thresholds = thresholds[index_type]
|
||||||
|
# Déterminer la couleur
|
||||||
|
if "vert" in index_thresholds and "max" in index_thresholds["vert"] and \
|
||||||
|
index_thresholds["vert"]["max"] is not None and value < index_thresholds["vert"]["max"]:
|
||||||
|
suffix = get_suffix_for_index(index_type, "vert")
|
||||||
|
return "Vert", suffix
|
||||||
|
elif "orange" in index_thresholds and "min" in index_thresholds["orange"] and "max" in index_thresholds["orange"] and \
|
||||||
|
index_thresholds["orange"]["min"] is not None and index_thresholds["orange"]["max"] is not None and \
|
||||||
|
index_thresholds["orange"]["min"] <= value < index_thresholds["orange"]["max"]:
|
||||||
|
suffix = get_suffix_for_index(index_type, "orange")
|
||||||
|
return "Orange", suffix
|
||||||
|
elif "rouge" in index_thresholds and "min" in index_thresholds["rouge"] and \
|
||||||
|
index_thresholds["rouge"]["min"] is not None and value >= index_thresholds["rouge"]["min"]:
|
||||||
|
suffix = get_suffix_for_index(index_type, "rouge")
|
||||||
|
return "Rouge", suffix
|
||||||
|
|
||||||
|
return "Non déterminé", ""
|
||||||
|
|
||||||
|
def get_suffix_for_index(index_type, color):
|
||||||
|
"""Retourne le suffixe approprié pour chaque indice et couleur."""
|
||||||
|
suffixes = DICTIONNAIRE_CRITICITES
|
||||||
|
|
||||||
|
if index_type in suffixes and color in suffixes[index_type]:
|
||||||
|
return suffixes[index_type][color]
|
||||||
|
return ""
|
||||||
|
|
||||||
|
def get_weight_for_color(color):
|
||||||
|
"""Retourne le poids correspondant à une couleur."""
|
||||||
|
weights = POIDS_COULEURS
|
||||||
|
return weights.get(color, 0)
|
||||||
133
batch_ia/utils/files.py
Normal file
133
batch_ia/utils/files.py
Normal file
@ -0,0 +1,133 @@
|
|||||||
|
import os
|
||||||
|
import re
|
||||||
|
|
||||||
|
from .config import (
|
||||||
|
CORPUS_DIR
|
||||||
|
)
|
||||||
|
|
||||||
|
def strip_prefix(name):
|
||||||
|
"""Supprime le préfixe numérique éventuel d'un nom de fichier ou de dossier."""
|
||||||
|
return re.sub(r'^\d+[-_ ]*', '', name).lower()
|
||||||
|
|
||||||
|
def find_prefixed_directory(pattern, base_path=None):
|
||||||
|
"""
|
||||||
|
Recherche un sous-répertoire dont le nom (sans préfixe) correspond au pattern.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
pattern: Nom du répertoire sans préfixe
|
||||||
|
base_path: Répertoire de base où chercher
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Le chemin relatif du répertoire trouvé (avec préfixe) ou None
|
||||||
|
"""
|
||||||
|
if base_path:
|
||||||
|
search_path = os.path.join(CORPUS_DIR, base_path)
|
||||||
|
else:
|
||||||
|
search_path = CORPUS_DIR
|
||||||
|
|
||||||
|
if not os.path.exists(search_path):
|
||||||
|
# print(f"Chemin inexistant: {search_path}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
for d in os.listdir(search_path):
|
||||||
|
dir_path = os.path.join(search_path, d)
|
||||||
|
if os.path.isdir(dir_path) and strip_prefix(d) == pattern.lower():
|
||||||
|
return os.path.relpath(dir_path, CORPUS_DIR)
|
||||||
|
|
||||||
|
# print(f"Aucun répertoire correspondant à: '{pattern}' trouvé dans {search_path}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
def find_corpus_file(pattern, base_path=None):
|
||||||
|
"""
|
||||||
|
Recherche récursive dans le corpus d'un fichier en ignorant les préfixes numériques dans les dossiers et fichiers.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
pattern: Chemin relatif type "sous-dossier/nom-fichier"
|
||||||
|
base_path: Dossier de base à partir duquel chercher
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Chemin relatif du fichier trouvé ou None
|
||||||
|
"""
|
||||||
|
|
||||||
|
if base_path:
|
||||||
|
search_path = os.path.join(CORPUS_DIR, base_path)
|
||||||
|
else:
|
||||||
|
search_path = CORPUS_DIR
|
||||||
|
|
||||||
|
# # print(f"Recherche de: '{pattern}' dans {search_path}")
|
||||||
|
|
||||||
|
if not os.path.exists(search_path):
|
||||||
|
# print(pattern)
|
||||||
|
# print(base_path)
|
||||||
|
# print(f"Chemin inexistant: {search_path}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
if '/' not in pattern:
|
||||||
|
# Recherche directe d'un fichier
|
||||||
|
for file in os.listdir(search_path):
|
||||||
|
if not file.endswith('.md'):
|
||||||
|
continue
|
||||||
|
if strip_prefix(os.path.splitext(file)[0]) == pattern.lower():
|
||||||
|
rel_path = os.path.relpath(os.path.join(search_path, file), CORPUS_DIR)
|
||||||
|
# # print(f"Fichier trouvé: {rel_path}")
|
||||||
|
return rel_path
|
||||||
|
else:
|
||||||
|
# Séparation du chemin en dossier/fichier
|
||||||
|
first, rest = pattern.split('/', 1)
|
||||||
|
matched_dir = find_prefixed_directory(first, base_path)
|
||||||
|
if matched_dir:
|
||||||
|
return find_corpus_file(rest, matched_dir)
|
||||||
|
|
||||||
|
# print(f"Aucun fichier correspondant à: '{pattern}' trouvé dans {base_path}.")
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def read_corpus_file(file_path, remove_first_title=False, shift_titles=0):
|
||||||
|
"""
|
||||||
|
Lit un fichier du corpus et applique les transformations demandées.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_path: Chemin relatif du fichier dans le corpus
|
||||||
|
remove_first_title: Si True, supprime la première ligne de titre
|
||||||
|
shift_titles: Nombre de niveaux à ajouter aux titres
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Le contenu du fichier avec les transformations appliquées
|
||||||
|
"""
|
||||||
|
full_path = os.path.join(CORPUS_DIR, file_path)
|
||||||
|
|
||||||
|
if not os.path.exists(full_path):
|
||||||
|
# print(f"Fichier non trouvé: {full_path}")
|
||||||
|
return f"Fichier non trouvé: {file_path}"
|
||||||
|
|
||||||
|
# # print(f"Lecture du fichier: {full_path}")
|
||||||
|
with open(full_path, 'r', encoding='utf-8') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
|
||||||
|
# Supprimer la première ligne si c'est un titre et si demandé
|
||||||
|
if remove_first_title and lines and lines[0].startswith('#'):
|
||||||
|
# # print(f"Suppression du titre: {lines[0].strip()}")
|
||||||
|
lines = lines[1:]
|
||||||
|
|
||||||
|
# Décaler les niveaux de titre si demandé
|
||||||
|
if shift_titles > 0:
|
||||||
|
for i in range(len(lines)):
|
||||||
|
if lines[i].startswith('#'):
|
||||||
|
lines[i] = '#' * shift_titles + lines[i]
|
||||||
|
|
||||||
|
# Nettoyer les retours à la ligne superflus
|
||||||
|
content = ''.join(lines)
|
||||||
|
# Supprimer les retours à la ligne en fin de contenu
|
||||||
|
content = content.rstrip('\n') + '\n'
|
||||||
|
|
||||||
|
return content
|
||||||
|
|
||||||
|
def write_report(report, fichier):
|
||||||
|
"""Écrit le rapport généré dans le fichier spécifié."""
|
||||||
|
|
||||||
|
report = re.sub(r'<!----.*?-->', '', report)
|
||||||
|
report = re.sub(r'\n\n\n+', '\n\n', report)
|
||||||
|
|
||||||
|
with open(fichier, 'w', encoding='utf-8') as f:
|
||||||
|
f.write(report)
|
||||||
|
# print(f"Rapport généré avec succès: {TEMPLATE_PATH}")
|
||||||
591
batch_ia/utils/graphs.py
Normal file
591
batch_ia/utils/graphs.py
Normal file
@ -0,0 +1,591 @@
|
|||||||
|
import os
|
||||||
|
import sys
|
||||||
|
from networkx.drawing.nx_agraph import read_dot
|
||||||
|
|
||||||
|
from .config import (
|
||||||
|
REFERENCE_GRAPH_PATH,
|
||||||
|
determine_threshold_color, get_weight_for_color
|
||||||
|
)
|
||||||
|
|
||||||
|
def parse_graphs(graphe_path):
|
||||||
|
"""
|
||||||
|
Charge et analyse les graphes DOT (analyse et référence).
|
||||||
|
"""
|
||||||
|
print(graphe_path)
|
||||||
|
# Charger le graphe à analyser
|
||||||
|
if not os.path.exists(graphe_path):
|
||||||
|
print(f"Fichier de graphe à analyser introuvable: {graphe_path}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
# Charger le graphe de référence
|
||||||
|
reference_path = REFERENCE_GRAPH_PATH
|
||||||
|
if not os.path.exists(reference_path):
|
||||||
|
print(f"Fichier de graphe de référence introuvable: {reference_path}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Charger les graphes avec NetworkX
|
||||||
|
graph = read_dot(graphe_path)
|
||||||
|
ref_graph = read_dot(reference_path)
|
||||||
|
|
||||||
|
# Convertir les attributs en types appropriés pour les deux graphes
|
||||||
|
for g in [graph, ref_graph]:
|
||||||
|
for node, attrs in g.nodes(data=True):
|
||||||
|
for key, value in list(attrs.items()):
|
||||||
|
# Convertir les valeurs numériques
|
||||||
|
if key in ['niveau', 'ihh_acteurs', 'ihh_pays', 'isg', 'ivc']:
|
||||||
|
try:
|
||||||
|
if key in ['isg', 'ivc', 'ihh_acteurs', 'ihh_pays', 'niveau']:
|
||||||
|
attrs[key] = int(value.strip('"'))
|
||||||
|
else:
|
||||||
|
attrs[key] = float(value.strip('"'))
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
# Garder la valeur originale si la conversion échoue
|
||||||
|
pass
|
||||||
|
elif key == 'label':
|
||||||
|
# Nettoyer les guillemets des étiquettes
|
||||||
|
attrs[key] = value.strip('"')
|
||||||
|
|
||||||
|
# Convertir les attributs des arêtes
|
||||||
|
for u, v, attrs in g.edges(data=True):
|
||||||
|
for key, value in list(attrs.items()):
|
||||||
|
if key in ['ics', 'cout', 'delai', 'technique']:
|
||||||
|
try:
|
||||||
|
attrs[key] = float(value.strip('"'))
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
elif key == 'label' and '%' in value:
|
||||||
|
# Extraire le pourcentage
|
||||||
|
try:
|
||||||
|
percentage = value.strip('"').replace('%', '')
|
||||||
|
attrs['percentage'] = float(percentage)
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
return graph, ref_graph
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Erreur lors de l'analyse des graphes: {str(e)}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
def extract_data_from_graph(graph, ref_graph):
|
||||||
|
"""
|
||||||
|
Extrait toutes les données pertinentes des graphes DOT.
|
||||||
|
"""
|
||||||
|
data = {
|
||||||
|
"products": {}, # Produits finaux (N0)
|
||||||
|
"components": {}, # Composants (N1)
|
||||||
|
"minerals": {}, # Minerais (N2)
|
||||||
|
"operations": {}, # Opérations (N10)
|
||||||
|
"countries": {}, # Pays (N11)
|
||||||
|
"geo_countries": {}, # Pays géographiques (N99)
|
||||||
|
"actors": {} # Acteurs (N12)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Extraire tous les pays géographiques du graphe de référence
|
||||||
|
for node, attrs in ref_graph.nodes(data=True):
|
||||||
|
if attrs.get('niveau') == 99:
|
||||||
|
country_name = attrs.get('label', node)
|
||||||
|
isg_value = attrs.get('isg', 0)
|
||||||
|
|
||||||
|
data["geo_countries"][country_name] = {
|
||||||
|
"id": node,
|
||||||
|
"isg": isg_value
|
||||||
|
}
|
||||||
|
|
||||||
|
# Extraire les nœuds du graphe à analyser
|
||||||
|
for node, attrs in graph.nodes(data=True):
|
||||||
|
level = attrs.get('niveau', -1)
|
||||||
|
label = attrs.get('label', node)
|
||||||
|
|
||||||
|
if level == 0 or level == 1000: # Produit final
|
||||||
|
data["products"][node] = {
|
||||||
|
"label": label,
|
||||||
|
"components": [],
|
||||||
|
"assembly": None,
|
||||||
|
"level": level
|
||||||
|
}
|
||||||
|
elif level == 1 or level == 1001: # Composant
|
||||||
|
data["components"][node] = {
|
||||||
|
"label": label,
|
||||||
|
"minerals": [],
|
||||||
|
"manufacturing": None
|
||||||
|
}
|
||||||
|
elif level == 2: # Minerai
|
||||||
|
data["minerals"][node] = {
|
||||||
|
"label": label,
|
||||||
|
"ivc": attrs.get('ivc', 0),
|
||||||
|
"extraction": None,
|
||||||
|
"treatment": None,
|
||||||
|
"ics_values": {}
|
||||||
|
}
|
||||||
|
elif level == 10 or level == 1010: # Opération
|
||||||
|
op_type = label.lower()
|
||||||
|
data["operations"][node] = {
|
||||||
|
"label": label,
|
||||||
|
"type": op_type,
|
||||||
|
"ihh_acteurs": attrs.get('ihh_acteurs', 0),
|
||||||
|
"ihh_pays": attrs.get('ihh_pays', 0),
|
||||||
|
"countries": {}
|
||||||
|
}
|
||||||
|
elif level == 11 or level == 1011: # Pays
|
||||||
|
data["countries"][node] = {
|
||||||
|
"label": label,
|
||||||
|
"actors": {},
|
||||||
|
"geo_country": None,
|
||||||
|
"market_share": 0
|
||||||
|
}
|
||||||
|
elif level == 12 or level == 1012: # Acteur
|
||||||
|
data["actors"][node] = {
|
||||||
|
"label": label,
|
||||||
|
"country": None,
|
||||||
|
"market_share": 0
|
||||||
|
}
|
||||||
|
|
||||||
|
# Extraire les relations et attributs des arêtes
|
||||||
|
for source, target, edge_attrs in graph.edges(data=True):
|
||||||
|
if source not in graph.nodes or target not in graph.nodes:
|
||||||
|
continue
|
||||||
|
|
||||||
|
source_level = graph.nodes[source].get('niveau', -1)
|
||||||
|
target_level = graph.nodes[target].get('niveau', -1)
|
||||||
|
|
||||||
|
# Extraire part de marché
|
||||||
|
market_share = 0
|
||||||
|
if 'percentage' in edge_attrs:
|
||||||
|
market_share = edge_attrs['percentage']
|
||||||
|
elif 'label' in edge_attrs and '%' in edge_attrs['label']:
|
||||||
|
try:
|
||||||
|
market_share = float(edge_attrs['label'].strip('"').replace('%', ''))
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Relations produit → composant
|
||||||
|
if (source_level == 0 and target_level == 1) or (source_level == 1000 and target_level == 1001):
|
||||||
|
if target not in data["products"][source]["components"]:
|
||||||
|
data["products"][source]["components"].append(target)
|
||||||
|
|
||||||
|
# Relations produit → opération (assemblage)
|
||||||
|
elif (source_level == 0 and target_level == 10) or (source_level == 1000 and target_level == 1010):
|
||||||
|
if graph.nodes[target].get('label', '').lower() == 'assemblage':
|
||||||
|
data["products"][source]["assembly"] = target
|
||||||
|
|
||||||
|
# Relations composant → minerai avec ICS
|
||||||
|
elif (source_level == 1 or source_level == 1001) and target_level == 2:
|
||||||
|
if target not in data["components"][source]["minerals"]:
|
||||||
|
data["components"][source]["minerals"].append(target)
|
||||||
|
|
||||||
|
# Stocker l'ICS s'il est présent
|
||||||
|
if 'ics' in edge_attrs:
|
||||||
|
ics_value = edge_attrs['ics']
|
||||||
|
data["minerals"][target]["ics_values"][source] = ics_value
|
||||||
|
|
||||||
|
# Relations composant → opération (fabrication)
|
||||||
|
elif (source_level == 1 or source_level == 1001) and target_level == 10:
|
||||||
|
if graph.nodes[target].get('label', '').lower() == 'fabrication':
|
||||||
|
data["components"][source]["manufacturing"] = target
|
||||||
|
|
||||||
|
# Relations minerai → opération (extraction/traitement)
|
||||||
|
elif source_level == 2 and target_level == 10:
|
||||||
|
op_label = graph.nodes[target].get('label', '').lower()
|
||||||
|
if 'extraction' in op_label:
|
||||||
|
data["minerals"][source]["extraction"] = target
|
||||||
|
elif 'traitement' in op_label:
|
||||||
|
data["minerals"][source]["treatment"] = target
|
||||||
|
|
||||||
|
# Relations opération → pays avec part de marché
|
||||||
|
elif (source_level == 10 and target_level == 11) or (source_level == 1010 and target_level == 1011):
|
||||||
|
data["operations"][source]["countries"][target] = market_share
|
||||||
|
data["countries"][target]["market_share"] = market_share
|
||||||
|
|
||||||
|
# Relations pays → acteur avec part de marché
|
||||||
|
elif (source_level == 11 and target_level == 12) or (source_level == 1011 and target_level == 1012):
|
||||||
|
data["countries"][source]["actors"][target] = market_share
|
||||||
|
data["actors"][target]["market_share"] = market_share
|
||||||
|
data["actors"][target]["country"] = source
|
||||||
|
|
||||||
|
# Relations pays → pays géographique
|
||||||
|
elif (source_level == 11 or source_level == 1011) and target_level == 99:
|
||||||
|
country_name = graph.nodes[target].get('label', '')
|
||||||
|
data["countries"][source]["geo_country"] = country_name
|
||||||
|
|
||||||
|
# Compléter les opérations manquantes pour les produits et composants
|
||||||
|
# en les récupérant du graphe de référence si elles existent
|
||||||
|
|
||||||
|
# Pour les produits finaux (N0)
|
||||||
|
for product_id, product_data in data["products"].items():
|
||||||
|
if product_data["assembly"] is None:
|
||||||
|
# Chercher l'opération d'assemblage dans le graphe de référence
|
||||||
|
for source, target, edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (source == product_id and
|
||||||
|
((ref_graph.nodes[source].get('niveau') == 0 and
|
||||||
|
ref_graph.nodes[target].get('niveau') == 10) or
|
||||||
|
(ref_graph.nodes[source].get('niveau') == 1000 and
|
||||||
|
ref_graph.nodes[target].get('niveau') == 1010)) and
|
||||||
|
ref_graph.nodes[target].get('label', '').lower() == 'assemblage'):
|
||||||
|
|
||||||
|
# L'opération existe dans le graphe de référence
|
||||||
|
assembly_id = target
|
||||||
|
product_data["assembly"] = assembly_id
|
||||||
|
|
||||||
|
# Ajouter l'opération si elle n'existe pas déjà
|
||||||
|
if assembly_id not in data["operations"]:
|
||||||
|
data["operations"][assembly_id] = {
|
||||||
|
"label": ref_graph.nodes[assembly_id].get('label', assembly_id),
|
||||||
|
"type": "assemblage",
|
||||||
|
"ihh_acteurs": ref_graph.nodes[assembly_id].get('ihh_acteurs', 0),
|
||||||
|
"ihh_pays": ref_graph.nodes[assembly_id].get('ihh_pays', 0),
|
||||||
|
"countries": {}
|
||||||
|
}
|
||||||
|
|
||||||
|
# Extraire les relations de l'opération vers les pays
|
||||||
|
for op_source, op_target, op_edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (op_source == assembly_id and
|
||||||
|
(ref_graph.nodes[op_target].get('niveau') == 11 or ref_graph.nodes[op_target].get('niveau') == 1011)):
|
||||||
|
|
||||||
|
country_id = op_target
|
||||||
|
|
||||||
|
# Extraire part de marché
|
||||||
|
market_share = 0
|
||||||
|
if 'percentage' in op_edge_attrs:
|
||||||
|
market_share = op_edge_attrs['percentage']
|
||||||
|
elif 'label' in op_edge_attrs and '%' in op_edge_attrs['label']:
|
||||||
|
try:
|
||||||
|
market_share = float(op_edge_attrs['label'].strip('"').replace('%', ''))
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Ajouter le pays à l'opération
|
||||||
|
data["operations"][assembly_id]["countries"][country_id] = market_share
|
||||||
|
|
||||||
|
# Ajouter le pays s'il n'existe pas déjà
|
||||||
|
if country_id not in data["countries"]:
|
||||||
|
data["countries"][country_id] = {
|
||||||
|
"label": ref_graph.nodes[country_id].get('label', country_id),
|
||||||
|
"actors": {},
|
||||||
|
"geo_country": None,
|
||||||
|
"market_share": market_share
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
data["countries"][country_id]["market_share"] = market_share
|
||||||
|
|
||||||
|
# Extraire les relations du pays vers les acteurs
|
||||||
|
for country_source, country_target, country_edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (country_source == country_id and
|
||||||
|
(ref_graph.nodes[country_target].get('niveau') == 12 or ref_graph.nodes[country_target].get('niveau') == 1012)):
|
||||||
|
|
||||||
|
actor_id = country_target
|
||||||
|
|
||||||
|
# Extraire part de marché
|
||||||
|
actor_market_share = 0
|
||||||
|
if 'percentage' in country_edge_attrs:
|
||||||
|
actor_market_share = country_edge_attrs['percentage']
|
||||||
|
elif 'label' in country_edge_attrs and '%' in country_edge_attrs['label']:
|
||||||
|
try:
|
||||||
|
actor_market_share = float(country_edge_attrs['label'].strip('"').replace('%', ''))
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Ajouter l'acteur au pays
|
||||||
|
data["countries"][country_id]["actors"][actor_id] = actor_market_share
|
||||||
|
|
||||||
|
# Ajouter l'acteur s'il n'existe pas déjà
|
||||||
|
if actor_id not in data["actors"]:
|
||||||
|
data["actors"][actor_id] = {
|
||||||
|
"label": ref_graph.nodes[actor_id].get('label', actor_id),
|
||||||
|
"country": country_id,
|
||||||
|
"market_share": actor_market_share
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
data["actors"][actor_id]["market_share"] = actor_market_share
|
||||||
|
data["actors"][actor_id]["country"] = country_id
|
||||||
|
|
||||||
|
# Extraire la relation du pays vers le pays géographique
|
||||||
|
for geo_source, geo_target, geo_edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (geo_source == country_id and
|
||||||
|
ref_graph.nodes[geo_target].get('niveau') == 99):
|
||||||
|
|
||||||
|
geo_country_name = ref_graph.nodes[geo_target].get('label', '')
|
||||||
|
data["countries"][country_id]["geo_country"] = geo_country_name
|
||||||
|
|
||||||
|
break # Une seule opération d'assemblage par produit
|
||||||
|
|
||||||
|
# Pour les composants (N1)
|
||||||
|
for component_id, component_data in data["components"].items():
|
||||||
|
if component_data["manufacturing"] is None:
|
||||||
|
# Chercher l'opération de fabrication dans le graphe de référence
|
||||||
|
for source, target, edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (source == component_id and
|
||||||
|
((ref_graph.nodes[source].get('niveau') == 1 and
|
||||||
|
ref_graph.nodes[target].get('niveau') == 10) or
|
||||||
|
(ref_graph.nodes[source].get('niveau') == 1001 and
|
||||||
|
ref_graph.nodes[target].get('niveau') == 1010)) and
|
||||||
|
ref_graph.nodes[target].get('label', '').lower() == 'fabrication'):
|
||||||
|
|
||||||
|
# L'opération existe dans le graphe de référence
|
||||||
|
manufacturing_id = target
|
||||||
|
component_data["manufacturing"] = manufacturing_id
|
||||||
|
|
||||||
|
# Ajouter l'opération si elle n'existe pas déjà
|
||||||
|
if manufacturing_id not in data["operations"]:
|
||||||
|
data["operations"][manufacturing_id] = {
|
||||||
|
"label": ref_graph.nodes[manufacturing_id].get('label', manufacturing_id),
|
||||||
|
"type": "fabrication",
|
||||||
|
"ihh_acteurs": ref_graph.nodes[manufacturing_id].get('ihh_acteurs', 0),
|
||||||
|
"ihh_pays": ref_graph.nodes[manufacturing_id].get('ihh_pays', 0),
|
||||||
|
"countries": {}
|
||||||
|
}
|
||||||
|
|
||||||
|
# Extraire les relations de l'opération vers les pays
|
||||||
|
for op_source, op_target, op_edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (op_source == manufacturing_id and
|
||||||
|
(ref_graph.nodes[op_target].get('niveau') == 11 or ref_graph.nodes[op_target].get('niveau') == 1011)):
|
||||||
|
|
||||||
|
country_id = op_target
|
||||||
|
|
||||||
|
# Extraire part de marché
|
||||||
|
market_share = 0
|
||||||
|
if 'percentage' in op_edge_attrs:
|
||||||
|
market_share = op_edge_attrs['percentage']
|
||||||
|
elif 'label' in op_edge_attrs and '%' in op_edge_attrs['label']:
|
||||||
|
try:
|
||||||
|
market_share = float(op_edge_attrs['label'].strip('"').replace('%', ''))
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Ajouter le pays à l'opération
|
||||||
|
data["operations"][manufacturing_id]["countries"][country_id] = market_share
|
||||||
|
|
||||||
|
# Ajouter le pays s'il n'existe pas déjà
|
||||||
|
if country_id not in data["countries"]:
|
||||||
|
data["countries"][country_id] = {
|
||||||
|
"label": ref_graph.nodes[country_id].get('label', country_id),
|
||||||
|
"actors": {},
|
||||||
|
"geo_country": None,
|
||||||
|
"market_share": market_share
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
data["countries"][country_id]["market_share"] = market_share
|
||||||
|
|
||||||
|
# Extraire les relations du pays vers les acteurs
|
||||||
|
for country_source, country_target, country_edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (country_source == country_id and
|
||||||
|
(ref_graph.nodes[country_target].get('niveau') == 12 or ref_graph.nodes[country_target].get('niveau') == 1012)):
|
||||||
|
|
||||||
|
actor_id = country_target
|
||||||
|
|
||||||
|
# Extraire part de marché
|
||||||
|
actor_market_share = 0
|
||||||
|
if 'percentage' in country_edge_attrs:
|
||||||
|
actor_market_share = country_edge_attrs['percentage']
|
||||||
|
elif 'label' in country_edge_attrs and '%' in country_edge_attrs['label']:
|
||||||
|
try:
|
||||||
|
actor_market_share = float(country_edge_attrs['label'].strip('"').replace('%', ''))
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Ajouter l'acteur au pays
|
||||||
|
data["countries"][country_id]["actors"][actor_id] = actor_market_share
|
||||||
|
|
||||||
|
# Ajouter l'acteur s'il n'existe pas déjà
|
||||||
|
if actor_id not in data["actors"]:
|
||||||
|
data["actors"][actor_id] = {
|
||||||
|
"label": ref_graph.nodes[actor_id].get('label', actor_id),
|
||||||
|
"country": country_id,
|
||||||
|
"market_share": actor_market_share
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
data["actors"][actor_id]["market_share"] = actor_market_share
|
||||||
|
data["actors"][actor_id]["country"] = country_id
|
||||||
|
|
||||||
|
# Extraire la relation du pays vers le pays géographique
|
||||||
|
for geo_source, geo_target, geo_edge_attrs in ref_graph.edges(data=True):
|
||||||
|
if (geo_source == country_id and
|
||||||
|
ref_graph.nodes[geo_target].get('niveau') == 99):
|
||||||
|
|
||||||
|
geo_country_name = ref_graph.nodes[geo_target].get('label', '')
|
||||||
|
data["countries"][country_id]["geo_country"] = geo_country_name
|
||||||
|
|
||||||
|
break # Une seule opération de fabrication par composant
|
||||||
|
|
||||||
|
return data
|
||||||
|
|
||||||
|
def calculate_vulnerabilities(data, config):
|
||||||
|
"""
|
||||||
|
Calcule les vulnérabilités combinées pour toutes les opérations et minerais.
|
||||||
|
"""
|
||||||
|
thresholds = config.get('thresholds', {})
|
||||||
|
results = {
|
||||||
|
"ihh_isg_combined": {}, # Pour chaque opération
|
||||||
|
"ics_ivc_combined": {}, # Pour chaque minerai
|
||||||
|
"chains": [] # Pour stocker tous les chemins possibles
|
||||||
|
}
|
||||||
|
|
||||||
|
# 1. Calculer ISG_combiné pour chaque opération
|
||||||
|
for op_id, operation in data["operations"].items():
|
||||||
|
isg_weighted_sum = 0
|
||||||
|
total_share = 0
|
||||||
|
|
||||||
|
# Parcourir chaque pays impliqué dans l'opération
|
||||||
|
for country_id, share in operation["countries"].items():
|
||||||
|
country = data["countries"][country_id]
|
||||||
|
geo_country = country.get("geo_country")
|
||||||
|
|
||||||
|
if geo_country and geo_country in data["geo_countries"]:
|
||||||
|
isg_value = data["geo_countries"][geo_country]["isg"]
|
||||||
|
isg_weighted_sum += isg_value * share
|
||||||
|
total_share += share
|
||||||
|
|
||||||
|
# Calculer la moyenne pondérée
|
||||||
|
isg_combined = 0
|
||||||
|
if total_share > 0:
|
||||||
|
isg_combined = isg_weighted_sum / total_share
|
||||||
|
|
||||||
|
# Déterminer couleurs et poids
|
||||||
|
ihh_value = operation["ihh_pays"]
|
||||||
|
ihh_color, ihh_suffix = determine_threshold_color(ihh_value, "IHH", thresholds)
|
||||||
|
isg_color, isg_suffix = determine_threshold_color(isg_combined, "ISG", thresholds)
|
||||||
|
|
||||||
|
# Calculer poids combiné
|
||||||
|
ihh_weight = get_weight_for_color(ihh_color)
|
||||||
|
isg_weight = get_weight_for_color(isg_color)
|
||||||
|
combined_weight = ihh_weight * isg_weight
|
||||||
|
|
||||||
|
# Déterminer vulnérabilité combinée
|
||||||
|
if combined_weight in [6, 9]:
|
||||||
|
vulnerability = "ÉLEVÉE à CRITIQUE"
|
||||||
|
elif combined_weight in [3, 4]:
|
||||||
|
vulnerability = "MOYENNE"
|
||||||
|
else: # 1, 2
|
||||||
|
vulnerability = "FAIBLE"
|
||||||
|
|
||||||
|
# Stocker résultats
|
||||||
|
results["ihh_isg_combined"][op_id] = {
|
||||||
|
"ihh_value": ihh_value,
|
||||||
|
"ihh_color": ihh_color,
|
||||||
|
"ihh_suffix": ihh_suffix,
|
||||||
|
"isg_combined": isg_combined,
|
||||||
|
"isg_color": isg_color,
|
||||||
|
"isg_suffix": isg_suffix,
|
||||||
|
"combined_weight": combined_weight,
|
||||||
|
"vulnerability": vulnerability
|
||||||
|
}
|
||||||
|
|
||||||
|
# 2. Calculer ICS_moyen pour chaque minerai
|
||||||
|
for mineral_id, mineral in data["minerals"].items():
|
||||||
|
ics_values = list(mineral["ics_values"].values())
|
||||||
|
ics_average = 0
|
||||||
|
|
||||||
|
if ics_values:
|
||||||
|
ics_average = sum(ics_values) / len(ics_values)
|
||||||
|
|
||||||
|
ivc_value = mineral.get("ivc", 0)
|
||||||
|
|
||||||
|
# Déterminer couleurs et poids
|
||||||
|
ics_color, ics_suffix = determine_threshold_color(ics_average, "ICS", thresholds)
|
||||||
|
ivc_color, ivc_suffix = determine_threshold_color(ivc_value, "IVC", thresholds)
|
||||||
|
|
||||||
|
# Calculer poids combiné
|
||||||
|
ics_weight = get_weight_for_color(ics_color)
|
||||||
|
ivc_weight = get_weight_for_color(ivc_color)
|
||||||
|
combined_weight = ics_weight * ivc_weight
|
||||||
|
|
||||||
|
# Déterminer vulnérabilité combinée
|
||||||
|
if combined_weight in [6, 9]:
|
||||||
|
vulnerability = "ÉLEVÉE à CRITIQUE"
|
||||||
|
elif combined_weight in [3, 4]:
|
||||||
|
vulnerability = "MOYENNE"
|
||||||
|
else: # 1, 2
|
||||||
|
vulnerability = "FAIBLE"
|
||||||
|
|
||||||
|
# Stocker résultats
|
||||||
|
results["ics_ivc_combined"][mineral_id] = {
|
||||||
|
"ics_average": ics_average,
|
||||||
|
"ics_color": ics_color,
|
||||||
|
"ics_suffix": ics_suffix,
|
||||||
|
"ivc_value": ivc_value,
|
||||||
|
"ivc_color": ivc_color,
|
||||||
|
"ivc_suffix": ivc_suffix,
|
||||||
|
"combined_weight": combined_weight,
|
||||||
|
"vulnerability": vulnerability
|
||||||
|
}
|
||||||
|
|
||||||
|
# 3. Identifier tous les chemins et leurs vulnérabilités
|
||||||
|
for product_id, product in data["products"].items():
|
||||||
|
for component_id in product["components"]:
|
||||||
|
component = data["components"][component_id]
|
||||||
|
|
||||||
|
for mineral_id in component["minerals"]:
|
||||||
|
mineral = data["minerals"][mineral_id]
|
||||||
|
|
||||||
|
# Collecter toutes les vulnérabilités dans ce chemin
|
||||||
|
path_vulnerabilities = []
|
||||||
|
|
||||||
|
# Assemblage (si présent)
|
||||||
|
assembly_id = product["assembly"]
|
||||||
|
if assembly_id and assembly_id in results["ihh_isg_combined"]:
|
||||||
|
path_vulnerabilities.append({
|
||||||
|
"type": "assemblage",
|
||||||
|
"vulnerability": results["ihh_isg_combined"][assembly_id]["vulnerability"],
|
||||||
|
"operation_id": assembly_id
|
||||||
|
})
|
||||||
|
|
||||||
|
# Fabrication (si présent)
|
||||||
|
manufacturing_id = component["manufacturing"]
|
||||||
|
if manufacturing_id and manufacturing_id in results["ihh_isg_combined"]:
|
||||||
|
path_vulnerabilities.append({
|
||||||
|
"type": "fabrication",
|
||||||
|
"vulnerability": results["ihh_isg_combined"][manufacturing_id]["vulnerability"],
|
||||||
|
"operation_id": manufacturing_id
|
||||||
|
})
|
||||||
|
|
||||||
|
# Minerai (ICS+IVC)
|
||||||
|
if mineral_id in results["ics_ivc_combined"]:
|
||||||
|
path_vulnerabilities.append({
|
||||||
|
"type": "minerai",
|
||||||
|
"vulnerability": results["ics_ivc_combined"][mineral_id]["vulnerability"],
|
||||||
|
"mineral_id": mineral_id
|
||||||
|
})
|
||||||
|
|
||||||
|
# Extraction (si présent)
|
||||||
|
extraction_id = mineral["extraction"]
|
||||||
|
if extraction_id and extraction_id in results["ihh_isg_combined"]:
|
||||||
|
path_vulnerabilities.append({
|
||||||
|
"type": "extraction",
|
||||||
|
"vulnerability": results["ihh_isg_combined"][extraction_id]["vulnerability"],
|
||||||
|
"operation_id": extraction_id
|
||||||
|
})
|
||||||
|
|
||||||
|
# Traitement (si présent)
|
||||||
|
treatment_id = mineral["treatment"]
|
||||||
|
if treatment_id and treatment_id in results["ihh_isg_combined"]:
|
||||||
|
path_vulnerabilities.append({
|
||||||
|
"type": "traitement",
|
||||||
|
"vulnerability": results["ihh_isg_combined"][treatment_id]["vulnerability"],
|
||||||
|
"operation_id": treatment_id
|
||||||
|
})
|
||||||
|
|
||||||
|
# Classifier le chemin
|
||||||
|
path_info = {
|
||||||
|
"product": product_id,
|
||||||
|
"component": component_id,
|
||||||
|
"mineral": mineral_id,
|
||||||
|
"vulnerabilities": path_vulnerabilities
|
||||||
|
}
|
||||||
|
|
||||||
|
# Déterminer le niveau de risque du chemin
|
||||||
|
critical_count = path_vulnerabilities.count({"vulnerability": "ÉLEVÉE à CRITIQUE"})
|
||||||
|
medium_count = path_vulnerabilities.count({"vulnerability": "MOYENNE"})
|
||||||
|
|
||||||
|
if any(v["vulnerability"] == "ÉLEVÉE à CRITIQUE" for v in path_vulnerabilities):
|
||||||
|
path_info["risk_level"] = "critique"
|
||||||
|
elif medium_count >= 3:
|
||||||
|
path_info["risk_level"] = "majeur"
|
||||||
|
elif any(v["vulnerability"] == "MOYENNE" for v in path_vulnerabilities):
|
||||||
|
path_info["risk_level"] = "moyen"
|
||||||
|
else:
|
||||||
|
path_info["risk_level"] = "faible"
|
||||||
|
|
||||||
|
results["chains"].append(path_info)
|
||||||
|
|
||||||
|
return results
|
||||||
286
batch_ia/utils/ia.py
Normal file
286
batch_ia/utils/ia.py
Normal file
@ -0,0 +1,286 @@
|
|||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
import requests
|
||||||
|
import json
|
||||||
|
import time
|
||||||
|
import zipfile
|
||||||
|
import streamlit as st
|
||||||
|
|
||||||
|
from nettoyer_pgpt import (
|
||||||
|
delete_documents_by_criteria
|
||||||
|
)
|
||||||
|
|
||||||
|
from utils.config import (
|
||||||
|
TEMP_SECTIONS,
|
||||||
|
session_uuid,
|
||||||
|
API_URL, PROMPT_METHODOLOGIE
|
||||||
|
)
|
||||||
|
|
||||||
|
def ingest_document(file_path: Path) -> bool:
|
||||||
|
"""Ingère un document dans PrivateGPT"""
|
||||||
|
try:
|
||||||
|
with open(file_path, "rb") as f:
|
||||||
|
file_name = file_path.name
|
||||||
|
|
||||||
|
files = {"file": (file_name, f, "text/markdown")}
|
||||||
|
# Ajouter des métadonnées pour identifier facilement ce fichier d'entrée
|
||||||
|
metadata = {
|
||||||
|
"type": "input_file",
|
||||||
|
"session_id": session_uuid,
|
||||||
|
"document_type": "rapport_analyse_input"
|
||||||
|
}
|
||||||
|
response = requests.post(
|
||||||
|
f"{API_URL}/ingest/file",
|
||||||
|
files=files,
|
||||||
|
data={"metadata": json.dumps(metadata)} if "metadata" in requests.get(f"{API_URL}/ingest/file").text else None
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
print(f"✅ Document '{file_path}' ingéré avec succès sous le nom '{file_name}'")
|
||||||
|
return True
|
||||||
|
except FileNotFoundError:
|
||||||
|
print(f"❌ Fichier '{file_path}' introuvable")
|
||||||
|
return False
|
||||||
|
except requests.RequestException as e:
|
||||||
|
print(f"❌ Erreur lors de l'ingestion du document: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def generate_text(input_file, full_prompt, system_message, temperature = "0.3", use_context = True):
|
||||||
|
"""Génère du texte avec l'API PrivateGPT"""
|
||||||
|
try:
|
||||||
|
|
||||||
|
# Définir les paramètres de la requête
|
||||||
|
payload = {
|
||||||
|
"messages": [
|
||||||
|
{"role": "system", "content": system_message},
|
||||||
|
{"role": "user", "content": full_prompt}
|
||||||
|
],
|
||||||
|
"use_context": use_context, # Active la recherche RAG dans les documents ingérés
|
||||||
|
"temperature": temperature, # Température réduite pour plus de cohérence
|
||||||
|
"stream": False
|
||||||
|
}
|
||||||
|
|
||||||
|
# Tenter d'ajouter un filtre de contexte (fonctionnalité expérimentale qui peut ne pas être supportée)
|
||||||
|
if input_file:
|
||||||
|
try:
|
||||||
|
# Vérifier si le filtre de contexte est supporté sans faire de requête supplémentaire
|
||||||
|
liste_des_fichiers = list(TEMP_SECTIONS.glob(f"*{session_uuid}*.md"))
|
||||||
|
filter_metadata = {
|
||||||
|
"document_name": [input_file.name] + [f.name for f in liste_des_fichiers]
|
||||||
|
}
|
||||||
|
payload["filter_metadata"] = filter_metadata
|
||||||
|
except Exception as e:
|
||||||
|
print(f"ℹ️ Remarque: Impossible d'appliquer le filtre de contexte: {e}")
|
||||||
|
|
||||||
|
# Envoyer la requête
|
||||||
|
response = requests.post(
|
||||||
|
f"{API_URL}/chat/completions",
|
||||||
|
json=payload,
|
||||||
|
headers={"accept": "application/json"}
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
# Extraire la réponse générée
|
||||||
|
result = response.json()
|
||||||
|
if "choices" in result and len(result["choices"]) > 0:
|
||||||
|
return result["choices"][0]["message"]["content"]
|
||||||
|
else:
|
||||||
|
print("⚠️ Format de réponse inattendu:", json.dumps(result, indent=2))
|
||||||
|
return None
|
||||||
|
|
||||||
|
except requests.RequestException as e:
|
||||||
|
print(f"❌ Erreur lors de la génération de texte: {e}")
|
||||||
|
if hasattr(e, 'response') and e.response is not None:
|
||||||
|
print(f"Détails: {e.response.text}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
def ia_analyse(file_names):
|
||||||
|
for file in file_names:
|
||||||
|
ingest_document(file)
|
||||||
|
time.sleep(5)
|
||||||
|
|
||||||
|
reponse = {}
|
||||||
|
for file in file_names:
|
||||||
|
produit_final = re.search(r"chemins critiques (.+)\.md$", file.name).group(1)
|
||||||
|
|
||||||
|
# Préparer le prompt avec le contexte précédent si disponible et demandé
|
||||||
|
full_prompt = f"""
|
||||||
|
Rédigez une synthèse du fichier {file.name} dédiée au produit final '{produit_final}'.
|
||||||
|
Cette synthèse, destinée spécifiquement au Directeur des Risques, membre du COMEX d'une grande entreprise utilisant ce produit, doit être claire et concise (environ 10 lignes).
|
||||||
|
|
||||||
|
En utilisant impérativement la méthodologie fournie, expliquez en termes simples mais précis, pourquoi et comment les vulnérabilités identifiées constituent un risque concret pour l'entreprise. Mentionnez clairement :
|
||||||
|
|
||||||
|
- Les composants spécifiques du produit '{produit_final}' concernés par ces vulnérabilités.
|
||||||
|
- Les minerais précis responsables de ces vulnérabilités et leur rôle dans l’impact sur les composants.
|
||||||
|
- Les points critiques exacts identifiés dans la chaîne d'approvisionnement (par exemple : faible substituabilité, forte concentration géographique, instabilité géopolitique, concurrence élevée entre secteurs industriels).
|
||||||
|
- Identifier autant que faire se peut, les pays générant la forte concentration géographiques ou qui sont sujet à instabilité géopolitique, les secteurs en concurrence avec le numérique pour les minerais.
|
||||||
|
|
||||||
|
Respectez strictement les consignes suivantes :
|
||||||
|
|
||||||
|
- N'utilisez aucun acronyme ni valeur numérique ; uniquement leur équivalent textuel (ex : criticité de substituabilité, vulnérabilité élevée ou critique, etc.).
|
||||||
|
- N'incluez à ce stade aucune préconisation ni recommandation.
|
||||||
|
|
||||||
|
Votre texte doit être parfaitement adapté à une compréhension rapide par des dirigeants d’entreprise.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
# Définir les paramètres de la requête
|
||||||
|
system_message = f"""
|
||||||
|
Vous êtes un assistant stratégique expert chargé de rédiger des synthèses destinées à des décideurs de très haut niveau (Directeurs des Risques, membres du COMEX, stratèges industriels). Vous analysez exclusivement les vulnérabilités systémiques affectant les produits numériques, à partir des données précises fournies dans le fichier {file.name}.
|
||||||
|
|
||||||
|
Votre analyse doit être rigoureuse, accessible, pertinente pour la prise de décision stratégique, et conforme à la méthodologie définie ci-dessous :
|
||||||
|
|
||||||
|
{PROMPT_METHODOLOGIE}
|
||||||
|
|
||||||
|
/no_think
|
||||||
|
"""
|
||||||
|
|
||||||
|
reponse[produit_final] = f"\n**{produit_final}**\n\n" + generate_text(file, full_prompt, system_message).split("</think>")[-1].strip()
|
||||||
|
# print(reponse[produit_final])
|
||||||
|
|
||||||
|
corps = "\n\n".join(reponse.values())
|
||||||
|
print("Corps")
|
||||||
|
|
||||||
|
st.session_state["step"] = 2
|
||||||
|
|
||||||
|
full_prompt = corps + "\n\n" + PROMPT_METHODOLOGIE
|
||||||
|
|
||||||
|
system_message = """
|
||||||
|
Vous êtes un expert en rédaction de rapports stratégiques destinés à un COMEX ou une Direction des Risques.
|
||||||
|
|
||||||
|
Votre mission est d'écrire une introduction professionnelle, claire et synthétique (maximum 7 lignes) à partir des éléments suivants :
|
||||||
|
1. Un corps d’analyse décrivant les vulnérabilités identifiées pour un produit numérique.
|
||||||
|
2. La méthodologie détaillée utilisée pour cette analyse (fourni en deuxième partie).
|
||||||
|
|
||||||
|
Votre introduction doit :
|
||||||
|
- Présenter brièvement le sujet traité (vulnérabilités du produit final), quels sont les produits finaux et les minerais concernés.
|
||||||
|
- Annoncer clairement le contenu et l'objectif de l'analyse présentée dans le corps.
|
||||||
|
- Résumer succinctement les axes méthodologiques principaux (concentration géographique ou industrielle, stabilité géopolitique, criticité de substituabilité, concurrence intersectorielle des minerais).
|
||||||
|
- Être facilement compréhensible par des décideurs de haut niveau (pas d'acronymes, ni chiffres ; uniquement des formulations textuelles).
|
||||||
|
- Être fluide, agréable à lire, avec un ton sobre et professionnel.
|
||||||
|
|
||||||
|
Répondez uniquement avec l'introduction rédigée. Ne fournissez aucune autre explication complémentaire.
|
||||||
|
|
||||||
|
/no_think
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
introduction = generate_text("", full_prompt, system_message).split("</think>")[-1].strip()
|
||||||
|
print("Introduction")
|
||||||
|
|
||||||
|
st.session_state["step"] = 3
|
||||||
|
|
||||||
|
full_prompt = corps + "\n\n" + PROMPT_METHODOLOGIE
|
||||||
|
|
||||||
|
system_message = """
|
||||||
|
Vous êtes un expert stratégique en gestion des risques liés à la chaîne de valeur numérique. Vous conseillez directement le COMEX et la Direction des Risques de grandes entreprises utilisatrices de produits numériques. Ces entreprises n'ont pour levier d’action que le choix de leurs fournisseurs ou l'allongement de la durée de vie de leur matériel.
|
||||||
|
|
||||||
|
À partir des vulnérabilités identifiées dans la première partie du prompt (corps d'analyse) et en tenant compte du contexte et de la méthodologie décrite en deuxième partie, rédigez un texte clair, structuré en deux parties distinctes :
|
||||||
|
|
||||||
|
1. **Préconisations stratégiques :**
|
||||||
|
Proposez clairement des axes concrets pour limiter les risques identifiés dans l’analyse. Ces préconisations doivent impérativement être réalistes et directement actionnables par les dirigeants compte tenu de leurs leviers limités.
|
||||||
|
|
||||||
|
2. **Indicateurs de suivi :**
|
||||||
|
Identifiez précisément les indicateurs pertinents à suivre pour évaluer régulièrement l’évolution de ces risques. Ces indicateurs doivent être inspirés directement des axes méthodologiques fournis (concentration géographique, stabilité géopolitique, substituabilité, concurrence intersectorielle) ou s’appuyer sur des bonnes pratiques reconnues.
|
||||||
|
|
||||||
|
Votre rédaction doit être fluide, concise, très professionnelle, et directement accessible à un COMEX. Évitez strictement toute explication complémentaire ou ajout superflu. Ne proposez que le texte demandé.
|
||||||
|
"""
|
||||||
|
|
||||||
|
preconisations = generate_text("", full_prompt, system_message, "0.5").split("</think>")[-1].strip()
|
||||||
|
print("Préconisations")
|
||||||
|
|
||||||
|
st.session_state["step"] = 4
|
||||||
|
|
||||||
|
full_prompt = corps + "\n\n" + preconisations
|
||||||
|
system_message = """
|
||||||
|
Vous êtes un expert stratégique spécialisé dans les risques liés à la chaîne de valeur du numérique. Vous conseillez directement le COMEX et la Direction des Risques de grandes entreprises dépendantes du numérique, dont les leviers d’action se limitent au choix des fournisseurs et à l’allongement de la durée d’utilisation du matériel.
|
||||||
|
|
||||||
|
À partir du résultat de l'analyse des vulnérabilités présenté en première partie du prompt (corps) et des préconisations stratégiques formulées en deuxième partie, rédigez une conclusion synthétique et percutante (environ 6 à 8 lignes maximum) afin de :
|
||||||
|
|
||||||
|
- Résumer clairement les principaux risques identifiés.
|
||||||
|
- Souligner brièvement les axes prioritaires proposés pour agir concrètement.
|
||||||
|
- Inviter de manière dynamique le COMEX à passer immédiatement à l'action.
|
||||||
|
|
||||||
|
Votre rédaction doit être fluide, professionnelle, claire et immédiatement exploitable par des dirigeants. Ne fournissez aucune explication supplémentaire. Ne répondez que par la conclusion demandée.
|
||||||
|
|
||||||
|
/no_think
|
||||||
|
"""
|
||||||
|
|
||||||
|
conclusion = generate_text("", full_prompt, system_message, "0.7").split("</think>")[-1].strip()
|
||||||
|
print("Conclusion")
|
||||||
|
|
||||||
|
st.session_state["step"] = 5
|
||||||
|
|
||||||
|
analyse = "# Rapport d'analyse\n\n" + \
|
||||||
|
"\n\n## Introduction\n\n" + \
|
||||||
|
introduction + \
|
||||||
|
"\n\n## Analyse des produits finaux\n\n" + \
|
||||||
|
corps + \
|
||||||
|
"\n\n## Préconisations\n\n" + \
|
||||||
|
preconisations + \
|
||||||
|
"\n\n## Conclusion\n\n" + \
|
||||||
|
conclusion + \
|
||||||
|
"\n\n## Méthodologie\n\n" + \
|
||||||
|
PROMPT_METHODOLOGIE
|
||||||
|
|
||||||
|
# fichier_a_reviser = Path(TEMPLATE_PATH.name.replace(".md", " - analyse à relire.md"))
|
||||||
|
# write_report(analyse, TEMP_SECTIONS / fichier_a_reviser)
|
||||||
|
# ingest_document(TEMP_SECTIONS / fichier_a_reviser)
|
||||||
|
|
||||||
|
full_prompt = """
|
||||||
|
Suivre scrupuleusement les consignes.
|
||||||
|
"""
|
||||||
|
|
||||||
|
system_message = f"""
|
||||||
|
Vous êtes un réviseur professionnel expert en écriture stratégique, maîtrisant parfaitement la langue française et habitué à réviser des textes destinés à des dirigeants de haut niveau (COMEX).
|
||||||
|
|
||||||
|
Votre tâche unique est d'améliorer strictement la qualité rédactionnelle du texte suivant, sans modifier en aucune manière :
|
||||||
|
- la structure existante (sections, titres, sous-titres),
|
||||||
|
- l'ordre des paragraphes et des idées,
|
||||||
|
- le sens précis du contenu original,
|
||||||
|
- sans ajouter aucune information nouvelle.
|
||||||
|
|
||||||
|
Votre révision doit impérativement respecter les points suivants :
|
||||||
|
- Éliminer toutes répétitions ou redondances et varier systématiquement les tournures entre les paragraphes.
|
||||||
|
- Rendre chaque phrase claire, directe et concise. Si une phrase est trop longue, scindez-la clairement en plusieurs phrases courtes.
|
||||||
|
- Structurer chaque paragraphe en 2 à 3 parties cohérentes, reliées entre elles par des termes logiques (coordination, implication, opposition, etc.) et séparées par des retours à la ligne.
|
||||||
|
- Remplacer systématiquement les acronymes par ces expressions précises :
|
||||||
|
- ICS → « capacité à substituer un minerai »
|
||||||
|
- IHH → « concentration géographique ou industrielle »
|
||||||
|
- ISG → « stabilité géopolitique »
|
||||||
|
- IVC → « concurrence intersectorielle pour les minerais »
|
||||||
|
|
||||||
|
Votre texte final doit être parfaitement fluide, agréable à lire, adapté à un COMEX, avec un ton professionnel et sobre.
|
||||||
|
|
||||||
|
**Important : Ne répondez strictement que par le texte révisé ci-dessous, sans aucun commentaire ou explication supplémentaire.**
|
||||||
|
|
||||||
|
Voici le texte à réviser précisément :
|
||||||
|
|
||||||
|
{analyse}
|
||||||
|
|
||||||
|
/no_think
|
||||||
|
"""
|
||||||
|
revision = generate_text("", full_prompt, system_message, "0.1", False).split("</think>")[-1].strip()
|
||||||
|
print("Relecture")
|
||||||
|
|
||||||
|
return revision
|
||||||
|
|
||||||
|
def supprimer_fichiers(session_uuid):
|
||||||
|
try:
|
||||||
|
delete_documents_by_criteria(session_uuid)
|
||||||
|
for temp_file in TEMP_SECTIONS.glob(f"*{session_uuid}*.md"):
|
||||||
|
temp_file.unlink()
|
||||||
|
return True
|
||||||
|
except:
|
||||||
|
return False
|
||||||
|
|
||||||
|
def generer_rapport_final(rapport, analyse, resultat):
|
||||||
|
try:
|
||||||
|
rapport = Path(rapport)
|
||||||
|
analyse = Path(analyse)
|
||||||
|
with zipfile.ZipFile(resultat, "w") as zipf:
|
||||||
|
zipf.write(rapport, arcname=rapport.name)
|
||||||
|
zipf.write(analyse, arcname=analyse.name)
|
||||||
|
return True
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Erreur lors du zip : {e}")
|
||||||
|
return False
|
||||||
770
batch_ia/utils/sections.py
Normal file
770
batch_ia/utils/sections.py
Normal file
@ -0,0 +1,770 @@
|
|||||||
|
import os
|
||||||
|
import re
|
||||||
|
|
||||||
|
from .config import (
|
||||||
|
CORPUS_DIR,
|
||||||
|
TEMPLATE_PATH,
|
||||||
|
determine_threshold_color
|
||||||
|
)
|
||||||
|
|
||||||
|
from .files import (
|
||||||
|
find_prefixed_directory,
|
||||||
|
find_corpus_file,
|
||||||
|
write_report,
|
||||||
|
read_corpus_file
|
||||||
|
)
|
||||||
|
|
||||||
|
from .sections_utils import (
|
||||||
|
trouver_dossier_composant,
|
||||||
|
extraire_sections_par_mot_cle
|
||||||
|
)
|
||||||
|
|
||||||
|
def generate_introduction_section(data):
|
||||||
|
"""
|
||||||
|
Génère la section d'introduction du rapport.
|
||||||
|
"""
|
||||||
|
products = [p["label"] for p in data["products"].values()]
|
||||||
|
components = [c["label"] for c in data["components"].values()]
|
||||||
|
minerals = [m["label"] for m in data["minerals"].values()]
|
||||||
|
|
||||||
|
template = []
|
||||||
|
template.append("## Introduction\n")
|
||||||
|
template.append("Ce rapport analyse les vulnérabilités de la chaîne de fabrication du numérique pour :\n")
|
||||||
|
|
||||||
|
template.append("* les produits finaux : " + ", ".join(products))
|
||||||
|
template.append("* les composants : " + ", ".join(components))
|
||||||
|
template.append("* les minerais : " + ", ".join(minerals) + "\n")
|
||||||
|
|
||||||
|
return "\n".join(template)
|
||||||
|
|
||||||
|
def generate_methodology_section():
|
||||||
|
"""
|
||||||
|
Génère la section méthodologie du rapport.
|
||||||
|
"""
|
||||||
|
template = []
|
||||||
|
template.append("## Méthodologie d'analyse des risques\n")
|
||||||
|
template.append("### Indices et seuils\n")
|
||||||
|
template.append("La méthode d'évaluation intègre 4 indices et leurs combinaisons pour identifier les chemins critiques.\n")
|
||||||
|
|
||||||
|
# IHH
|
||||||
|
template.append("#### IHH (Herfindahl-Hirschmann) : concentration géographiques ou industrielle d'une opération\n")
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
ihh_context_file = "Criticités/Fiche technique IHH/00-contexte-et-objectif.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, ihh_context_file)):
|
||||||
|
template.append(read_corpus_file(ihh_context_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
ihh_context_file = find_corpus_file("contexte-et-objectif", "Criticités/Fiche technique IHH")
|
||||||
|
if ihh_context_file:
|
||||||
|
template.append(read_corpus_file(ihh_context_file, remove_first_title=True))
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
ihh_calc_file = "Criticités/Fiche technique IHH/01-mode-de-calcul/_intro.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, ihh_calc_file)):
|
||||||
|
template.append(read_corpus_file(ihh_calc_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
ihh_calc_file = find_corpus_file("mode-de-calcul/_intro", "Criticités/Fiche technique IHH")
|
||||||
|
if ihh_calc_file:
|
||||||
|
template.append(read_corpus_file(ihh_calc_file, remove_first_title=True))
|
||||||
|
|
||||||
|
template.append(" * Seuils : <15 = Vert (Faible), 15-25 = Orange (Modérée), >25 = Rouge (Élevée)\n")
|
||||||
|
|
||||||
|
# ISG
|
||||||
|
template.append("#### ISG (Stabilité Géopolitique) : stabilité des pays\n")
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
isg_context_file = "Criticités/Fiche technique ISG/00-contexte-et-objectif.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, isg_context_file)):
|
||||||
|
template.append(read_corpus_file(isg_context_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
isg_context_file = find_corpus_file("contexte-et-objectif", "Criticités/Fiche technique ISG")
|
||||||
|
if isg_context_file:
|
||||||
|
template.append(read_corpus_file(isg_context_file, remove_first_title=True))
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
isg_calc_file = "Criticités/Fiche technique ISG/01-mode-de-calcul/_intro.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, isg_calc_file)):
|
||||||
|
template.append(read_corpus_file(isg_calc_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
isg_calc_file = find_corpus_file("mode-de-calcul/_intro", "Criticités/Fiche technique ISG")
|
||||||
|
if isg_calc_file:
|
||||||
|
template.append(read_corpus_file(isg_calc_file, remove_first_title=True))
|
||||||
|
|
||||||
|
template.append(" * Seuils : <40 = Vert (Stable), 40-60 = Orange, >60 = Rouge (Instable)\n")
|
||||||
|
|
||||||
|
# ICS
|
||||||
|
template.append("#### ICS (Criticité de Substituabilité) : capacité à remplacer / substituer un élément\n")
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
ics_context_file = "Criticités/Fiche technique ICS/00-contexte-et-objectif.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, ics_context_file)):
|
||||||
|
template.append(read_corpus_file(ics_context_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
ics_context_file = find_corpus_file("contexte-et-objectif", "Criticités/Fiche technique ICS")
|
||||||
|
if ics_context_file:
|
||||||
|
template.append(read_corpus_file(ics_context_file, remove_first_title=True))
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
ics_calc_file = "Criticités/Fiche technique ICS/01-mode-de-calcul/_intro.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, ics_calc_file)):
|
||||||
|
template.append(read_corpus_file(ics_calc_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
ics_calc_file = find_corpus_file("mode-de-calcul/_intro", "Criticités/Fiche technique ICS")
|
||||||
|
if ics_calc_file:
|
||||||
|
template.append(read_corpus_file(ics_calc_file, remove_first_title=True))
|
||||||
|
|
||||||
|
template.append(" * Seuils : <0.3 = Vert (Facile), 0.3-0.6 = Orange (Moyenne), >0.6 = Rouge (Difficile)\n")
|
||||||
|
|
||||||
|
# IVC
|
||||||
|
template.append("#### IVC (Vulnérabilité de Concurrence) : pression concurrentielle avec d'autres secteurs\n")
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
ivc_context_file = "Criticités/Fiche technique IVC/00-contexte-et-objectif.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, ivc_context_file)):
|
||||||
|
template.append(read_corpus_file(ivc_context_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
ivc_context_file = find_corpus_file("contexte-et-objectif", "Criticités/Fiche technique IVC")
|
||||||
|
if ivc_context_file:
|
||||||
|
template.append(read_corpus_file(ivc_context_file, remove_first_title=True))
|
||||||
|
|
||||||
|
# Essayer d'abord avec le chemin exact
|
||||||
|
ivc_calc_file = "Criticités/Fiche technique IVC/01-mode-de-calcul/_intro.md"
|
||||||
|
if os.path.exists(os.path.join(CORPUS_DIR, ivc_calc_file)):
|
||||||
|
template.append(read_corpus_file(ivc_calc_file, remove_first_title=True))
|
||||||
|
else:
|
||||||
|
# Fallback à la recherche par motif
|
||||||
|
ivc_calc_file = find_corpus_file("mode-de-calcul/_intro", "Criticités/Fiche technique IVC")
|
||||||
|
if ivc_calc_file:
|
||||||
|
template.append(read_corpus_file(ivc_calc_file, remove_first_title=True))
|
||||||
|
|
||||||
|
template.append(" * Seuils : <5 = Vert (Faible), 5-15 = Orange (Modérée), >15 = Rouge (Forte)\n")
|
||||||
|
|
||||||
|
# Combinaison des indices
|
||||||
|
template.append("### Combinaison des indices\n")
|
||||||
|
|
||||||
|
# IHH et ISG
|
||||||
|
template.append("**IHH et ISG**\n")
|
||||||
|
template.append("Ces deux indices s'appliquent à toutes les opérations et se combinent dans l'évaluation du risque (niveau d'impact et probabilité de survenance) :\n")
|
||||||
|
template.append("* l'IHH donne le niveau d'impact => une forte concentration implique un fort impact si le risque est avéré")
|
||||||
|
template.append("* l'ISG donne la probabilité de survenance => plus les pays sont instables (et donc plus l'ISG est élevé) et plus la survenance du risque est élevée\n")
|
||||||
|
|
||||||
|
template.append("Pour évaluer le risque pour une opération, les ISG des pays sont pondérés par les parts de marché respectives pour donner un ISG combiné dont le calcul est :")
|
||||||
|
template.append("ISG_combiné = (Somme des ISG des pays multipliée par leur part de marché) / Sommes de leur part de marché\n")
|
||||||
|
|
||||||
|
template.append("On établit alors une matrice (Vert = 1, Orange = 2, Rouge = 3) et en faisant le produit des poids de l'ISG combiné et de l'IHH\n")
|
||||||
|
|
||||||
|
template.append("| ISG combiné / IHH | Vert | Orange | Rouge |")
|
||||||
|
template.append("| :-- | :-- | :-- | :-- |")
|
||||||
|
template.append("| Vert | 1 | 2 | 3 |")
|
||||||
|
template.append("| Orange | 2 | 4 | 6 |")
|
||||||
|
template.append("| Rouge | 3 | 6 | 9 |\n")
|
||||||
|
|
||||||
|
template.append("Les vulnérabilités se classent en trois niveaux pour chaque opération :\n")
|
||||||
|
template.append("* Vulnérabilité combinée élevée à critique : poids 6 et 9")
|
||||||
|
template.append("* Vulnérabilité combinée moyenne : poids 3 et 4")
|
||||||
|
template.append("* Vulnérabilité combinée faible : poids 1 et 2\n")
|
||||||
|
|
||||||
|
# ICS et IVC
|
||||||
|
template.append("**ICS et IVC**\n")
|
||||||
|
template.append("Ces deux indices se combinent dans l'évaluation du risque pour un minerai :\n")
|
||||||
|
template.append("* l'ICS donne le niveau d'impact => une faible substituabilité (et donc un ICS élevé) implique un fort impact si le risque est avéré ; l'ICS est associé à la relation entre un composant et un minerai")
|
||||||
|
template.append("* l'IVC donne la probabilité de l'impact => une forte concurrence intersectorielle (IVC élevé) implique une plus forte probabilité de survenance\n")
|
||||||
|
|
||||||
|
template.append("Par simplification, on intègre un ICS moyen d'un minerai comme étant la moyenne des ICS pour chacun des composants dans lesquels il intervient.\n")
|
||||||
|
|
||||||
|
template.append("On établit alors une matrice (Vert = 1, Orange = 2, Rouge = 3) et en faisant le produit des poids de l'ICS moyen et de l'IVC.\n")
|
||||||
|
|
||||||
|
template.append("| ICS_moyen / IVC | Vert | Orange | Rouge |")
|
||||||
|
template.append("| :-- | :-- | :-- | :-- |")
|
||||||
|
template.append("| Vert | 1 | 2 | 3 |")
|
||||||
|
template.append("| Orange | 2 | 4 | 6 |")
|
||||||
|
template.append("| Rouge | 3 | 6 | 9 |\n")
|
||||||
|
|
||||||
|
template.append("Les vulnérabilités se classent en trois niveaux pour chaque minerai :\n")
|
||||||
|
template.append("* Vulnérabilité combinée élevée à critique : poids 6 et 9")
|
||||||
|
template.append("* Vulnérabilité combinée moyenne : poids 3 et 4")
|
||||||
|
template.append("* Vulnérabilité combinée faible : poids 1 et 2\n")
|
||||||
|
|
||||||
|
return "\n".join(template)
|
||||||
|
|
||||||
|
def generate_operations_section(data, results, config):
|
||||||
|
"""
|
||||||
|
Génère la section détaillant les opérations (assemblage, fabrication, extraction, traitement).
|
||||||
|
"""
|
||||||
|
# # print("DEBUG: Génération de la section des opérations")
|
||||||
|
# # print(f"DEBUG: Nombre de produits: {len(data['products'])}")
|
||||||
|
# # print(f"DEBUG: Nombre de composants: {len(data['components'])}")
|
||||||
|
# # print(f"DEBUG: Nombre d'opérations: {len(data['operations'])}")
|
||||||
|
|
||||||
|
template = []
|
||||||
|
template.append("## Détails des opérations\n")
|
||||||
|
|
||||||
|
# 1. Traiter les produits finaux (assemblage)
|
||||||
|
for product_id, product in data["products"].items():
|
||||||
|
# # print(f"DEBUG: Produit {product_id} ({product['label']}), assembly = {product['assembly']}")
|
||||||
|
if product["assembly"]:
|
||||||
|
template.append(f"### {product['label']} et Assemblage\n")
|
||||||
|
|
||||||
|
# Récupérer la présentation synthétique
|
||||||
|
# product_slug = product['label'].lower().replace(' ', '-')
|
||||||
|
sous_repertoire = f"{product['label']}"
|
||||||
|
if product["level"] == 0:
|
||||||
|
type = "Assemblage"
|
||||||
|
else:
|
||||||
|
type = "Connexe"
|
||||||
|
sous_repertoire = trouver_dossier_composant(sous_repertoire, type, "Fiche assemblage ")
|
||||||
|
product_slug = sous_repertoire.split(' ', 2)[2]
|
||||||
|
presentation_file = find_corpus_file("présentation-synthétique", f"{type}/Fiche assemblage {product_slug}")
|
||||||
|
if presentation_file:
|
||||||
|
template.append(read_corpus_file(presentation_file, remove_first_title=True))
|
||||||
|
template.append("")
|
||||||
|
|
||||||
|
# Récupérer les principaux assembleurs
|
||||||
|
assembleurs_file = find_corpus_file("principaux-assembleurs", f"{type}/Fiche assemblage {product_slug}")
|
||||||
|
if assembleurs_file:
|
||||||
|
template.append(read_corpus_file(assembleurs_file, shift_titles=2))
|
||||||
|
template.append("")
|
||||||
|
|
||||||
|
# ISG des pays impliqués
|
||||||
|
assembly_id = product["assembly"]
|
||||||
|
operation = data["operations"][assembly_id]
|
||||||
|
|
||||||
|
template.append("##### ISG des pays impliqués\n")
|
||||||
|
template.append("| Pays | Part de marché | ISG | Criticité |")
|
||||||
|
template.append("| :-- | :-- | :-- | :-- |")
|
||||||
|
|
||||||
|
isg_weighted_sum = 0
|
||||||
|
total_share = 0
|
||||||
|
|
||||||
|
for country_id, share in operation["countries"].items():
|
||||||
|
country = data["countries"][country_id]
|
||||||
|
geo_country = country.get("geo_country")
|
||||||
|
|
||||||
|
if geo_country and geo_country in data["geo_countries"]:
|
||||||
|
isg_value = data["geo_countries"][geo_country]["isg"]
|
||||||
|
color, suffix = determine_threshold_color(isg_value, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"| {country['label']} | {share}% | {isg_value} | {color} ({suffix}) |")
|
||||||
|
|
||||||
|
isg_weighted_sum += isg_value * share
|
||||||
|
total_share += share
|
||||||
|
|
||||||
|
# Calculer ISG combiné
|
||||||
|
if total_share > 0:
|
||||||
|
isg_combined = isg_weighted_sum / total_share
|
||||||
|
color, suffix = determine_threshold_color(isg_combined, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"\n**ISG combiné: {isg_combined:.0f} - {color} ({suffix})**")
|
||||||
|
|
||||||
|
# IHH
|
||||||
|
ihh_file = find_corpus_file("matrice-des-risques-liés-à-l-assemblage/indice-de-herfindahl-hirschmann", f"{type}/Fiche assemblage {product_slug}")
|
||||||
|
if ihh_file:
|
||||||
|
template.append(read_corpus_file(ihh_file, shift_titles=1))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# Vulnérabilité combinée
|
||||||
|
if assembly_id in results["ihh_isg_combined"]:
|
||||||
|
combined = results["ihh_isg_combined"][assembly_id]
|
||||||
|
template.append("#### Vulnérabilité combinée IHH-ISG\n")
|
||||||
|
template.append(f"* IHH: {combined['ihh_value']} - {combined['ihh_color']} ({combined['ihh_suffix']})")
|
||||||
|
template.append(f"* ISG combiné: {combined['isg_combined']:.0f} - {combined['isg_color']} ({combined['isg_suffix']})")
|
||||||
|
template.append(f"* Poids combiné: {combined['combined_weight']}")
|
||||||
|
template.append(f"* Niveau de vulnérabilité: **{combined['vulnerability']}**\n")
|
||||||
|
|
||||||
|
# 2. Traiter les composants (fabrication)
|
||||||
|
for component_id, component in data["components"].items():
|
||||||
|
# # print(f"DEBUG: Composant {component_id} ({component['label']}), manufacturing = {component['manufacturing']}")
|
||||||
|
if component["manufacturing"]:
|
||||||
|
template.append(f"### {component['label']} et Fabrication\n")
|
||||||
|
|
||||||
|
# Récupérer la présentation synthétique
|
||||||
|
# component_slug = component['label'].lower().replace(' ', '-')
|
||||||
|
sous_repertoire = f"{component['label']}"
|
||||||
|
sous_repertoire = trouver_dossier_composant(sous_repertoire, "Fabrication", "Fiche fabrication ")
|
||||||
|
component_slug = sous_repertoire.split(' ', 2)[2]
|
||||||
|
presentation_file = find_corpus_file("présentation-synthétique", f"Fabrication/Fiche fabrication {component_slug}")
|
||||||
|
if presentation_file:
|
||||||
|
template.append(read_corpus_file(presentation_file, remove_first_title=True))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# Récupérer les principaux fabricants
|
||||||
|
fabricants_file = find_corpus_file("principaux-fabricants", f"Fabrication/Fiche fabrication {component_slug}")
|
||||||
|
if fabricants_file:
|
||||||
|
template.append(read_corpus_file(fabricants_file, shift_titles=2))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# ISG des pays impliqués
|
||||||
|
manufacturing_id = component["manufacturing"]
|
||||||
|
operation = data["operations"][manufacturing_id]
|
||||||
|
|
||||||
|
template.append("##### ISG des pays impliqués\n")
|
||||||
|
template.append("| Pays | Part de marché | ISG | Criticité |")
|
||||||
|
template.append("| :-- | :-- | :-- | :-- |")
|
||||||
|
|
||||||
|
isg_weighted_sum = 0
|
||||||
|
total_share = 0
|
||||||
|
|
||||||
|
for country_id, share in operation["countries"].items():
|
||||||
|
country = data["countries"][country_id]
|
||||||
|
geo_country = country.get("geo_country")
|
||||||
|
|
||||||
|
if geo_country and geo_country in data["geo_countries"]:
|
||||||
|
isg_value = data["geo_countries"][geo_country]["isg"]
|
||||||
|
color, suffix = determine_threshold_color(isg_value, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"| {country['label']} | {share}% | {isg_value} | {color} ({suffix}) |")
|
||||||
|
|
||||||
|
isg_weighted_sum += isg_value * share
|
||||||
|
total_share += share
|
||||||
|
|
||||||
|
# Calculer ISG combiné
|
||||||
|
if total_share > 0:
|
||||||
|
isg_combined = isg_weighted_sum / total_share
|
||||||
|
color, suffix = determine_threshold_color(isg_combined, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"\n**ISG combiné: {isg_combined:.0f} - {color} ({suffix})**\n\n")
|
||||||
|
|
||||||
|
# IHH
|
||||||
|
ihh_file = find_corpus_file("matrice-des-risques-liés-à-la-fabrication/indice-de-herfindahl-hirschmann", f"Fabrication/Fiche fabrication {component_slug}")
|
||||||
|
if ihh_file:
|
||||||
|
template.append(read_corpus_file(ihh_file, shift_titles=1))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# Vulnérabilité combinée
|
||||||
|
if manufacturing_id in results["ihh_isg_combined"]:
|
||||||
|
combined = results["ihh_isg_combined"][manufacturing_id]
|
||||||
|
template.append("#### Vulnérabilité combinée IHH-ISG\n")
|
||||||
|
template.append(f"* IHH: {combined['ihh_value']} - {combined['ihh_color']} ({combined['ihh_suffix']})")
|
||||||
|
template.append(f"* ISG combiné: {combined['isg_combined']:.0f} - {combined['isg_color']} ({combined['isg_suffix']})")
|
||||||
|
template.append(f"* Poids combiné: {combined['combined_weight']}")
|
||||||
|
template.append(f"* Niveau de vulnérabilité: **{combined['vulnerability']}**\n")
|
||||||
|
|
||||||
|
# 3. Traiter les minerais (détaillés dans une section séparée)
|
||||||
|
|
||||||
|
result = "\n".join(template)
|
||||||
|
# # print(f"DEBUG: Fin de génération de la section des opérations. Taille: {len(result)} caractères")
|
||||||
|
if len(result) <= 30: # Juste le titre de section
|
||||||
|
# # print("DEBUG: ALERTE - La section des opérations est vide ou presque vide!")
|
||||||
|
# Ajout d'une section de débogage dans le rapport
|
||||||
|
template.append("### DÉBOGAGE - Opérations manquantes\n")
|
||||||
|
template.append("Aucune opération d'assemblage ou de fabrication n'a été trouvée dans les données.\n")
|
||||||
|
template.append("Informations disponibles:\n")
|
||||||
|
template.append(f"* Nombre de produits: {len(data['products'])}\n")
|
||||||
|
template.append(f"* Nombre de composants: {len(data['components'])}\n")
|
||||||
|
template.append(f"* Nombre d'opérations: {len(data['operations'])}\n")
|
||||||
|
template.append("\nDétail des produits et de leurs opérations d'assemblage:\n")
|
||||||
|
for pid, p in data["products"].items():
|
||||||
|
template.append(f"* {p['label']}: {'Assemblage: ' + str(p['assembly']) if p['assembly'] else 'Pas d\'assemblage'}\n")
|
||||||
|
template.append("\nDétail des composants et de leurs opérations de fabrication:\n")
|
||||||
|
for cid, c in data["components"].items():
|
||||||
|
template.append(f"* {c['label']}: {'Fabrication: ' + str(c['manufacturing']) if c['manufacturing'] else 'Pas de fabrication'}\n")
|
||||||
|
result = "\n".join(template)
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
def generate_minerals_section(data, results, config):
|
||||||
|
"""
|
||||||
|
Génère la section détaillant les minerais et leurs opérations d'extraction et traitement.
|
||||||
|
"""
|
||||||
|
template = []
|
||||||
|
template.append("## Détails des minerais\n")
|
||||||
|
|
||||||
|
for mineral_id, mineral in data["minerals"].items():
|
||||||
|
mineral_slug = mineral['label'].lower().replace(' ', '-')
|
||||||
|
fiche_dir = f"{CORPUS_DIR}/Minerai/Fiche minerai {mineral_slug}"
|
||||||
|
if not os.path.exists(fiche_dir):
|
||||||
|
continue
|
||||||
|
|
||||||
|
template.append(f"---\n\n### {mineral['label']}\n")
|
||||||
|
|
||||||
|
# Récupérer la présentation synthétique
|
||||||
|
presentation_file = find_corpus_file("présentation-synthétique", f"Minerai/Fiche minerai {mineral_slug}")
|
||||||
|
if presentation_file:
|
||||||
|
template.append(read_corpus_file(presentation_file, remove_first_title=True))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# ICS
|
||||||
|
template.append("#### ICS\n")
|
||||||
|
|
||||||
|
ics_intro_file = find_corpus_file("risque-de-substituabilité/_intro", f"Minerai/Fiche minerai {mineral_slug}")
|
||||||
|
if ics_intro_file:
|
||||||
|
template.append(read_corpus_file(ics_intro_file, remove_first_title=True))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# Calcul de l'ICS moyen
|
||||||
|
ics_values = list(mineral["ics_values"].values())
|
||||||
|
if ics_values:
|
||||||
|
ics_average = sum(ics_values) / len(ics_values)
|
||||||
|
color, suffix = determine_threshold_color(ics_average, "ICS", config.get('thresholds'))
|
||||||
|
|
||||||
|
template.append("##### Valeurs d'ICS par composant\n")
|
||||||
|
template.append("| Composant | ICS | Criticité |")
|
||||||
|
template.append("| :-- | :-- | :-- |")
|
||||||
|
|
||||||
|
for comp_id, ics_value in mineral["ics_values"].items():
|
||||||
|
comp_name = data["components"][comp_id]["label"]
|
||||||
|
comp_color, comp_suffix = determine_threshold_color(ics_value, "ICS", config.get('thresholds'))
|
||||||
|
template.append(f"| {comp_name} | {ics_value:.2f} | {comp_color} ({comp_suffix}) |")
|
||||||
|
|
||||||
|
template.append(f"\n**ICS moyen : {ics_average:.2f} - {color} ({suffix})**\n")
|
||||||
|
|
||||||
|
# IVC
|
||||||
|
template.append("#### IVC\n\n")
|
||||||
|
|
||||||
|
# Valeur IVC
|
||||||
|
ivc_value = mineral.get("ivc", 0)
|
||||||
|
color, suffix = determine_threshold_color(ivc_value, "IVC", config.get('thresholds'))
|
||||||
|
template.append(f"**IVC: {ivc_value} - {color} ({suffix})**\n")
|
||||||
|
|
||||||
|
# Récupérer toutes les sections de vulnérabilité de concurrence
|
||||||
|
ivc_sections = []
|
||||||
|
ivc_dir = find_prefixed_directory("vulnérabilité-de-concurrence", f"Minerai/Fiche minerai {mineral_slug}")
|
||||||
|
corpus_path = os.path.join(CORPUS_DIR, ivc_dir) if os.path.exists(os.path.join(CORPUS_DIR, ivc_dir)) else None
|
||||||
|
if corpus_path:
|
||||||
|
for file in sorted(os.listdir(corpus_path)):
|
||||||
|
if file.endswith('.md') and "_intro.md" not in file and "sources" not in file:
|
||||||
|
ivc_sections.append(os.path.join(ivc_dir, file))
|
||||||
|
|
||||||
|
# Inclure chaque section IVC
|
||||||
|
for section_file in ivc_sections:
|
||||||
|
content = read_corpus_file(section_file, remove_first_title=False)
|
||||||
|
# Nettoyer les balises des fichiers IVC
|
||||||
|
content = re.sub(r'```.*?```', '', content, flags=re.DOTALL)
|
||||||
|
|
||||||
|
# Mettre le titre en italique s'il commence par un # (format Markdown pour titre)
|
||||||
|
if content and '\n' in content:
|
||||||
|
first_line, rest = content.split('\n', 1)
|
||||||
|
if first_line.strip().startswith('#'):
|
||||||
|
# Extraire le texte du titre sans les # et les espaces
|
||||||
|
title_text = first_line.strip().lstrip('#').strip()
|
||||||
|
content = f"\n*{title_text}*\n{rest.strip()}"
|
||||||
|
|
||||||
|
# Ne pas ajouter de contenu vide
|
||||||
|
if content.strip():
|
||||||
|
template.append(content.strip())
|
||||||
|
|
||||||
|
# ICS et IVC combinés
|
||||||
|
if mineral_id in results["ics_ivc_combined"]:
|
||||||
|
combined = results["ics_ivc_combined"][mineral_id]
|
||||||
|
template.append("\n#### Vulnérabilité combinée ICS-IVC\n")
|
||||||
|
template.append(f"* ICS moyen: {combined['ics_average']:.2f} - {combined['ics_color']} ({combined['ics_suffix']})")
|
||||||
|
template.append(f"* IVC: {combined['ivc_value']} - {combined['ivc_color']} ({combined['ivc_suffix']})")
|
||||||
|
template.append(f"* Poids combiné: {combined['combined_weight']}")
|
||||||
|
template.append(f"* Niveau de vulnérabilité: **{combined['vulnerability']}**\n")
|
||||||
|
|
||||||
|
# Extraction
|
||||||
|
if mineral["extraction"]:
|
||||||
|
template.append("#### Extraction\n")
|
||||||
|
|
||||||
|
# Récupérer les principaux producteurs
|
||||||
|
producers_file = find_corpus_file("principaux-producteurs-extraction", f"Minerai/Fiche minerai {mineral_slug}")
|
||||||
|
if producers_file:
|
||||||
|
template.append(read_corpus_file(producers_file, remove_first_title=True))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# ISG des pays impliqués
|
||||||
|
extraction_id = mineral["extraction"]
|
||||||
|
operation = data["operations"][extraction_id]
|
||||||
|
|
||||||
|
template.append("##### ISG des pays impliqués\n")
|
||||||
|
template.append("| Pays | Part de marché | ISG | Criticité |")
|
||||||
|
template.append("| :-- | :-- | :-- | :-- |")
|
||||||
|
|
||||||
|
isg_weighted_sum = 0
|
||||||
|
total_share = 0
|
||||||
|
|
||||||
|
for country_id, share in operation["countries"].items():
|
||||||
|
country = data["countries"][country_id]
|
||||||
|
geo_country = country.get("geo_country")
|
||||||
|
|
||||||
|
if geo_country and geo_country in data["geo_countries"]:
|
||||||
|
isg_value = data["geo_countries"][geo_country]["isg"]
|
||||||
|
color, suffix = determine_threshold_color(isg_value, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"| {country['label']} | {share}% | {isg_value} | {color} ({suffix}) |")
|
||||||
|
|
||||||
|
isg_weighted_sum += isg_value * share
|
||||||
|
total_share += share
|
||||||
|
|
||||||
|
# Calculer ISG combiné
|
||||||
|
if total_share > 0:
|
||||||
|
isg_combined = isg_weighted_sum / total_share
|
||||||
|
color, suffix = determine_threshold_color(isg_combined, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"\n**ISG combiné: {isg_combined:.0f} - {color} ({suffix})**\n")
|
||||||
|
|
||||||
|
# IHH extraction
|
||||||
|
ihh_file = find_corpus_file("matrice-des-risques/indice-de-herfindahl-hirschmann-extraction", f"Minerai/Fiche minerai {mineral_slug}")
|
||||||
|
if ihh_file:
|
||||||
|
template.append(read_corpus_file(ihh_file, shift_titles=1))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# Vulnérabilité combinée
|
||||||
|
if extraction_id in results["ihh_isg_combined"]:
|
||||||
|
combined = results["ihh_isg_combined"][extraction_id]
|
||||||
|
template.append("##### Vulnérabilité combinée IHH-ISG pour l'extraction\n")
|
||||||
|
template.append(f"* IHH: {combined['ihh_value']} - {combined['ihh_color']} ({combined['ihh_suffix']})")
|
||||||
|
template.append(f"* ISG combiné: {combined['isg_combined']:.0f} - {combined['isg_color']} ({combined['isg_suffix']})")
|
||||||
|
template.append(f"* Poids combiné: {combined['combined_weight']}")
|
||||||
|
template.append(f"* Niveau de vulnérabilité: **{combined['vulnerability']}**\n")
|
||||||
|
|
||||||
|
# Traitement
|
||||||
|
if mineral["treatment"]:
|
||||||
|
template.append("#### Traitement\n")
|
||||||
|
|
||||||
|
# Récupérer les principaux producteurs
|
||||||
|
producers_file = find_corpus_file("principaux-producteurs-traitement", f"Minerai/Fiche minerai {mineral_slug}")
|
||||||
|
if producers_file:
|
||||||
|
template.append(read_corpus_file(producers_file, remove_first_title=True))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# ISG des pays impliqués
|
||||||
|
treatment_id = mineral["treatment"]
|
||||||
|
operation = data["operations"][treatment_id]
|
||||||
|
|
||||||
|
template.append("##### ISG des pays impliqués\n")
|
||||||
|
template.append("| Pays | Part de marché | ISG | Criticité |")
|
||||||
|
template.append("| :-- | :-- | :-- | :-- |")
|
||||||
|
|
||||||
|
isg_weighted_sum = 0
|
||||||
|
total_share = 0
|
||||||
|
|
||||||
|
for country_id, share in operation["countries"].items():
|
||||||
|
country = data["countries"][country_id]
|
||||||
|
geo_country = country.get("geo_country")
|
||||||
|
|
||||||
|
if geo_country and geo_country in data["geo_countries"]:
|
||||||
|
isg_value = data["geo_countries"][geo_country]["isg"]
|
||||||
|
color, suffix = determine_threshold_color(isg_value, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"| {country['label']} | {share}% | {isg_value} | {color} ({suffix}) |")
|
||||||
|
|
||||||
|
isg_weighted_sum += isg_value * share
|
||||||
|
total_share += share
|
||||||
|
|
||||||
|
# Calculer ISG combiné
|
||||||
|
if total_share > 0:
|
||||||
|
isg_combined = isg_weighted_sum / total_share
|
||||||
|
color, suffix = determine_threshold_color(isg_combined, "ISG", config.get('thresholds'))
|
||||||
|
template.append(f"\n**ISG combiné: {isg_combined:.0f} - {color} ({suffix})**\n")
|
||||||
|
|
||||||
|
# IHH traitement
|
||||||
|
ihh_file = find_corpus_file("matrice-des-risques/indice-de-herfindahl-hirschmann-traitement", f"Minerai/Fiche minerai {mineral_slug}")
|
||||||
|
if ihh_file:
|
||||||
|
template.append(read_corpus_file(ihh_file, shift_titles=1))
|
||||||
|
template.append("\n")
|
||||||
|
|
||||||
|
# Vulnérabilité combinée
|
||||||
|
if treatment_id in results["ihh_isg_combined"]:
|
||||||
|
combined = results["ihh_isg_combined"][treatment_id]
|
||||||
|
template.append("##### Vulnérabilité combinée IHH-ISG pour le traitement\n")
|
||||||
|
template.append(f"* IHH: {combined['ihh_value']} - {combined['ihh_color']} ({combined['ihh_suffix']})")
|
||||||
|
template.append(f"* ISG combiné: {combined['isg_combined']:.0f} - {combined['isg_color']} ({combined['isg_suffix']})")
|
||||||
|
template.append(f"* Poids combiné: {combined['combined_weight']}")
|
||||||
|
template.append(f"* Niveau de vulnérabilité: **{combined['vulnerability']}**\n")
|
||||||
|
|
||||||
|
return "\n".join(template)
|
||||||
|
|
||||||
|
def generate_critical_paths_section(data, results):
|
||||||
|
"""
|
||||||
|
Génère la section des chemins critiques.
|
||||||
|
"""
|
||||||
|
template = []
|
||||||
|
template.append("## Chemins critiques\n")
|
||||||
|
|
||||||
|
# Récupérer les chaînes par niveau de risque
|
||||||
|
critical_chains = []
|
||||||
|
major_chains = []
|
||||||
|
medium_chains = []
|
||||||
|
|
||||||
|
for chain in results["chains"]:
|
||||||
|
if chain["risk_level"] == "critique":
|
||||||
|
critical_chains.append(chain)
|
||||||
|
elif chain["risk_level"] == "majeur":
|
||||||
|
major_chains.append(chain)
|
||||||
|
elif chain["risk_level"] == "moyen":
|
||||||
|
medium_chains.append(chain)
|
||||||
|
|
||||||
|
# 1. Chaînes critiques
|
||||||
|
template.append("### Chaînes avec risque critique\n")
|
||||||
|
template.append("*Ces chaînes comprennent au moins une vulnérabilité combinée élevée à critique*\n")
|
||||||
|
|
||||||
|
if critical_chains:
|
||||||
|
for chain in critical_chains:
|
||||||
|
product_name = data["products"][chain["product"]]["label"]
|
||||||
|
component_name = data["components"][chain["component"]]["label"]
|
||||||
|
mineral_name = data["minerals"][chain["mineral"]]["label"]
|
||||||
|
|
||||||
|
template.append(f"#### {product_name} → {component_name} → {mineral_name}\n")
|
||||||
|
|
||||||
|
# Vulnérabilités
|
||||||
|
template.append("**Vulnérabilités identifiées:**\n")
|
||||||
|
for vuln in chain["vulnerabilities"]:
|
||||||
|
vuln_type = vuln["type"].capitalize()
|
||||||
|
vuln_level = vuln["vulnerability"]
|
||||||
|
|
||||||
|
if vuln_type == "Minerai":
|
||||||
|
mineral_id = vuln["mineral_id"]
|
||||||
|
template.append(f"* {vuln_type} ({mineral_name}): {vuln_level}")
|
||||||
|
if mineral_id in results["ics_ivc_combined"]:
|
||||||
|
combined = results["ics_ivc_combined"][mineral_id]
|
||||||
|
template.append(f" * ICS moyen: {combined['ics_average']:.2f} - {combined['ics_color']}")
|
||||||
|
template.append(f" * IVC: {combined['ivc_value']} - {combined['ivc_color']}")
|
||||||
|
else:
|
||||||
|
op_id = vuln["operation_id"]
|
||||||
|
op_label = data["operations"][op_id]["label"]
|
||||||
|
template.append(f"* {vuln_type} ({op_label}): {vuln_level}")
|
||||||
|
if op_id in results["ihh_isg_combined"]:
|
||||||
|
combined = results["ihh_isg_combined"][op_id]
|
||||||
|
template.append(f" * IHH: {combined['ihh_value']} - {combined['ihh_color']}")
|
||||||
|
template.append(f" * ISG combiné: {combined['isg_combined']:.0f} - {combined['isg_color']}")
|
||||||
|
|
||||||
|
template.append("\n")
|
||||||
|
else:
|
||||||
|
template.append("Aucune chaîne à risque critique identifiée.\n")
|
||||||
|
|
||||||
|
# 2. Chaînes majeures
|
||||||
|
template.append("### Chaînes avec risque majeur\n")
|
||||||
|
template.append("*Ces chaînes comprennent au moins trois vulnérabilités combinées moyennes*\n")
|
||||||
|
|
||||||
|
if major_chains:
|
||||||
|
for chain in major_chains:
|
||||||
|
product_name = data["products"][chain["product"]]["label"]
|
||||||
|
component_name = data["components"][chain["component"]]["label"]
|
||||||
|
mineral_name = data["minerals"][chain["mineral"]]["label"]
|
||||||
|
|
||||||
|
template.append(f"#### {product_name} → {component_name} → {mineral_name}\n")
|
||||||
|
|
||||||
|
# Vulnérabilités
|
||||||
|
template.append("**Vulnérabilités identifiées:**\n")
|
||||||
|
for vuln in chain["vulnerabilities"]:
|
||||||
|
vuln_type = vuln["type"].capitalize()
|
||||||
|
vuln_level = vuln["vulnerability"]
|
||||||
|
|
||||||
|
if vuln_type == "Minerai":
|
||||||
|
mineral_id = vuln["mineral_id"]
|
||||||
|
template.append(f"* {vuln_type} ({mineral_name}): {vuln_level}\n")
|
||||||
|
if mineral_id in results["ics_ivc_combined"]:
|
||||||
|
combined = results["ics_ivc_combined"][mineral_id]
|
||||||
|
template.append(f" * ICS moyen: {combined['ics_average']:.2f} - {combined['ics_color']}\n")
|
||||||
|
template.append(f" * IVC: {combined['ivc_value']} - {combined['ivc_color']}\n")
|
||||||
|
else:
|
||||||
|
op_id = vuln["operation_id"]
|
||||||
|
op_label = data["operations"][op_id]["label"]
|
||||||
|
template.append(f"* {vuln_type} ({op_label}): {vuln_level}\n")
|
||||||
|
if op_id in results["ihh_isg_combined"]:
|
||||||
|
combined = results["ihh_isg_combined"][op_id]
|
||||||
|
template.append(f" * IHH: {combined['ihh_value']} - {combined['ihh_color']}\n")
|
||||||
|
template.append(f" * ISG combiné: {combined['isg_combined']:.0f} - {combined['isg_color']}\n")
|
||||||
|
|
||||||
|
template.append("\n")
|
||||||
|
else:
|
||||||
|
template.append("Aucune chaîne à risque majeur identifiée.\n")
|
||||||
|
|
||||||
|
# 3. Chaînes moyennes
|
||||||
|
template.append("### Chaînes avec risque moyen\n")
|
||||||
|
template.append("*Ces chaînes comprennent au moins une vulnérabilité combinée moyenne*\n")
|
||||||
|
|
||||||
|
if medium_chains:
|
||||||
|
for chain in medium_chains:
|
||||||
|
product_name = data["products"][chain["product"]]["label"]
|
||||||
|
component_name = data["components"][chain["component"]]["label"]
|
||||||
|
mineral_name = data["minerals"][chain["mineral"]]["label"]
|
||||||
|
|
||||||
|
template.append(f"#### {product_name} → {component_name} → {mineral_name}\n")
|
||||||
|
|
||||||
|
# Vulnérabilités
|
||||||
|
template.append("**Vulnérabilités identifiées:**\n")
|
||||||
|
for vuln in chain["vulnerabilities"]:
|
||||||
|
vuln_type = vuln["type"].capitalize()
|
||||||
|
vuln_level = vuln["vulnerability"]
|
||||||
|
|
||||||
|
if vuln_type == "Minerai":
|
||||||
|
mineral_id = vuln["mineral_id"]
|
||||||
|
template.append(f"* {vuln_type} ({mineral_name}): {vuln_level}\n")
|
||||||
|
if mineral_id in results["ics_ivc_combined"]:
|
||||||
|
combined = results["ics_ivc_combined"][mineral_id]
|
||||||
|
template.append(f" * ICS moyen: {combined['ics_average']:.2f} - {combined['ics_color']}\n")
|
||||||
|
template.append(f" * IVC: {combined['ivc_value']} - {combined['ivc_color']}\n")
|
||||||
|
else:
|
||||||
|
op_id = vuln["operation_id"]
|
||||||
|
op_label = data["operations"][op_id]["label"]
|
||||||
|
template.append(f"* {vuln_type} ({op_label}): {vuln_level}\n")
|
||||||
|
if op_id in results["ihh_isg_combined"]:
|
||||||
|
combined = results["ihh_isg_combined"][op_id]
|
||||||
|
template.append(f" * IHH: {combined['ihh_value']} - {combined['ihh_color']}\n")
|
||||||
|
template.append(f" * ISG combiné: {combined['isg_combined']:.0f} - {combined['isg_color']}\n")
|
||||||
|
|
||||||
|
template.append("\n")
|
||||||
|
else:
|
||||||
|
template.append("Aucune chaîne à risque moyen identifiée.\n")
|
||||||
|
|
||||||
|
return "\n".join(template)
|
||||||
|
|
||||||
|
def slugify(text):
|
||||||
|
return re.sub(r'\W+', '-', text.strip()).strip('-').lower()
|
||||||
|
|
||||||
|
def generate_report(data, results, config):
|
||||||
|
"""
|
||||||
|
Génère le rapport complet structuré selon les spécifications.
|
||||||
|
"""
|
||||||
|
# Titre principal
|
||||||
|
report_titre = ["# Évaluation des vulnérabilités critiques\n"]
|
||||||
|
|
||||||
|
# Section d'introduction
|
||||||
|
report_introduction = generate_introduction_section(data)
|
||||||
|
# report.append(generate_introduction_section(data))
|
||||||
|
|
||||||
|
# Section méthodologie
|
||||||
|
report_methodologie = generate_methodology_section()
|
||||||
|
# report.append(generate_methodology_section())
|
||||||
|
|
||||||
|
# Section détails des opérations
|
||||||
|
report_operations = generate_operations_section(data, results, config)
|
||||||
|
# report.append(generate_operations_section(data, results, config))
|
||||||
|
|
||||||
|
# Section détails des minerais
|
||||||
|
report_minerals = generate_minerals_section(data, results, config)
|
||||||
|
# report.append(generate_minerals_section(data, results, config))
|
||||||
|
|
||||||
|
# Section chemins critiques
|
||||||
|
report_critical_paths = generate_critical_paths_section(data, results)
|
||||||
|
|
||||||
|
suffixe = " - chemins critiques"
|
||||||
|
fichier = TEMPLATE_PATH.name.replace(".md", f"{suffixe}.md")
|
||||||
|
fichier_path = TEMPLATE_PATH.parent / fichier
|
||||||
|
# Élever les titres Markdown dans report_critical_paths
|
||||||
|
report_critical_paths = re.sub(r'^(#{2,})', lambda m: '#' * (len(m.group(1)) - 1), report_critical_paths, flags=re.MULTILINE)
|
||||||
|
write_report(report_critical_paths, fichier_path)
|
||||||
|
|
||||||
|
# Récupérer les sections critiques décomposées par mot-clé
|
||||||
|
chemins_critiques_sections = extraire_sections_par_mot_cle(fichier_path)
|
||||||
|
|
||||||
|
file_names = []
|
||||||
|
|
||||||
|
# Pour chaque mot-clé, écrire un fichier individuel
|
||||||
|
for mot_cle, contenu in chemins_critiques_sections.items():
|
||||||
|
print(mot_cle)
|
||||||
|
mot_cle_slug = slugify(mot_cle)
|
||||||
|
suffixe = f" - chemins critiques {mot_cle_slug}"
|
||||||
|
fichier_personnalise = TEMPLATE_PATH.with_name(
|
||||||
|
TEMPLATE_PATH.name.replace(".md", f"{suffixe}.md")
|
||||||
|
)
|
||||||
|
# Ajouter du texte au début du contenu
|
||||||
|
introduction = f"# Détail des chemins critiques pour : {mot_cle}\n\n"
|
||||||
|
contenu = introduction + contenu
|
||||||
|
write_report(contenu, fichier_personnalise)
|
||||||
|
file_names.append(fichier_personnalise)
|
||||||
|
# report.append(generate_critical_paths_section(data, results))
|
||||||
|
|
||||||
|
# Ordre de composition final
|
||||||
|
report = (
|
||||||
|
report_titre +
|
||||||
|
[report_introduction] +
|
||||||
|
[report_critical_paths] +
|
||||||
|
[report_operations] +
|
||||||
|
[report_minerals] +
|
||||||
|
[report_methodologie]
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(report), file_names
|
||||||
92
batch_ia/utils/sections_utils.py
Normal file
92
batch_ia/utils/sections_utils.py
Normal file
@ -0,0 +1,92 @@
|
|||||||
|
import os
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
from collections import defaultdict
|
||||||
|
|
||||||
|
from .config import (
|
||||||
|
CORPUS_DIR
|
||||||
|
)
|
||||||
|
|
||||||
|
def composant_match(nom_composant, nom_dossier):
|
||||||
|
"""
|
||||||
|
Vérifie si le nom du composant correspond approximativement à un nom de dossier (lettres et chiffres dans le même ordre).
|
||||||
|
"""
|
||||||
|
def clean(s):
|
||||||
|
return ''.join(c.lower() for c in s if c.isalnum())
|
||||||
|
|
||||||
|
cleaned_comp = clean(nom_composant)
|
||||||
|
cleaned_dir = clean(nom_dossier)
|
||||||
|
|
||||||
|
# Vérifie que chaque caractère de cleaned_comp est présent dans cleaned_dir dans le bon ordre
|
||||||
|
it = iter(cleaned_dir)
|
||||||
|
return all(c in it for c in cleaned_comp)
|
||||||
|
|
||||||
|
def trouver_dossier_composant(nom_composant, base_path, prefixe):
|
||||||
|
"""
|
||||||
|
Parcourt les sous-répertoires de base_path et retourne celui qui correspond au composant.
|
||||||
|
"""
|
||||||
|
search_path = os.path.join(CORPUS_DIR, base_path)
|
||||||
|
if not os.path.exists(search_path):
|
||||||
|
return None
|
||||||
|
|
||||||
|
for d in os.listdir(search_path):
|
||||||
|
if os.path.isdir(os.path.join(search_path, d)):
|
||||||
|
if composant_match(f"{prefixe}{nom_composant}", d):
|
||||||
|
return os.path.join(base_path, d)
|
||||||
|
return None
|
||||||
|
|
||||||
|
def extraire_sections_par_mot_cle(fichier_markdown: Path) -> dict:
|
||||||
|
"""
|
||||||
|
Extrait les sections de niveau 3 uniquement dans la section
|
||||||
|
'## Chaînes avec risque critique' du fichier Markdown,
|
||||||
|
et les regroupe par mot-clé (ce qui se trouve entre '### ' et ' →').
|
||||||
|
Réduit chaque titre d’un niveau (#).
|
||||||
|
"""
|
||||||
|
with fichier_markdown.open(encoding="utf-8") as f:
|
||||||
|
contenu = f.read()
|
||||||
|
|
||||||
|
# Extraire uniquement la section '## Chaînes avec risque critique'
|
||||||
|
match_section = re.search(
|
||||||
|
r"## Chaînes avec risque critique(.*?)(?=\n## |\Z)", contenu, re.DOTALL
|
||||||
|
)
|
||||||
|
if not match_section:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
section_critique = match_section.group(1)
|
||||||
|
|
||||||
|
# Extraire les mots-clés entre '### ' et ' →'
|
||||||
|
mots_cles = set(re.findall(r"^### (.+?) →", section_critique, re.MULTILINE))
|
||||||
|
|
||||||
|
# Extraire tous les blocs de niveau 3 dans cette section uniquement
|
||||||
|
blocs_sections = re.findall(r"(### .+?)(?=\n### |\n## |\Z)", section_critique, re.DOTALL)
|
||||||
|
|
||||||
|
# Regrouper les blocs par mot-clé
|
||||||
|
regroupement = defaultdict(list)
|
||||||
|
for bloc in blocs_sections:
|
||||||
|
match = re.match(r"### (.+?) →", bloc)
|
||||||
|
if match:
|
||||||
|
mot = match.group(1)
|
||||||
|
if mot in mots_cles:
|
||||||
|
# Réduction du niveau des titres
|
||||||
|
bloc_modifie = re.sub(r"^###", "##", bloc, flags=re.MULTILINE)
|
||||||
|
bloc_modifie = re.sub(r"^###", "##", bloc_modifie, flags=re.MULTILINE)
|
||||||
|
regroupement[mot].append(bloc_modifie)
|
||||||
|
|
||||||
|
return {mot: "\n\n".join(blocs) for mot, blocs in regroupement.items()}
|
||||||
|
|
||||||
|
def nettoyer_texte_fr(texte: str) -> str:
|
||||||
|
# Apostrophes droites -> typographiques
|
||||||
|
texte = texte.replace("'", "’")
|
||||||
|
# Guillemets droits -> guillemets français (avec espace fine insécable)
|
||||||
|
texte = re.sub(r'"(.*?)"', r'« \1 »', texte)
|
||||||
|
# Espaces fines insécables avant : ; ! ?
|
||||||
|
texte = re.sub(r' (?=[:;!?])', '\u202F', texte)
|
||||||
|
# Unités : espace insécable entre chiffre et unité
|
||||||
|
texte = re.sub(r'(\d) (?=\w+)', lambda m: f"{m.group(1)}\u202F", texte)
|
||||||
|
# Suppression des doubles espaces
|
||||||
|
texte = re.sub(r' {2,}', ' ', texte)
|
||||||
|
# Remplacement optionnel des tirets simples (optionnel)
|
||||||
|
texte = texte.replace(" - ", " – ")
|
||||||
|
# Nettoyage ponctuation multiple accidentelle
|
||||||
|
texte = re.sub(r'\s+([.,;!?])', r'\1', texte)
|
||||||
|
return texte
|
||||||
@ -10,11 +10,12 @@ Le module components comprend plusieurs fichiers clés :
|
|||||||
- **header.py** : Composant d'en-tête unifié pour toutes les pages
|
- **header.py** : Composant d'en-tête unifié pour toutes les pages
|
||||||
- **footer.py** : Pied de page standardisé incluant les mentions légales et informations de contact
|
- **footer.py** : Pied de page standardisé incluant les mentions légales et informations de contact
|
||||||
- **fiches.py** : Composants spécifiques à l'affichage et à la manipulation des fiches
|
- **fiches.py** : Composants spécifiques à l'affichage et à la manipulation des fiches
|
||||||
|
- **connexion.py** : Composants spécifiques à la gestion de la connexion / déconnexion sur la base d'un token Gitea
|
||||||
|
|
||||||
## Fonctionnalités
|
## Fonctionnalités
|
||||||
|
|
||||||
### Barre latérale (sidebar.py)
|
### Barre latérale (sidebar.py)
|
||||||
- Menu de navigation principal entre les différentes sections
|
- Menu de navigation prin>>>>>>>>>>>>> cipal entre les différentes sections
|
||||||
- Options de configuration et de personnalisation
|
- Options de configuration et de personnalisation
|
||||||
- Affichage des informations sur l'impact environnemental
|
- Affichage des informations sur l'impact environnemental
|
||||||
- Gestion du thème (clair/sombre)
|
- Gestion du thème (clair/sombre)
|
||||||
|
|||||||
@ -42,9 +42,10 @@ def connexion():
|
|||||||
with st.form("auth_form"):
|
with st.form("auth_form"):
|
||||||
# Ajout d'un champ identifiant fictif pour activer l'autocomplétion navigateur
|
# Ajout d'un champ identifiant fictif pour activer l'autocomplétion navigateur
|
||||||
# et permettre de stocker le token comme un mot de passe par le navigateur
|
# et permettre de stocker le token comme un mot de passe par le navigateur
|
||||||
|
# L'identifiant n'est donc pas utilisé par la suite ; il est caché en CSS
|
||||||
identifiant = st.text_input(str(_("auth.username")), value="fabnum-connexion", key="nom_utilisateur")
|
identifiant = st.text_input(str(_("auth.username")), value="fabnum-connexion", key="nom_utilisateur")
|
||||||
token = st.text_input(str(_("auth.token")), type="password")
|
token = st.text_input(str(_("auth.token")), type="password")
|
||||||
submitted = st.form_submit_button(str(_("auth.login")))
|
submitted = st.form_submit_button(str(_("auth.login")), icon=":material/login:")
|
||||||
|
|
||||||
if submitted and token:
|
if submitted and token:
|
||||||
erreur = True
|
erreur = True
|
||||||
@ -100,7 +101,7 @@ def bouton_deconnexion():
|
|||||||
""")
|
""")
|
||||||
|
|
||||||
st.sidebar.markdown(f"{str(_('auth.logged_as'))} `{st.session_state.username}`")
|
st.sidebar.markdown(f"{str(_('auth.logged_as'))} `{st.session_state.username}`")
|
||||||
if st.sidebar.button(str(_("auth.logout"))):
|
if st.sidebar.button(str(_("auth.logout")), icon=":material/logout:"):
|
||||||
st.session_state.logged_in = False
|
st.session_state.logged_in = False
|
||||||
st.session_state.username = ""
|
st.session_state.username = ""
|
||||||
st.session_state.token = ""
|
st.session_state.token = ""
|
||||||
|
|||||||
@ -21,6 +21,8 @@ def afficher_menu():
|
|||||||
str(_("navigation.instructions")),
|
str(_("navigation.instructions")),
|
||||||
str(_("navigation.personnalisation")),
|
str(_("navigation.personnalisation")),
|
||||||
str(_("navigation.analyse")),
|
str(_("navigation.analyse")),
|
||||||
|
*([str(_("navigation.ia_nalyse"))] if st.session_state.get("logged_in", False) else []),
|
||||||
|
*([str(_("navigation.plan_d_action"))]),
|
||||||
str(_("navigation.visualisations")),
|
str(_("navigation.visualisations")),
|
||||||
str(_("navigation.fiches"))
|
str(_("navigation.fiches"))
|
||||||
]
|
]
|
||||||
@ -99,7 +101,12 @@ def afficher_menu():
|
|||||||
st.session_state.theme_mode = theme
|
st.session_state.theme_mode = theme
|
||||||
st.rerun()
|
st.rerun()
|
||||||
|
|
||||||
|
#
|
||||||
|
# Important :
|
||||||
|
# Avec Selinux, il faut donner les bons droits
|
||||||
|
#
|
||||||
|
# sudo chcon -Rt httpd_sys_content_t /chemin/d/acces/assets/
|
||||||
|
#
|
||||||
def afficher_impact(total_bytes):
|
def afficher_impact(total_bytes):
|
||||||
impact_label = str(_("sidebar.impact"))
|
impact_label = str(_("sidebar.impact"))
|
||||||
loading_text = str(_("sidebar.loading"))
|
loading_text = str(_("sidebar.loading"))
|
||||||
|
|||||||
15
fabnum-dev.service
Normal file
15
fabnum-dev.service
Normal file
@ -0,0 +1,15 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Fabnum Dev - Streamlit App
|
||||||
|
After=network.target
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
User=fabnum
|
||||||
|
WorkingDirectory=/home/fabnum/fabnum-dev
|
||||||
|
ExecStart=/home/fabnum/fabnum-dev/venv/bin/streamlit run /home/fabnum/fabnum-dev
|
||||||
|
/fabnum.py --server.port 8502
|
||||||
|
Restart=always
|
||||||
|
RestartSec=5
|
||||||
|
Environment=PYTHONUNBUFFERED=1
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.targe
|
||||||
24
fabnum.py
24
fabnum.py
@ -19,6 +19,8 @@ from utils.gitea import (
|
|||||||
# Import du module de traductions
|
# Import du module de traductions
|
||||||
from utils.translations import init_translations, _, set_language
|
from utils.translations import init_translations, _, set_language
|
||||||
|
|
||||||
|
from utils.widgets import html_expander
|
||||||
|
|
||||||
def afficher_instructions_avec_expanders(markdown_content):
|
def afficher_instructions_avec_expanders(markdown_content):
|
||||||
"""
|
"""
|
||||||
Affiche le contenu markdown avec les sections de niveau 2 (## Titre) dans des expanders
|
Affiche le contenu markdown avec les sections de niveau 2 (## Titre) dans des expanders
|
||||||
@ -57,8 +59,8 @@ def afficher_instructions_avec_expanders(markdown_content):
|
|||||||
|
|
||||||
# Affichage dans un expander
|
# Affichage dans un expander
|
||||||
status = True if i == 1 else False
|
status = True if i == 1 else False
|
||||||
with st.expander(f"## {titre_section}", expanded=status):
|
# with st.expander(f"## {titre_section}", expanded=status):
|
||||||
st.markdown(contenu_section, unsafe_allow_html=True)
|
html_expander(f"{titre_section}", content=contenu_section, open_by_default=status, details_class="details_introduction")
|
||||||
|
|
||||||
from utils.graph_utils import (
|
from utils.graph_utils import (
|
||||||
charger_graphe
|
charger_graphe
|
||||||
@ -76,6 +78,8 @@ from app.fiches import interface_fiches
|
|||||||
from app.visualisations import interface_visualisations
|
from app.visualisations import interface_visualisations
|
||||||
from app.personnalisation import interface_personnalisation
|
from app.personnalisation import interface_personnalisation
|
||||||
from app.analyse import interface_analyse
|
from app.analyse import interface_analyse
|
||||||
|
from app.ia_nalyse import interface_ia_nalyse
|
||||||
|
from app.plan_d_action import interface_plan_d_action
|
||||||
|
|
||||||
# Initialisation des traductions (langue française par défaut)
|
# Initialisation des traductions (langue française par défaut)
|
||||||
init_translations()
|
init_translations()
|
||||||
@ -166,6 +170,8 @@ instructions_tab = _("navigation.instructions")
|
|||||||
fiches_tab = _("navigation.fiches")
|
fiches_tab = _("navigation.fiches")
|
||||||
personnalisation_tab = _("navigation.personnalisation")
|
personnalisation_tab = _("navigation.personnalisation")
|
||||||
analyse_tab = _("navigation.analyse")
|
analyse_tab = _("navigation.analyse")
|
||||||
|
ia_nalyse_tab = _("navigation.ia_nalyse")
|
||||||
|
plan_d_action_tab = _("navigation.plan_d_action")
|
||||||
visualisations_tab = _("navigation.visualisations")
|
visualisations_tab = _("navigation.visualisations")
|
||||||
|
|
||||||
if st.session_state.onglet == instructions_tab:
|
if st.session_state.onglet == instructions_tab:
|
||||||
@ -179,15 +185,27 @@ elif st.session_state.onglet == fiches_tab:
|
|||||||
else:
|
else:
|
||||||
# Charger le graphe une seule fois
|
# Charger le graphe une seule fois
|
||||||
# Le graphe n'est pas nécessaire pour Instructions ou Fiches
|
# Le graphe n'est pas nécessaire pour Instructions ou Fiches
|
||||||
G_temp, G_temp_ivc, dot_file_path = charger_graphe()
|
dot_file_path = charger_graphe()
|
||||||
|
|
||||||
if dot_file_path and st.session_state.onglet == analyse_tab:
|
if dot_file_path and st.session_state.onglet == analyse_tab:
|
||||||
|
G_temp = st.session_state["G_temp"]
|
||||||
interface_analyse(G_temp)
|
interface_analyse(G_temp)
|
||||||
|
|
||||||
|
elif dot_file_path and st.session_state.onglet == ia_nalyse_tab:
|
||||||
|
G_temp = st.session_state["G_temp"]
|
||||||
|
interface_ia_nalyse(G_temp)
|
||||||
|
|
||||||
|
elif dot_file_path and st.session_state.onglet == plan_d_action_tab:
|
||||||
|
G_temp = st.session_state["G_temp"]
|
||||||
|
interface_plan_d_action(G_temp)
|
||||||
|
|
||||||
elif dot_file_path and st.session_state.onglet == visualisations_tab:
|
elif dot_file_path and st.session_state.onglet == visualisations_tab:
|
||||||
|
G_temp = st.session_state["G_temp"]
|
||||||
|
G_temp_ivc = st.session_state["G_temp_ivc"]
|
||||||
interface_visualisations(G_temp, G_temp_ivc)
|
interface_visualisations(G_temp, G_temp_ivc)
|
||||||
|
|
||||||
elif dot_file_path and st.session_state.onglet == personnalisation_tab:
|
elif dot_file_path and st.session_state.onglet == personnalisation_tab:
|
||||||
|
G_temp = st.session_state["G_temp"]
|
||||||
G_temp = interface_personnalisation(G_temp)
|
G_temp = interface_personnalisation(G_temp)
|
||||||
|
|
||||||
fermer_page()
|
fermer_page()
|
||||||
|
|||||||
16
pgpt/.docker/router.yml
Normal file
16
pgpt/.docker/router.yml
Normal file
@ -0,0 +1,16 @@
|
|||||||
|
http:
|
||||||
|
services:
|
||||||
|
ollama:
|
||||||
|
loadBalancer:
|
||||||
|
healthCheck:
|
||||||
|
interval: 5s
|
||||||
|
path: /
|
||||||
|
servers:
|
||||||
|
- url: http://ollama-cpu:11434
|
||||||
|
- url: http://ollama-cuda:11434
|
||||||
|
- url: http://host.docker.internal:11434
|
||||||
|
|
||||||
|
routers:
|
||||||
|
ollama-router:
|
||||||
|
rule: "PathPrefix(`/`)"
|
||||||
|
service: ollama
|
||||||
8
pgpt/.gitignore
vendored
Normal file
8
pgpt/.gitignore
vendored
Normal file
@ -0,0 +1,8 @@
|
|||||||
|
local_data/*
|
||||||
|
qdrant_data/*
|
||||||
|
models/
|
||||||
|
qdrant_data/
|
||||||
|
poetry.lock
|
||||||
|
.dockerignore
|
||||||
|
CITATION.cff
|
||||||
|
scripts
|
||||||
51
pgpt/Dockerfile.ollama
Normal file
51
pgpt/Dockerfile.ollama
Normal file
@ -0,0 +1,51 @@
|
|||||||
|
FROM python:3.11.6-slim-bookworm AS base
|
||||||
|
|
||||||
|
# Install poetry
|
||||||
|
RUN pip install pipx
|
||||||
|
RUN python3 -m pipx ensurepath
|
||||||
|
RUN pipx install poetry==1.8.3
|
||||||
|
ENV PATH="/root/.local/bin:$PATH"
|
||||||
|
ENV PATH=".venv/bin/:$PATH"
|
||||||
|
|
||||||
|
# https://python-poetry.org/docs/configuration/#virtualenvsin-project
|
||||||
|
ENV POETRY_VIRTUALENVS_IN_PROJECT=true
|
||||||
|
|
||||||
|
FROM base AS dependencies
|
||||||
|
WORKDIR /home/worker/app
|
||||||
|
COPY pyproject.toml poetry.lock ./
|
||||||
|
|
||||||
|
ARG POETRY_EXTRAS="ui vector-stores-qdrant llms-ollama embeddings-ollama"
|
||||||
|
RUN poetry install --no-root --extras "${POETRY_EXTRAS}"
|
||||||
|
|
||||||
|
FROM base AS app
|
||||||
|
ENV PYTHONUNBUFFERED=1
|
||||||
|
ENV PORT=8080
|
||||||
|
ENV APP_ENV=prod
|
||||||
|
ENV PYTHONPATH="$PYTHONPATH:/home/worker/app/private_gpt/"
|
||||||
|
EXPOSE 8080
|
||||||
|
|
||||||
|
# Prepare a non-root user
|
||||||
|
# More info about how to configure UIDs and GIDs in Docker:
|
||||||
|
# https://github.com/systemd/systemd/blob/main/docs/UIDS-GIDS.md
|
||||||
|
|
||||||
|
# Define the User ID (UID) for the non-root user
|
||||||
|
# UID 100 is chosen to avoid conflicts with existing system users
|
||||||
|
ARG UID=100
|
||||||
|
|
||||||
|
# Define the Group ID (GID) for the non-root user
|
||||||
|
# GID 65534 is often used for the 'nogroup' or 'nobody' group
|
||||||
|
ARG GID=65534
|
||||||
|
|
||||||
|
RUN adduser --system --gid ${GID} --uid ${UID} --home /home/worker worker
|
||||||
|
WORKDIR /home/worker/app
|
||||||
|
|
||||||
|
RUN chown worker /home/worker/app
|
||||||
|
RUN mkdir local_data && chown worker local_data
|
||||||
|
RUN mkdir models && chown worker models
|
||||||
|
COPY --chown=worker --from=dependencies /home/worker/app/.venv/ .venv
|
||||||
|
COPY --chown=worker private_gpt/ private_gpt
|
||||||
|
COPY --chown=worker *.yaml .
|
||||||
|
COPY --chown=worker scripts/ scripts
|
||||||
|
|
||||||
|
USER worker
|
||||||
|
ENTRYPOINT python -m private_gpt
|
||||||
121
pgpt/docker-compose.yaml
Normal file
121
pgpt/docker-compose.yaml
Normal file
@ -0,0 +1,121 @@
|
|||||||
|
services:
|
||||||
|
#-----------------------------------
|
||||||
|
#---- Private-GPT services ---------
|
||||||
|
#-----------------------------------
|
||||||
|
|
||||||
|
# Private-GPT service for the Ollama CPU and GPU modes
|
||||||
|
# This service builds from an external Dockerfile and runs the Ollama mode.
|
||||||
|
private-gpt-ollama:
|
||||||
|
image: ${PGPT_IMAGE:-zylonai/private-gpt}:${PGPT_TAG:-0.6.2}-ollama # x-release-please-version
|
||||||
|
user: root
|
||||||
|
build:
|
||||||
|
context: .
|
||||||
|
dockerfile: Dockerfile.ollama
|
||||||
|
volumes:
|
||||||
|
- /home/fabnum/fabnum-dev/Fiches:/home/worker/app/local_data/Fiches:Z
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8001:8001"
|
||||||
|
environment:
|
||||||
|
PORT: 8001
|
||||||
|
PGPT_PROFILES: docker
|
||||||
|
PGPT_MODE: ollama
|
||||||
|
PGPT_EMBED_MODE: ollama
|
||||||
|
PGPT_OLLAMA_API_BASE: http://ollama:11434
|
||||||
|
HF_TOKEN: ${HF_TOKEN:-}
|
||||||
|
profiles:
|
||||||
|
- ""
|
||||||
|
- ollama-cpu
|
||||||
|
- ollama-cuda
|
||||||
|
- ollama-api
|
||||||
|
depends_on:
|
||||||
|
ollama:
|
||||||
|
condition: service_healthy
|
||||||
|
|
||||||
|
# Private-GPT service for the local mode
|
||||||
|
# This service builds from a local Dockerfile and runs the application in local mode.
|
||||||
|
private-gpt-llamacpp-cpu:
|
||||||
|
image: ${PGPT_IMAGE:-zylonai/private-gpt}:${PGPT_TAG:-0.6.2}-llamacpp-cpu # x-release-please-version
|
||||||
|
user: root
|
||||||
|
build:
|
||||||
|
context: .
|
||||||
|
dockerfile: Dockerfile.llamacpp-cpu
|
||||||
|
volumes:
|
||||||
|
- ./local_data/:/home/worker/app/local_data
|
||||||
|
- ./models/:/home/worker/app/models
|
||||||
|
entrypoint: sh -c ".venv/bin/python scripts/setup && .venv/bin/python -m private_gpt"
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8001:8001"
|
||||||
|
environment:
|
||||||
|
PORT: 8001
|
||||||
|
PGPT_PROFILES: local
|
||||||
|
HF_TOKEN: ${HF_TOKEN:-}
|
||||||
|
profiles:
|
||||||
|
- llamacpp-cpu
|
||||||
|
|
||||||
|
#-----------------------------------
|
||||||
|
#---- Ollama services --------------
|
||||||
|
#-----------------------------------
|
||||||
|
|
||||||
|
# Traefik reverse proxy for the Ollama service
|
||||||
|
# This will route requests to the Ollama service based on the profile.
|
||||||
|
ollama:
|
||||||
|
image: traefik:v2.10
|
||||||
|
healthcheck:
|
||||||
|
test:
|
||||||
|
[
|
||||||
|
"CMD",
|
||||||
|
"sh",
|
||||||
|
"-c",
|
||||||
|
"wget -q --spider http://ollama:11434 || exit 1",
|
||||||
|
]
|
||||||
|
interval: 10s
|
||||||
|
retries: 3
|
||||||
|
start_period: 5s
|
||||||
|
timeout: 5s
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8080:8080"
|
||||||
|
command:
|
||||||
|
- "--providers.file.filename=/etc/router.yml"
|
||||||
|
- "--log.level=ERROR"
|
||||||
|
- "--api.insecure=true"
|
||||||
|
- "--providers.docker=true"
|
||||||
|
- "--providers.docker.exposedbydefault=false"
|
||||||
|
- "--entrypoints.web.address=:11434"
|
||||||
|
volumes:
|
||||||
|
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||||
|
- ./.docker/router.yml:/etc/router.yml:ro
|
||||||
|
extra_hosts:
|
||||||
|
- "host.docker.internal:host-gateway"
|
||||||
|
profiles:
|
||||||
|
- ""
|
||||||
|
- ollama-cpu
|
||||||
|
- ollama-cuda
|
||||||
|
- ollama-api
|
||||||
|
|
||||||
|
# Ollama service for the CPU mode
|
||||||
|
ollama-cpu:
|
||||||
|
image: ollama/ollama:latest
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:11434:11434"
|
||||||
|
volumes:
|
||||||
|
- ./models:/root/.ollama:Z
|
||||||
|
profiles:
|
||||||
|
- ""
|
||||||
|
- ollama-cpu
|
||||||
|
|
||||||
|
# Ollama service for the CUDA mode
|
||||||
|
ollama-cuda:
|
||||||
|
image: ollama/ollama:latest
|
||||||
|
ports:
|
||||||
|
- "11434:11434"
|
||||||
|
volumes:
|
||||||
|
- ./models:/root/.ollama
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
reservations:
|
||||||
|
devices:
|
||||||
|
- driver: nvidia
|
||||||
|
count: 1
|
||||||
|
capabilities: [gpu]
|
||||||
|
profiles:
|
||||||
|
- ollama-cuda
|
||||||
27
pgpt/private_gpt/__init__.py
Normal file
27
pgpt/private_gpt/__init__.py
Normal file
@ -0,0 +1,27 @@
|
|||||||
|
"""private-gpt."""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
|
||||||
|
# Set to 'DEBUG' to have extensive logging turned on, even for libraries
|
||||||
|
ROOT_LOG_LEVEL = "INFO"
|
||||||
|
|
||||||
|
PRETTY_LOG_FORMAT = (
|
||||||
|
"%(asctime)s.%(msecs)03d [%(levelname)-8s] %(name)+25s - %(message)s"
|
||||||
|
)
|
||||||
|
logging.basicConfig(level=ROOT_LOG_LEVEL, format=PRETTY_LOG_FORMAT, datefmt="%H:%M:%S")
|
||||||
|
logging.captureWarnings(True)
|
||||||
|
|
||||||
|
# Disable gradio analytics
|
||||||
|
# This is done this way because gradio does not solely rely on what values are
|
||||||
|
# passed to gr.Blocks(enable_analytics=...) but also on the environment
|
||||||
|
# variable GRADIO_ANALYTICS_ENABLED. `gradio.strings` actually reads this env
|
||||||
|
# directly, so to fully disable gradio analytics we need to set this env var.
|
||||||
|
os.environ["GRADIO_ANALYTICS_ENABLED"] = "False"
|
||||||
|
|
||||||
|
# Disable chromaDB telemetry
|
||||||
|
# It is already disabled, see PR#1144
|
||||||
|
# os.environ["ANONYMIZED_TELEMETRY"] = "False"
|
||||||
|
|
||||||
|
# adding tiktoken cache path within repo to be able to run in offline environment.
|
||||||
|
os.environ["TIKTOKEN_CACHE_DIR"] = "tiktoken_cache"
|
||||||
11
pgpt/private_gpt/__main__.py
Normal file
11
pgpt/private_gpt/__main__.py
Normal file
@ -0,0 +1,11 @@
|
|||||||
|
# start a fastapi server with uvicorn
|
||||||
|
|
||||||
|
import uvicorn
|
||||||
|
|
||||||
|
from private_gpt.main import app
|
||||||
|
from private_gpt.settings.settings import settings
|
||||||
|
|
||||||
|
# Set log_config=None to do not use the uvicorn logging configuration, and
|
||||||
|
# use ours instead. For reference, see below:
|
||||||
|
# https://github.com/tiangolo/fastapi/discussions/7457#discussioncomment-5141108
|
||||||
|
uvicorn.run(app, host="0.0.0.0", port=settings().server.port, log_config=None)
|
||||||
0
pgpt/private_gpt/components/__init__.py
Normal file
0
pgpt/private_gpt/components/__init__.py
Normal file
0
pgpt/private_gpt/components/embedding/__init__.py
Normal file
0
pgpt/private_gpt/components/embedding/__init__.py
Normal file
82
pgpt/private_gpt/components/embedding/custom/sagemaker.py
Normal file
82
pgpt/private_gpt/components/embedding/custom/sagemaker.py
Normal file
@ -0,0 +1,82 @@
|
|||||||
|
# mypy: ignore-errors
|
||||||
|
import json
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import boto3
|
||||||
|
from llama_index.core.base.embeddings.base import BaseEmbedding
|
||||||
|
from pydantic import Field, PrivateAttr
|
||||||
|
|
||||||
|
|
||||||
|
class SagemakerEmbedding(BaseEmbedding):
|
||||||
|
"""Sagemaker Embedding Endpoint.
|
||||||
|
|
||||||
|
To use, you must supply the endpoint name from your deployed
|
||||||
|
Sagemaker embedding model & the region where it is deployed.
|
||||||
|
|
||||||
|
To authenticate, the AWS client uses the following methods to
|
||||||
|
automatically load credentials:
|
||||||
|
https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html
|
||||||
|
|
||||||
|
If a specific credential profile should be used, you must pass
|
||||||
|
the name of the profile from the ~/.aws/credentials file that is to be used.
|
||||||
|
|
||||||
|
Make sure the credentials / roles used have the required policies to
|
||||||
|
access the Sagemaker endpoint.
|
||||||
|
See: https://docs.aws.amazon.com/IAM/latest/UserGuide/access_policies.html
|
||||||
|
"""
|
||||||
|
|
||||||
|
endpoint_name: str = Field(description="")
|
||||||
|
|
||||||
|
_boto_client: Any = boto3.client(
|
||||||
|
"sagemaker-runtime",
|
||||||
|
) # TODO make it an optional field
|
||||||
|
|
||||||
|
_async_not_implemented_warned: bool = PrivateAttr(default=False)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def class_name(cls) -> str:
|
||||||
|
return "SagemakerEmbedding"
|
||||||
|
|
||||||
|
def _async_not_implemented_warn_once(self) -> None:
|
||||||
|
if not self._async_not_implemented_warned:
|
||||||
|
print("Async embedding not available, falling back to sync method.")
|
||||||
|
self._async_not_implemented_warned = True
|
||||||
|
|
||||||
|
def _embed(self, sentences: list[str]) -> list[list[float]]:
|
||||||
|
request_params = {
|
||||||
|
"inputs": sentences,
|
||||||
|
}
|
||||||
|
|
||||||
|
resp = self._boto_client.invoke_endpoint(
|
||||||
|
EndpointName=self.endpoint_name,
|
||||||
|
Body=json.dumps(request_params),
|
||||||
|
ContentType="application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
response_body = resp["Body"]
|
||||||
|
response_str = response_body.read().decode("utf-8")
|
||||||
|
response_json = json.loads(response_str)
|
||||||
|
|
||||||
|
return response_json["vectors"]
|
||||||
|
|
||||||
|
def _get_query_embedding(self, query: str) -> list[float]:
|
||||||
|
"""Get query embedding."""
|
||||||
|
return self._embed([query])[0]
|
||||||
|
|
||||||
|
async def _aget_query_embedding(self, query: str) -> list[float]:
|
||||||
|
# Warn the user that sync is being used
|
||||||
|
self._async_not_implemented_warn_once()
|
||||||
|
return self._get_query_embedding(query)
|
||||||
|
|
||||||
|
async def _aget_text_embedding(self, text: str) -> list[float]:
|
||||||
|
# Warn the user that sync is being used
|
||||||
|
self._async_not_implemented_warn_once()
|
||||||
|
return self._get_text_embedding(text)
|
||||||
|
|
||||||
|
def _get_text_embedding(self, text: str) -> list[float]:
|
||||||
|
"""Get text embedding."""
|
||||||
|
return self._embed([text])[0]
|
||||||
|
|
||||||
|
def _get_text_embeddings(self, texts: list[str]) -> list[list[float]]:
|
||||||
|
"""Get text embeddings."""
|
||||||
|
return self._embed(texts)
|
||||||
167
pgpt/private_gpt/components/embedding/embedding_component.py
Normal file
167
pgpt/private_gpt/components/embedding/embedding_component.py
Normal file
@ -0,0 +1,167 @@
|
|||||||
|
import logging
|
||||||
|
|
||||||
|
from injector import inject, singleton
|
||||||
|
from llama_index.core.embeddings import BaseEmbedding, MockEmbedding
|
||||||
|
|
||||||
|
from private_gpt.paths import models_cache_path
|
||||||
|
from private_gpt.settings.settings import Settings
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@singleton
|
||||||
|
class EmbeddingComponent:
|
||||||
|
embedding_model: BaseEmbedding
|
||||||
|
|
||||||
|
@inject
|
||||||
|
def __init__(self, settings: Settings) -> None:
|
||||||
|
embedding_mode = settings.embedding.mode
|
||||||
|
logger.info("Initializing the embedding model in mode=%s", embedding_mode)
|
||||||
|
match embedding_mode:
|
||||||
|
case "huggingface":
|
||||||
|
try:
|
||||||
|
from llama_index.embeddings.huggingface import ( # type: ignore
|
||||||
|
HuggingFaceEmbedding,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Local dependencies not found, install with `poetry install --extras embeddings-huggingface`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
self.embedding_model = HuggingFaceEmbedding(
|
||||||
|
model_name=settings.huggingface.embedding_hf_model_name,
|
||||||
|
cache_folder=str(models_cache_path),
|
||||||
|
trust_remote_code=settings.huggingface.trust_remote_code,
|
||||||
|
)
|
||||||
|
case "sagemaker":
|
||||||
|
try:
|
||||||
|
from private_gpt.components.embedding.custom.sagemaker import (
|
||||||
|
SagemakerEmbedding,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Sagemaker dependencies not found, install with `poetry install --extras embeddings-sagemaker`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
self.embedding_model = SagemakerEmbedding(
|
||||||
|
endpoint_name=settings.sagemaker.embedding_endpoint_name,
|
||||||
|
)
|
||||||
|
case "openai":
|
||||||
|
try:
|
||||||
|
from llama_index.embeddings.openai import ( # type: ignore
|
||||||
|
OpenAIEmbedding,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"OpenAI dependencies not found, install with `poetry install --extras embeddings-openai`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
api_base = (
|
||||||
|
settings.openai.embedding_api_base or settings.openai.api_base
|
||||||
|
)
|
||||||
|
api_key = settings.openai.embedding_api_key or settings.openai.api_key
|
||||||
|
model = settings.openai.embedding_model
|
||||||
|
|
||||||
|
self.embedding_model = OpenAIEmbedding(
|
||||||
|
api_base=api_base,
|
||||||
|
api_key=api_key,
|
||||||
|
model=model,
|
||||||
|
)
|
||||||
|
case "ollama":
|
||||||
|
try:
|
||||||
|
from llama_index.embeddings.ollama import ( # type: ignore
|
||||||
|
OllamaEmbedding,
|
||||||
|
)
|
||||||
|
from ollama import Client # type: ignore
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Local dependencies not found, install with `poetry install --extras embeddings-ollama`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
ollama_settings = settings.ollama
|
||||||
|
|
||||||
|
# Calculate embedding model. If not provided tag, it will be use latest
|
||||||
|
model_name = (
|
||||||
|
ollama_settings.embedding_model + ":latest"
|
||||||
|
if ":" not in ollama_settings.embedding_model
|
||||||
|
else ollama_settings.embedding_model
|
||||||
|
)
|
||||||
|
|
||||||
|
self.embedding_model = OllamaEmbedding(
|
||||||
|
model_name=model_name,
|
||||||
|
base_url=ollama_settings.embedding_api_base,
|
||||||
|
)
|
||||||
|
|
||||||
|
if ollama_settings.autopull_models:
|
||||||
|
if ollama_settings.autopull_models:
|
||||||
|
from private_gpt.utils.ollama import (
|
||||||
|
check_connection,
|
||||||
|
pull_model,
|
||||||
|
)
|
||||||
|
|
||||||
|
# TODO: Reuse llama-index client when llama-index is updated
|
||||||
|
client = Client(
|
||||||
|
host=ollama_settings.embedding_api_base,
|
||||||
|
timeout=ollama_settings.request_timeout,
|
||||||
|
)
|
||||||
|
|
||||||
|
if not check_connection(client):
|
||||||
|
raise ValueError(
|
||||||
|
f"Failed to connect to Ollama, "
|
||||||
|
f"check if Ollama server is running on {ollama_settings.api_base}"
|
||||||
|
)
|
||||||
|
pull_model(client, model_name)
|
||||||
|
|
||||||
|
case "azopenai":
|
||||||
|
try:
|
||||||
|
from llama_index.embeddings.azure_openai import ( # type: ignore
|
||||||
|
AzureOpenAIEmbedding,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Azure OpenAI dependencies not found, install with `poetry install --extras embeddings-azopenai`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
azopenai_settings = settings.azopenai
|
||||||
|
self.embedding_model = AzureOpenAIEmbedding(
|
||||||
|
model=azopenai_settings.embedding_model,
|
||||||
|
deployment_name=azopenai_settings.embedding_deployment_name,
|
||||||
|
api_key=azopenai_settings.api_key,
|
||||||
|
azure_endpoint=azopenai_settings.azure_endpoint,
|
||||||
|
api_version=azopenai_settings.api_version,
|
||||||
|
)
|
||||||
|
case "gemini":
|
||||||
|
try:
|
||||||
|
from llama_index.embeddings.gemini import ( # type: ignore
|
||||||
|
GeminiEmbedding,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Gemini dependencies not found, install with `poetry install --extras embeddings-gemini`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
self.embedding_model = GeminiEmbedding(
|
||||||
|
api_key=settings.gemini.api_key,
|
||||||
|
model_name=settings.gemini.embedding_model,
|
||||||
|
)
|
||||||
|
case "mistralai":
|
||||||
|
try:
|
||||||
|
from llama_index.embeddings.mistralai import ( # type: ignore
|
||||||
|
MistralAIEmbedding,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Mistral dependencies not found, install with `poetry install --extras embeddings-mistral`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
api_key = settings.openai.api_key
|
||||||
|
model = settings.openai.embedding_model
|
||||||
|
|
||||||
|
self.embedding_model = MistralAIEmbedding(
|
||||||
|
api_key=api_key,
|
||||||
|
model=model,
|
||||||
|
)
|
||||||
|
case "mock":
|
||||||
|
# Not a random number, is the dimensionality used by
|
||||||
|
# the default embedding model
|
||||||
|
self.embedding_model = MockEmbedding(384)
|
||||||
0
pgpt/private_gpt/components/ingest/__init__.py
Normal file
0
pgpt/private_gpt/components/ingest/__init__.py
Normal file
517
pgpt/private_gpt/components/ingest/ingest_component.py
Normal file
517
pgpt/private_gpt/components/ingest/ingest_component.py
Normal file
@ -0,0 +1,517 @@
|
|||||||
|
import abc
|
||||||
|
import itertools
|
||||||
|
import logging
|
||||||
|
import multiprocessing
|
||||||
|
import multiprocessing.pool
|
||||||
|
import os
|
||||||
|
import threading
|
||||||
|
from pathlib import Path
|
||||||
|
from queue import Queue
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from llama_index.core.data_structs import IndexDict
|
||||||
|
from llama_index.core.embeddings.utils import EmbedType
|
||||||
|
from llama_index.core.indices import VectorStoreIndex, load_index_from_storage
|
||||||
|
from llama_index.core.indices.base import BaseIndex
|
||||||
|
from llama_index.core.ingestion import run_transformations
|
||||||
|
from llama_index.core.schema import BaseNode, Document, TransformComponent
|
||||||
|
from llama_index.core.storage import StorageContext
|
||||||
|
|
||||||
|
from private_gpt.components.ingest.ingest_helper import IngestionHelper
|
||||||
|
from private_gpt.paths import local_data_path
|
||||||
|
from private_gpt.settings.settings import Settings
|
||||||
|
from private_gpt.utils.eta import eta
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class BaseIngestComponent(abc.ABC):
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
storage_context: StorageContext,
|
||||||
|
embed_model: EmbedType,
|
||||||
|
transformations: list[TransformComponent],
|
||||||
|
*args: Any,
|
||||||
|
**kwargs: Any,
|
||||||
|
) -> None:
|
||||||
|
logger.debug("Initializing base ingest component type=%s", type(self).__name__)
|
||||||
|
self.storage_context = storage_context
|
||||||
|
self.embed_model = embed_model
|
||||||
|
self.transformations = transformations
|
||||||
|
|
||||||
|
@abc.abstractmethod
|
||||||
|
def ingest(self, file_name: str, file_data: Path) -> list[Document]:
|
||||||
|
pass
|
||||||
|
|
||||||
|
@abc.abstractmethod
|
||||||
|
def bulk_ingest(self, files: list[tuple[str, Path]]) -> list[Document]:
|
||||||
|
pass
|
||||||
|
|
||||||
|
@abc.abstractmethod
|
||||||
|
def delete(self, doc_id: str) -> None:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class BaseIngestComponentWithIndex(BaseIngestComponent, abc.ABC):
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
storage_context: StorageContext,
|
||||||
|
embed_model: EmbedType,
|
||||||
|
transformations: list[TransformComponent],
|
||||||
|
*args: Any,
|
||||||
|
**kwargs: Any,
|
||||||
|
) -> None:
|
||||||
|
super().__init__(storage_context, embed_model, transformations, *args, **kwargs)
|
||||||
|
|
||||||
|
self.show_progress = True
|
||||||
|
self._index_thread_lock = (
|
||||||
|
threading.Lock()
|
||||||
|
) # Thread lock! Not Multiprocessing lock
|
||||||
|
self._index = self._initialize_index()
|
||||||
|
|
||||||
|
def _initialize_index(self) -> BaseIndex[IndexDict]:
|
||||||
|
"""Initialize the index from the storage context."""
|
||||||
|
try:
|
||||||
|
# Load the index with store_nodes_override=True to be able to delete them
|
||||||
|
index = load_index_from_storage(
|
||||||
|
storage_context=self.storage_context,
|
||||||
|
store_nodes_override=True, # Force store nodes in index and document stores
|
||||||
|
show_progress=self.show_progress,
|
||||||
|
embed_model=self.embed_model,
|
||||||
|
transformations=self.transformations,
|
||||||
|
)
|
||||||
|
except ValueError:
|
||||||
|
# There are no index in the storage context, creating a new one
|
||||||
|
logger.info("Creating a new vector store index")
|
||||||
|
index = VectorStoreIndex.from_documents(
|
||||||
|
[],
|
||||||
|
storage_context=self.storage_context,
|
||||||
|
store_nodes_override=True, # Force store nodes in index and document stores
|
||||||
|
show_progress=self.show_progress,
|
||||||
|
embed_model=self.embed_model,
|
||||||
|
transformations=self.transformations,
|
||||||
|
)
|
||||||
|
index.storage_context.persist(persist_dir=local_data_path)
|
||||||
|
return index
|
||||||
|
|
||||||
|
def _save_index(self) -> None:
|
||||||
|
self._index.storage_context.persist(persist_dir=local_data_path)
|
||||||
|
|
||||||
|
def delete(self, doc_id: str) -> None:
|
||||||
|
with self._index_thread_lock:
|
||||||
|
# Delete the document from the index
|
||||||
|
self._index.delete_ref_doc(doc_id, delete_from_docstore=True)
|
||||||
|
|
||||||
|
# Save the index
|
||||||
|
self._save_index()
|
||||||
|
|
||||||
|
|
||||||
|
class SimpleIngestComponent(BaseIngestComponentWithIndex):
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
storage_context: StorageContext,
|
||||||
|
embed_model: EmbedType,
|
||||||
|
transformations: list[TransformComponent],
|
||||||
|
*args: Any,
|
||||||
|
**kwargs: Any,
|
||||||
|
) -> None:
|
||||||
|
super().__init__(storage_context, embed_model, transformations, *args, **kwargs)
|
||||||
|
|
||||||
|
def ingest(self, file_name: str, file_data: Path) -> list[Document]:
|
||||||
|
logger.info("Ingesting file_name=%s", file_name)
|
||||||
|
documents = IngestionHelper.transform_file_into_documents(file_name, file_data)
|
||||||
|
logger.info(
|
||||||
|
"Transformed file=%s into count=%s documents", file_name, len(documents)
|
||||||
|
)
|
||||||
|
logger.debug("Saving the documents in the index and doc store")
|
||||||
|
return self._save_docs(documents)
|
||||||
|
|
||||||
|
def bulk_ingest(self, files: list[tuple[str, Path]]) -> list[Document]:
|
||||||
|
saved_documents = []
|
||||||
|
for file_name, file_data in files:
|
||||||
|
documents = IngestionHelper.transform_file_into_documents(
|
||||||
|
file_name, file_data
|
||||||
|
)
|
||||||
|
saved_documents.extend(self._save_docs(documents))
|
||||||
|
return saved_documents
|
||||||
|
|
||||||
|
def _save_docs(self, documents: list[Document]) -> list[Document]:
|
||||||
|
logger.debug("Transforming count=%s documents into nodes", len(documents))
|
||||||
|
with self._index_thread_lock:
|
||||||
|
for document in documents:
|
||||||
|
self._index.insert(document, show_progress=True)
|
||||||
|
logger.debug("Persisting the index and nodes")
|
||||||
|
# persist the index and nodes
|
||||||
|
self._save_index()
|
||||||
|
logger.debug("Persisted the index and nodes")
|
||||||
|
return documents
|
||||||
|
|
||||||
|
|
||||||
|
class BatchIngestComponent(BaseIngestComponentWithIndex):
|
||||||
|
"""Parallelize the file reading and parsing on multiple CPU core.
|
||||||
|
|
||||||
|
This also makes the embeddings to be computed in batches (on GPU or CPU).
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
storage_context: StorageContext,
|
||||||
|
embed_model: EmbedType,
|
||||||
|
transformations: list[TransformComponent],
|
||||||
|
count_workers: int,
|
||||||
|
*args: Any,
|
||||||
|
**kwargs: Any,
|
||||||
|
) -> None:
|
||||||
|
super().__init__(storage_context, embed_model, transformations, *args, **kwargs)
|
||||||
|
# Make an efficient use of the CPU and GPU, the embedding
|
||||||
|
# must be in the transformations
|
||||||
|
assert (
|
||||||
|
len(self.transformations) >= 2
|
||||||
|
), "Embeddings must be in the transformations"
|
||||||
|
assert count_workers > 0, "count_workers must be > 0"
|
||||||
|
self.count_workers = count_workers
|
||||||
|
|
||||||
|
self._file_to_documents_work_pool = multiprocessing.Pool(
|
||||||
|
processes=self.count_workers
|
||||||
|
)
|
||||||
|
|
||||||
|
def ingest(self, file_name: str, file_data: Path) -> list[Document]:
|
||||||
|
logger.info("Ingesting file_name=%s", file_name)
|
||||||
|
documents = IngestionHelper.transform_file_into_documents(file_name, file_data)
|
||||||
|
logger.info(
|
||||||
|
"Transformed file=%s into count=%s documents", file_name, len(documents)
|
||||||
|
)
|
||||||
|
logger.debug("Saving the documents in the index and doc store")
|
||||||
|
return self._save_docs(documents)
|
||||||
|
|
||||||
|
def bulk_ingest(self, files: list[tuple[str, Path]]) -> list[Document]:
|
||||||
|
documents = list(
|
||||||
|
itertools.chain.from_iterable(
|
||||||
|
self._file_to_documents_work_pool.starmap(
|
||||||
|
IngestionHelper.transform_file_into_documents, files
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"Transformed count=%s files into count=%s documents",
|
||||||
|
len(files),
|
||||||
|
len(documents),
|
||||||
|
)
|
||||||
|
return self._save_docs(documents)
|
||||||
|
|
||||||
|
def _save_docs(self, documents: list[Document]) -> list[Document]:
|
||||||
|
logger.debug("Transforming count=%s documents into nodes", len(documents))
|
||||||
|
nodes = run_transformations(
|
||||||
|
documents, # type: ignore[arg-type]
|
||||||
|
self.transformations,
|
||||||
|
show_progress=self.show_progress,
|
||||||
|
)
|
||||||
|
# Locking the index to avoid concurrent writes
|
||||||
|
with self._index_thread_lock:
|
||||||
|
logger.info("Inserting count=%s nodes in the index", len(nodes))
|
||||||
|
self._index.insert_nodes(nodes, show_progress=True)
|
||||||
|
for document in documents:
|
||||||
|
self._index.docstore.set_document_hash(
|
||||||
|
document.get_doc_id(), document.hash
|
||||||
|
)
|
||||||
|
logger.debug("Persisting the index and nodes")
|
||||||
|
# persist the index and nodes
|
||||||
|
self._save_index()
|
||||||
|
logger.debug("Persisted the index and nodes")
|
||||||
|
return documents
|
||||||
|
|
||||||
|
|
||||||
|
class ParallelizedIngestComponent(BaseIngestComponentWithIndex):
|
||||||
|
"""Parallelize the file ingestion (file reading, embeddings, and index insertion).
|
||||||
|
|
||||||
|
This use the CPU and GPU in parallel (both running at the same time), and
|
||||||
|
reduce the memory pressure by not loading all the files in memory at the same time.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
storage_context: StorageContext,
|
||||||
|
embed_model: EmbedType,
|
||||||
|
transformations: list[TransformComponent],
|
||||||
|
count_workers: int,
|
||||||
|
*args: Any,
|
||||||
|
**kwargs: Any,
|
||||||
|
) -> None:
|
||||||
|
super().__init__(storage_context, embed_model, transformations, *args, **kwargs)
|
||||||
|
# To make an efficient use of the CPU and GPU, the embeddings
|
||||||
|
# must be in the transformations (to be computed in batches)
|
||||||
|
assert (
|
||||||
|
len(self.transformations) >= 2
|
||||||
|
), "Embeddings must be in the transformations"
|
||||||
|
assert count_workers > 0, "count_workers must be > 0"
|
||||||
|
self.count_workers = count_workers
|
||||||
|
# We are doing our own multiprocessing
|
||||||
|
# To do not collide with the multiprocessing of huggingface, we disable it
|
||||||
|
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||||||
|
|
||||||
|
self._ingest_work_pool = multiprocessing.pool.ThreadPool(
|
||||||
|
processes=self.count_workers
|
||||||
|
)
|
||||||
|
|
||||||
|
self._file_to_documents_work_pool = multiprocessing.Pool(
|
||||||
|
processes=self.count_workers
|
||||||
|
)
|
||||||
|
|
||||||
|
def ingest(self, file_name: str, file_data: Path) -> list[Document]:
|
||||||
|
logger.info("Ingesting file_name=%s", file_name)
|
||||||
|
# Running in a single (1) process to release the current
|
||||||
|
# thread, and take a dedicated CPU core for computation
|
||||||
|
documents = self._file_to_documents_work_pool.apply(
|
||||||
|
IngestionHelper.transform_file_into_documents, (file_name, file_data)
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"Transformed file=%s into count=%s documents", file_name, len(documents)
|
||||||
|
)
|
||||||
|
logger.debug("Saving the documents in the index and doc store")
|
||||||
|
return self._save_docs(documents)
|
||||||
|
|
||||||
|
def bulk_ingest(self, files: list[tuple[str, Path]]) -> list[Document]:
|
||||||
|
# Lightweight threads, used for parallelize the
|
||||||
|
# underlying IO calls made in the ingestion
|
||||||
|
|
||||||
|
documents = list(
|
||||||
|
itertools.chain.from_iterable(
|
||||||
|
self._ingest_work_pool.starmap(self.ingest, files)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return documents
|
||||||
|
|
||||||
|
def _save_docs(self, documents: list[Document]) -> list[Document]:
|
||||||
|
logger.debug("Transforming count=%s documents into nodes", len(documents))
|
||||||
|
nodes = run_transformations(
|
||||||
|
documents, # type: ignore[arg-type]
|
||||||
|
self.transformations,
|
||||||
|
show_progress=self.show_progress,
|
||||||
|
)
|
||||||
|
# Locking the index to avoid concurrent writes
|
||||||
|
with self._index_thread_lock:
|
||||||
|
logger.info("Inserting count=%s nodes in the index", len(nodes))
|
||||||
|
self._index.insert_nodes(nodes, show_progress=True)
|
||||||
|
for document in documents:
|
||||||
|
self._index.docstore.set_document_hash(
|
||||||
|
document.get_doc_id(), document.hash
|
||||||
|
)
|
||||||
|
logger.debug("Persisting the index and nodes")
|
||||||
|
# persist the index and nodes
|
||||||
|
self._save_index()
|
||||||
|
logger.debug("Persisted the index and nodes")
|
||||||
|
return documents
|
||||||
|
|
||||||
|
def __del__(self) -> None:
|
||||||
|
# We need to do the appropriate cleanup of the multiprocessing pools
|
||||||
|
# when the object is deleted. Using root logger to avoid
|
||||||
|
# the logger to be deleted before the pool
|
||||||
|
logging.debug("Closing the ingest work pool")
|
||||||
|
self._ingest_work_pool.close()
|
||||||
|
self._ingest_work_pool.join()
|
||||||
|
self._ingest_work_pool.terminate()
|
||||||
|
logging.debug("Closing the file to documents work pool")
|
||||||
|
self._file_to_documents_work_pool.close()
|
||||||
|
self._file_to_documents_work_pool.join()
|
||||||
|
self._file_to_documents_work_pool.terminate()
|
||||||
|
|
||||||
|
|
||||||
|
class PipelineIngestComponent(BaseIngestComponentWithIndex):
|
||||||
|
"""Pipeline ingestion - keeping the embedding worker pool as busy as possible.
|
||||||
|
|
||||||
|
This class implements a threaded ingestion pipeline, which comprises two threads
|
||||||
|
and two queues. The primary thread is responsible for reading and parsing files
|
||||||
|
into documents. These documents are then placed into a queue, which is
|
||||||
|
distributed to a pool of worker processes for embedding computation. After
|
||||||
|
embedding, the documents are transferred to another queue where they are
|
||||||
|
accumulated until a threshold is reached. Upon reaching this threshold, the
|
||||||
|
accumulated documents are flushed to the document store, index, and vector
|
||||||
|
store.
|
||||||
|
|
||||||
|
Exception handling ensures robustness against erroneous files. However, in the
|
||||||
|
pipelined design, one error can lead to the discarding of multiple files. Any
|
||||||
|
discarded files will be reported.
|
||||||
|
"""
|
||||||
|
|
||||||
|
NODE_FLUSH_COUNT = 5000 # Save the index every # nodes.
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
storage_context: StorageContext,
|
||||||
|
embed_model: EmbedType,
|
||||||
|
transformations: list[TransformComponent],
|
||||||
|
count_workers: int,
|
||||||
|
*args: Any,
|
||||||
|
**kwargs: Any,
|
||||||
|
) -> None:
|
||||||
|
super().__init__(storage_context, embed_model, transformations, *args, **kwargs)
|
||||||
|
self.count_workers = count_workers
|
||||||
|
assert (
|
||||||
|
len(self.transformations) >= 2
|
||||||
|
), "Embeddings must be in the transformations"
|
||||||
|
assert count_workers > 0, "count_workers must be > 0"
|
||||||
|
self.count_workers = count_workers
|
||||||
|
# We are doing our own multiprocessing
|
||||||
|
# To do not collide with the multiprocessing of huggingface, we disable it
|
||||||
|
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||||||
|
|
||||||
|
# doc_q stores parsed files as Document chunks.
|
||||||
|
# Using a shallow queue causes the filesystem parser to block
|
||||||
|
# when it reaches capacity. This ensures it doesn't outpace the
|
||||||
|
# computationally intensive embeddings phase, avoiding unnecessary
|
||||||
|
# memory consumption. The semaphore is used to bound the async worker
|
||||||
|
# embedding computations to cause the doc Q to fill and block.
|
||||||
|
self.doc_semaphore = multiprocessing.Semaphore(
|
||||||
|
self.count_workers
|
||||||
|
) # limit the doc queue to # items.
|
||||||
|
self.doc_q: Queue[tuple[str, str | None, list[Document] | None]] = Queue(20)
|
||||||
|
# node_q stores documents parsed into nodes (embeddings).
|
||||||
|
# Larger queue size so we don't block the embedding workers during a slow
|
||||||
|
# index update.
|
||||||
|
self.node_q: Queue[
|
||||||
|
tuple[str, str | None, list[Document] | None, list[BaseNode] | None]
|
||||||
|
] = Queue(40)
|
||||||
|
threading.Thread(target=self._doc_to_node, daemon=True).start()
|
||||||
|
threading.Thread(target=self._write_nodes, daemon=True).start()
|
||||||
|
|
||||||
|
def _doc_to_node(self) -> None:
|
||||||
|
# Parse documents into nodes
|
||||||
|
with multiprocessing.pool.ThreadPool(processes=self.count_workers) as pool:
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
cmd, file_name, documents = self.doc_q.get(
|
||||||
|
block=True
|
||||||
|
) # Documents for a file
|
||||||
|
if cmd == "process":
|
||||||
|
# Push CPU/GPU embedding work to the worker pool
|
||||||
|
# Acquire semaphore to control access to worker pool
|
||||||
|
self.doc_semaphore.acquire()
|
||||||
|
pool.apply_async(
|
||||||
|
self._doc_to_node_worker, (file_name, documents)
|
||||||
|
)
|
||||||
|
elif cmd == "quit":
|
||||||
|
break
|
||||||
|
finally:
|
||||||
|
if cmd != "process":
|
||||||
|
self.doc_q.task_done() # unblock Q joins
|
||||||
|
|
||||||
|
def _doc_to_node_worker(self, file_name: str, documents: list[Document]) -> None:
|
||||||
|
# CPU/GPU intensive work in its own process
|
||||||
|
try:
|
||||||
|
nodes = run_transformations(
|
||||||
|
documents, # type: ignore[arg-type]
|
||||||
|
self.transformations,
|
||||||
|
show_progress=self.show_progress,
|
||||||
|
)
|
||||||
|
self.node_q.put(("process", file_name, documents, list(nodes)))
|
||||||
|
finally:
|
||||||
|
self.doc_semaphore.release()
|
||||||
|
self.doc_q.task_done() # unblock Q joins
|
||||||
|
|
||||||
|
def _save_docs(
|
||||||
|
self, files: list[str], documents: list[Document], nodes: list[BaseNode]
|
||||||
|
) -> None:
|
||||||
|
try:
|
||||||
|
logger.info(
|
||||||
|
f"Saving {len(files)} files ({len(documents)} documents / {len(nodes)} nodes)"
|
||||||
|
)
|
||||||
|
self._index.insert_nodes(nodes)
|
||||||
|
for document in documents:
|
||||||
|
self._index.docstore.set_document_hash(
|
||||||
|
document.get_doc_id(), document.hash
|
||||||
|
)
|
||||||
|
self._save_index()
|
||||||
|
except Exception:
|
||||||
|
# Tell the user so they can investigate these files
|
||||||
|
logger.exception(f"Processing files {files}")
|
||||||
|
finally:
|
||||||
|
# Clearing work, even on exception, maintains a clean state.
|
||||||
|
nodes.clear()
|
||||||
|
documents.clear()
|
||||||
|
files.clear()
|
||||||
|
|
||||||
|
def _write_nodes(self) -> None:
|
||||||
|
# Save nodes to index. I/O intensive.
|
||||||
|
node_stack: list[BaseNode] = []
|
||||||
|
doc_stack: list[Document] = []
|
||||||
|
file_stack: list[str] = []
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
cmd, file_name, documents, nodes = self.node_q.get(block=True)
|
||||||
|
if cmd in ("flush", "quit"):
|
||||||
|
if file_stack:
|
||||||
|
self._save_docs(file_stack, doc_stack, node_stack)
|
||||||
|
if cmd == "quit":
|
||||||
|
break
|
||||||
|
elif cmd == "process":
|
||||||
|
node_stack.extend(nodes) # type: ignore[arg-type]
|
||||||
|
doc_stack.extend(documents) # type: ignore[arg-type]
|
||||||
|
file_stack.append(file_name) # type: ignore[arg-type]
|
||||||
|
# Constant saving is heavy on I/O - accumulate to a threshold
|
||||||
|
if len(node_stack) >= self.NODE_FLUSH_COUNT:
|
||||||
|
self._save_docs(file_stack, doc_stack, node_stack)
|
||||||
|
finally:
|
||||||
|
self.node_q.task_done()
|
||||||
|
|
||||||
|
def _flush(self) -> None:
|
||||||
|
self.doc_q.put(("flush", None, None))
|
||||||
|
self.doc_q.join()
|
||||||
|
self.node_q.put(("flush", None, None, None))
|
||||||
|
self.node_q.join()
|
||||||
|
|
||||||
|
def ingest(self, file_name: str, file_data: Path) -> list[Document]:
|
||||||
|
documents = IngestionHelper.transform_file_into_documents(file_name, file_data)
|
||||||
|
self.doc_q.put(("process", file_name, documents))
|
||||||
|
self._flush()
|
||||||
|
return documents
|
||||||
|
|
||||||
|
def bulk_ingest(self, files: list[tuple[str, Path]]) -> list[Document]:
|
||||||
|
docs = []
|
||||||
|
for file_name, file_data in eta(files):
|
||||||
|
try:
|
||||||
|
documents = IngestionHelper.transform_file_into_documents(
|
||||||
|
file_name, file_data
|
||||||
|
)
|
||||||
|
self.doc_q.put(("process", file_name, documents))
|
||||||
|
docs.extend(documents)
|
||||||
|
except Exception:
|
||||||
|
logger.exception(f"Skipping {file_data.name}")
|
||||||
|
self._flush()
|
||||||
|
return docs
|
||||||
|
|
||||||
|
|
||||||
|
def get_ingestion_component(
|
||||||
|
storage_context: StorageContext,
|
||||||
|
embed_model: EmbedType,
|
||||||
|
transformations: list[TransformComponent],
|
||||||
|
settings: Settings,
|
||||||
|
) -> BaseIngestComponent:
|
||||||
|
"""Get the ingestion component for the given configuration."""
|
||||||
|
ingest_mode = settings.embedding.ingest_mode
|
||||||
|
if ingest_mode == "batch":
|
||||||
|
return BatchIngestComponent(
|
||||||
|
storage_context=storage_context,
|
||||||
|
embed_model=embed_model,
|
||||||
|
transformations=transformations,
|
||||||
|
count_workers=settings.embedding.count_workers,
|
||||||
|
)
|
||||||
|
elif ingest_mode == "parallel":
|
||||||
|
return ParallelizedIngestComponent(
|
||||||
|
storage_context=storage_context,
|
||||||
|
embed_model=embed_model,
|
||||||
|
transformations=transformations,
|
||||||
|
count_workers=settings.embedding.count_workers,
|
||||||
|
)
|
||||||
|
elif ingest_mode == "pipeline":
|
||||||
|
return PipelineIngestComponent(
|
||||||
|
storage_context=storage_context,
|
||||||
|
embed_model=embed_model,
|
||||||
|
transformations=transformations,
|
||||||
|
count_workers=settings.embedding.count_workers,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
return SimpleIngestComponent(
|
||||||
|
storage_context=storage_context,
|
||||||
|
embed_model=embed_model,
|
||||||
|
transformations=transformations,
|
||||||
|
)
|
||||||
111
pgpt/private_gpt/components/ingest/ingest_helper.py
Normal file
111
pgpt/private_gpt/components/ingest/ingest_helper.py
Normal file
@ -0,0 +1,111 @@
|
|||||||
|
import logging
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from llama_index.core.readers import StringIterableReader
|
||||||
|
from llama_index.core.readers.base import BaseReader
|
||||||
|
from llama_index.core.readers.json import JSONReader
|
||||||
|
from llama_index.core.schema import Document
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# Inspired by the `llama_index.core.readers.file.base` module
|
||||||
|
def _try_loading_included_file_formats() -> dict[str, type[BaseReader]]:
|
||||||
|
try:
|
||||||
|
from llama_index.readers.file.docs import ( # type: ignore
|
||||||
|
DocxReader,
|
||||||
|
HWPReader,
|
||||||
|
PDFReader,
|
||||||
|
)
|
||||||
|
from llama_index.readers.file.epub import EpubReader # type: ignore
|
||||||
|
from llama_index.readers.file.image import ImageReader # type: ignore
|
||||||
|
from llama_index.readers.file.ipynb import IPYNBReader # type: ignore
|
||||||
|
from llama_index.readers.file.markdown import MarkdownReader # type: ignore
|
||||||
|
from llama_index.readers.file.mbox import MboxReader # type: ignore
|
||||||
|
from llama_index.readers.file.slides import PptxReader # type: ignore
|
||||||
|
from llama_index.readers.file.tabular import PandasCSVReader # type: ignore
|
||||||
|
from llama_index.readers.file.video_audio import ( # type: ignore
|
||||||
|
VideoAudioReader,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError("`llama-index-readers-file` package not found") from e
|
||||||
|
|
||||||
|
default_file_reader_cls: dict[str, type[BaseReader]] = {
|
||||||
|
".hwp": HWPReader,
|
||||||
|
".pdf": PDFReader,
|
||||||
|
".docx": DocxReader,
|
||||||
|
".pptx": PptxReader,
|
||||||
|
".ppt": PptxReader,
|
||||||
|
".pptm": PptxReader,
|
||||||
|
".jpg": ImageReader,
|
||||||
|
".png": ImageReader,
|
||||||
|
".jpeg": ImageReader,
|
||||||
|
".mp3": VideoAudioReader,
|
||||||
|
".mp4": VideoAudioReader,
|
||||||
|
".csv": PandasCSVReader,
|
||||||
|
".epub": EpubReader,
|
||||||
|
".md": MarkdownReader,
|
||||||
|
".mbox": MboxReader,
|
||||||
|
".ipynb": IPYNBReader,
|
||||||
|
}
|
||||||
|
return default_file_reader_cls
|
||||||
|
|
||||||
|
|
||||||
|
# Patching the default file reader to support other file types
|
||||||
|
FILE_READER_CLS = _try_loading_included_file_formats()
|
||||||
|
FILE_READER_CLS.update(
|
||||||
|
{
|
||||||
|
".json": JSONReader,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class IngestionHelper:
|
||||||
|
"""Helper class to transform a file into a list of documents.
|
||||||
|
|
||||||
|
This class should be used to transform a file into a list of documents.
|
||||||
|
These methods are thread-safe (and multiprocessing-safe).
|
||||||
|
"""
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def transform_file_into_documents(
|
||||||
|
file_name: str, file_data: Path
|
||||||
|
) -> list[Document]:
|
||||||
|
documents = IngestionHelper._load_file_to_documents(file_name, file_data)
|
||||||
|
for document in documents:
|
||||||
|
document.metadata["file_name"] = file_name
|
||||||
|
IngestionHelper._exclude_metadata(documents)
|
||||||
|
return documents
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _load_file_to_documents(file_name: str, file_data: Path) -> list[Document]:
|
||||||
|
logger.debug("Transforming file_name=%s into documents", file_name)
|
||||||
|
extension = Path(file_name).suffix
|
||||||
|
reader_cls = FILE_READER_CLS.get(extension)
|
||||||
|
if reader_cls is None:
|
||||||
|
logger.debug(
|
||||||
|
"No reader found for extension=%s, using default string reader",
|
||||||
|
extension,
|
||||||
|
)
|
||||||
|
# Read as a plain text
|
||||||
|
string_reader = StringIterableReader()
|
||||||
|
return string_reader.load_data([file_data.read_text()])
|
||||||
|
|
||||||
|
logger.debug("Specific reader found for extension=%s", extension)
|
||||||
|
documents = reader_cls().load_data(file_data)
|
||||||
|
|
||||||
|
# Sanitize NUL bytes in text which can't be stored in Postgres
|
||||||
|
for i in range(len(documents)):
|
||||||
|
documents[i].text = documents[i].text.replace("\u0000", "")
|
||||||
|
|
||||||
|
return documents
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _exclude_metadata(documents: list[Document]) -> None:
|
||||||
|
logger.debug("Excluding metadata from count=%s documents", len(documents))
|
||||||
|
for document in documents:
|
||||||
|
document.metadata["doc_id"] = document.doc_id
|
||||||
|
# We don't want the Embeddings search to receive this metadata
|
||||||
|
document.excluded_embed_metadata_keys = ["doc_id"]
|
||||||
|
# We don't want the LLM to receive these metadata in the context
|
||||||
|
document.excluded_llm_metadata_keys = ["file_name", "doc_id", "page_label"]
|
||||||
1
pgpt/private_gpt/components/llm/__init__.py
Normal file
1
pgpt/private_gpt/components/llm/__init__.py
Normal file
@ -0,0 +1 @@
|
|||||||
|
"""LLM implementations."""
|
||||||
0
pgpt/private_gpt/components/llm/custom/__init__.py
Normal file
0
pgpt/private_gpt/components/llm/custom/__init__.py
Normal file
276
pgpt/private_gpt/components/llm/custom/sagemaker.py
Normal file
276
pgpt/private_gpt/components/llm/custom/sagemaker.py
Normal file
@ -0,0 +1,276 @@
|
|||||||
|
# mypy: ignore-errors
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import io
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
|
import boto3 # type: ignore
|
||||||
|
from llama_index.core.base.llms.generic_utils import (
|
||||||
|
completion_response_to_chat_response,
|
||||||
|
stream_completion_response_to_chat_response,
|
||||||
|
)
|
||||||
|
from llama_index.core.bridge.pydantic import Field
|
||||||
|
from llama_index.core.llms import (
|
||||||
|
CompletionResponse,
|
||||||
|
CustomLLM,
|
||||||
|
LLMMetadata,
|
||||||
|
)
|
||||||
|
from llama_index.core.llms.callbacks import (
|
||||||
|
llm_chat_callback,
|
||||||
|
llm_completion_callback,
|
||||||
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
from llama_index.callbacks import CallbackManager
|
||||||
|
from llama_index.llms import (
|
||||||
|
ChatMessage,
|
||||||
|
ChatResponse,
|
||||||
|
ChatResponseGen,
|
||||||
|
CompletionResponseGen,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class LineIterator:
|
||||||
|
r"""A helper class for parsing the byte stream input from TGI container.
|
||||||
|
|
||||||
|
The output of the model will be in the following format:
|
||||||
|
```
|
||||||
|
b'data:{"token": {"text": " a"}}\n\n'
|
||||||
|
b'data:{"token": {"text": " challenging"}}\n\n'
|
||||||
|
b'data:{"token": {"text": " problem"
|
||||||
|
b'}}'
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
While usually each PayloadPart event from the event stream will contain a byte array
|
||||||
|
with a full json, this is not guaranteed and some of the json objects may be split
|
||||||
|
across PayloadPart events. For example:
|
||||||
|
```
|
||||||
|
{'PayloadPart': {'Bytes': b'{"outputs": '}}
|
||||||
|
{'PayloadPart': {'Bytes': b'[" problem"]}\n'}}
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
This class accounts for this by concatenating bytes written via the 'write' function
|
||||||
|
and then exposing a method which will return lines (ending with a '\n' character)
|
||||||
|
within the buffer via the 'scan_lines' function. It maintains the position of the
|
||||||
|
last read position to ensure that previous bytes are not exposed again. It will
|
||||||
|
also save any pending lines that doe not end with a '\n' to make sure truncations
|
||||||
|
are concatinated
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, stream: Any) -> None:
|
||||||
|
"""Line iterator initializer."""
|
||||||
|
self.byte_iterator = iter(stream)
|
||||||
|
self.buffer = io.BytesIO()
|
||||||
|
self.read_pos = 0
|
||||||
|
|
||||||
|
def __iter__(self) -> Any:
|
||||||
|
"""Self iterator."""
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __next__(self) -> Any:
|
||||||
|
"""Next element from iterator."""
|
||||||
|
while True:
|
||||||
|
self.buffer.seek(self.read_pos)
|
||||||
|
line = self.buffer.readline()
|
||||||
|
if line and line[-1] == ord("\n"):
|
||||||
|
self.read_pos += len(line)
|
||||||
|
return line[:-1]
|
||||||
|
try:
|
||||||
|
chunk = next(self.byte_iterator)
|
||||||
|
except StopIteration:
|
||||||
|
if self.read_pos < self.buffer.getbuffer().nbytes:
|
||||||
|
continue
|
||||||
|
raise
|
||||||
|
if "PayloadPart" not in chunk:
|
||||||
|
logger.warning("Unknown event type=%s", chunk)
|
||||||
|
continue
|
||||||
|
self.buffer.seek(0, io.SEEK_END)
|
||||||
|
self.buffer.write(chunk["PayloadPart"]["Bytes"])
|
||||||
|
|
||||||
|
|
||||||
|
class SagemakerLLM(CustomLLM):
|
||||||
|
"""Sagemaker Inference Endpoint models.
|
||||||
|
|
||||||
|
To use, you must supply the endpoint name from your deployed
|
||||||
|
Sagemaker model & the region where it is deployed.
|
||||||
|
|
||||||
|
To authenticate, the AWS client uses the following methods to
|
||||||
|
automatically load credentials:
|
||||||
|
https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html
|
||||||
|
|
||||||
|
If a specific credential profile should be used, you must pass
|
||||||
|
the name of the profile from the ~/.aws/credentials file that is to be used.
|
||||||
|
|
||||||
|
Make sure the credentials / roles used have the required policies to
|
||||||
|
access the Sagemaker endpoint.
|
||||||
|
See: https://docs.aws.amazon.com/IAM/latest/UserGuide/access_policies.html
|
||||||
|
"""
|
||||||
|
|
||||||
|
endpoint_name: str = Field(description="")
|
||||||
|
temperature: float = Field(description="The temperature to use for sampling.")
|
||||||
|
max_new_tokens: int = Field(description="The maximum number of tokens to generate.")
|
||||||
|
context_window: int = Field(
|
||||||
|
description="The maximum number of context tokens for the model."
|
||||||
|
)
|
||||||
|
messages_to_prompt: Any = Field(
|
||||||
|
description="The function to convert messages to a prompt.", exclude=True
|
||||||
|
)
|
||||||
|
completion_to_prompt: Any = Field(
|
||||||
|
description="The function to convert a completion to a prompt.", exclude=True
|
||||||
|
)
|
||||||
|
generate_kwargs: dict[str, Any] = Field(
|
||||||
|
default_factory=dict, description="Kwargs used for generation."
|
||||||
|
)
|
||||||
|
model_kwargs: dict[str, Any] = Field(
|
||||||
|
default_factory=dict, description="Kwargs used for model initialization."
|
||||||
|
)
|
||||||
|
verbose: bool = Field(description="Whether to print verbose output.")
|
||||||
|
|
||||||
|
_boto_client: Any = boto3.client(
|
||||||
|
"sagemaker-runtime",
|
||||||
|
) # TODO make it an optional field
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
endpoint_name: str | None = "",
|
||||||
|
temperature: float = 0.1,
|
||||||
|
max_new_tokens: int = 512, # to review defaults
|
||||||
|
context_window: int = 2048, # to review defaults
|
||||||
|
messages_to_prompt: Any = None,
|
||||||
|
completion_to_prompt: Any = None,
|
||||||
|
callback_manager: CallbackManager | None = None,
|
||||||
|
generate_kwargs: dict[str, Any] | None = None,
|
||||||
|
model_kwargs: dict[str, Any] | None = None,
|
||||||
|
verbose: bool = True,
|
||||||
|
) -> None:
|
||||||
|
"""SagemakerLLM initializer."""
|
||||||
|
model_kwargs = model_kwargs or {}
|
||||||
|
model_kwargs.update({"n_ctx": context_window, "verbose": verbose})
|
||||||
|
|
||||||
|
messages_to_prompt = messages_to_prompt or {}
|
||||||
|
completion_to_prompt = completion_to_prompt or {}
|
||||||
|
|
||||||
|
generate_kwargs = generate_kwargs or {}
|
||||||
|
generate_kwargs.update(
|
||||||
|
{"temperature": temperature, "max_tokens": max_new_tokens}
|
||||||
|
)
|
||||||
|
|
||||||
|
super().__init__(
|
||||||
|
endpoint_name=endpoint_name,
|
||||||
|
temperature=temperature,
|
||||||
|
context_window=context_window,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
messages_to_prompt=messages_to_prompt,
|
||||||
|
completion_to_prompt=completion_to_prompt,
|
||||||
|
callback_manager=callback_manager,
|
||||||
|
generate_kwargs=generate_kwargs,
|
||||||
|
model_kwargs=model_kwargs,
|
||||||
|
verbose=verbose,
|
||||||
|
)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def inference_params(self):
|
||||||
|
# TODO expose the rest of params
|
||||||
|
return {
|
||||||
|
"do_sample": True,
|
||||||
|
"top_p": 0.7,
|
||||||
|
"temperature": self.temperature,
|
||||||
|
"top_k": 50,
|
||||||
|
"max_new_tokens": self.max_new_tokens,
|
||||||
|
}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def metadata(self) -> LLMMetadata:
|
||||||
|
"""Get LLM metadata."""
|
||||||
|
return LLMMetadata(
|
||||||
|
context_window=self.context_window,
|
||||||
|
num_output=self.max_new_tokens,
|
||||||
|
model_name="Sagemaker LLama 2",
|
||||||
|
)
|
||||||
|
|
||||||
|
@llm_completion_callback()
|
||||||
|
def complete(self, prompt: str, **kwargs: Any) -> CompletionResponse:
|
||||||
|
self.generate_kwargs.update({"stream": False})
|
||||||
|
|
||||||
|
is_formatted = kwargs.pop("formatted", False)
|
||||||
|
if not is_formatted:
|
||||||
|
prompt = self.completion_to_prompt(prompt)
|
||||||
|
|
||||||
|
request_params = {
|
||||||
|
"inputs": prompt,
|
||||||
|
"stream": False,
|
||||||
|
"parameters": self.inference_params,
|
||||||
|
}
|
||||||
|
|
||||||
|
resp = self._boto_client.invoke_endpoint(
|
||||||
|
EndpointName=self.endpoint_name,
|
||||||
|
Body=json.dumps(request_params),
|
||||||
|
ContentType="application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
response_body = resp["Body"]
|
||||||
|
response_str = response_body.read().decode("utf-8")
|
||||||
|
response_dict = json.loads(response_str)
|
||||||
|
|
||||||
|
return CompletionResponse(
|
||||||
|
text=response_dict[0]["generated_text"][len(prompt) :], raw=resp
|
||||||
|
)
|
||||||
|
|
||||||
|
@llm_completion_callback()
|
||||||
|
def stream_complete(self, prompt: str, **kwargs: Any) -> CompletionResponseGen:
|
||||||
|
def get_stream():
|
||||||
|
text = ""
|
||||||
|
|
||||||
|
request_params = {
|
||||||
|
"inputs": prompt,
|
||||||
|
"stream": True,
|
||||||
|
"parameters": self.inference_params,
|
||||||
|
}
|
||||||
|
resp = self._boto_client.invoke_endpoint_with_response_stream(
|
||||||
|
EndpointName=self.endpoint_name,
|
||||||
|
Body=json.dumps(request_params),
|
||||||
|
ContentType="application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
event_stream = resp["Body"]
|
||||||
|
start_json = b"{"
|
||||||
|
stop_token = "<|endoftext|>"
|
||||||
|
first_token = True
|
||||||
|
|
||||||
|
for line in LineIterator(event_stream):
|
||||||
|
if line != b"" and start_json in line:
|
||||||
|
data = json.loads(line[line.find(start_json) :].decode("utf-8"))
|
||||||
|
special = data["token"]["special"]
|
||||||
|
stop = data["token"]["text"] == stop_token
|
||||||
|
if not special and not stop:
|
||||||
|
delta = data["token"]["text"]
|
||||||
|
# trim the leading space for the first token if present
|
||||||
|
if first_token:
|
||||||
|
delta = delta.lstrip()
|
||||||
|
first_token = False
|
||||||
|
text += delta
|
||||||
|
yield CompletionResponse(delta=delta, text=text, raw=data)
|
||||||
|
|
||||||
|
return get_stream()
|
||||||
|
|
||||||
|
@llm_chat_callback()
|
||||||
|
def chat(self, messages: Sequence[ChatMessage], **kwargs: Any) -> ChatResponse:
|
||||||
|
prompt = self.messages_to_prompt(messages)
|
||||||
|
completion_response = self.complete(prompt, formatted=True, **kwargs)
|
||||||
|
return completion_response_to_chat_response(completion_response)
|
||||||
|
|
||||||
|
@llm_chat_callback()
|
||||||
|
def stream_chat(
|
||||||
|
self, messages: Sequence[ChatMessage], **kwargs: Any
|
||||||
|
) -> ChatResponseGen:
|
||||||
|
prompt = self.messages_to_prompt(messages)
|
||||||
|
completion_response = self.stream_complete(prompt, formatted=True, **kwargs)
|
||||||
|
return stream_completion_response_to_chat_response(completion_response)
|
||||||
225
pgpt/private_gpt/components/llm/llm_component.py
Normal file
225
pgpt/private_gpt/components/llm/llm_component.py
Normal file
@ -0,0 +1,225 @@
|
|||||||
|
import logging
|
||||||
|
from collections.abc import Callable
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from injector import inject, singleton
|
||||||
|
from llama_index.core.llms import LLM, MockLLM
|
||||||
|
from llama_index.core.settings import Settings as LlamaIndexSettings
|
||||||
|
from llama_index.core.utils import set_global_tokenizer
|
||||||
|
from transformers import AutoTokenizer # type: ignore
|
||||||
|
|
||||||
|
from private_gpt.components.llm.prompt_helper import get_prompt_style
|
||||||
|
from private_gpt.paths import models_cache_path, models_path
|
||||||
|
from private_gpt.settings.settings import Settings
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@singleton
|
||||||
|
class LLMComponent:
|
||||||
|
llm: LLM
|
||||||
|
|
||||||
|
@inject
|
||||||
|
def __init__(self, settings: Settings) -> None:
|
||||||
|
llm_mode = settings.llm.mode
|
||||||
|
if settings.llm.tokenizer and settings.llm.mode != "mock":
|
||||||
|
# Try to download the tokenizer. If it fails, the LLM will still work
|
||||||
|
# using the default one, which is less accurate.
|
||||||
|
try:
|
||||||
|
set_global_tokenizer(
|
||||||
|
AutoTokenizer.from_pretrained(
|
||||||
|
pretrained_model_name_or_path=settings.llm.tokenizer,
|
||||||
|
cache_dir=str(models_cache_path),
|
||||||
|
token=settings.huggingface.access_token,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
f"Failed to download tokenizer {settings.llm.tokenizer}: {e!s}"
|
||||||
|
f"Please follow the instructions in the documentation to download it if needed: "
|
||||||
|
f"https://docs.privategpt.dev/installation/getting-started/troubleshooting#tokenizer-setup."
|
||||||
|
f"Falling back to default tokenizer."
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info("Initializing the LLM in mode=%s", llm_mode)
|
||||||
|
match settings.llm.mode:
|
||||||
|
case "llamacpp":
|
||||||
|
try:
|
||||||
|
from llama_index.llms.llama_cpp import LlamaCPP # type: ignore
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Local dependencies not found, install with `poetry install --extras llms-llama-cpp`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
prompt_style = get_prompt_style(settings.llm.prompt_style)
|
||||||
|
settings_kwargs = {
|
||||||
|
"tfs_z": settings.llamacpp.tfs_z, # ollama and llama-cpp
|
||||||
|
"top_k": settings.llamacpp.top_k, # ollama and llama-cpp
|
||||||
|
"top_p": settings.llamacpp.top_p, # ollama and llama-cpp
|
||||||
|
"repeat_penalty": settings.llamacpp.repeat_penalty, # ollama llama-cpp
|
||||||
|
"n_gpu_layers": -1,
|
||||||
|
"offload_kqv": True,
|
||||||
|
}
|
||||||
|
self.llm = LlamaCPP(
|
||||||
|
model_path=str(models_path / settings.llamacpp.llm_hf_model_file),
|
||||||
|
temperature=settings.llm.temperature,
|
||||||
|
max_new_tokens=settings.llm.max_new_tokens,
|
||||||
|
context_window=settings.llm.context_window,
|
||||||
|
generate_kwargs={},
|
||||||
|
callback_manager=LlamaIndexSettings.callback_manager,
|
||||||
|
# All to GPU
|
||||||
|
model_kwargs=settings_kwargs,
|
||||||
|
# transform inputs into Llama2 format
|
||||||
|
messages_to_prompt=prompt_style.messages_to_prompt,
|
||||||
|
completion_to_prompt=prompt_style.completion_to_prompt,
|
||||||
|
verbose=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
case "sagemaker":
|
||||||
|
try:
|
||||||
|
from private_gpt.components.llm.custom.sagemaker import SagemakerLLM
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Sagemaker dependencies not found, install with `poetry install --extras llms-sagemaker`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
self.llm = SagemakerLLM(
|
||||||
|
endpoint_name=settings.sagemaker.llm_endpoint_name,
|
||||||
|
max_new_tokens=settings.llm.max_new_tokens,
|
||||||
|
context_window=settings.llm.context_window,
|
||||||
|
)
|
||||||
|
case "openai":
|
||||||
|
try:
|
||||||
|
from llama_index.llms.openai import OpenAI # type: ignore
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"OpenAI dependencies not found, install with `poetry install --extras llms-openai`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
openai_settings = settings.openai
|
||||||
|
self.llm = OpenAI(
|
||||||
|
api_base=openai_settings.api_base,
|
||||||
|
api_key=openai_settings.api_key,
|
||||||
|
model=openai_settings.model,
|
||||||
|
)
|
||||||
|
case "openailike":
|
||||||
|
try:
|
||||||
|
from llama_index.llms.openai_like import OpenAILike # type: ignore
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"OpenAILike dependencies not found, install with `poetry install --extras llms-openai-like`"
|
||||||
|
) from e
|
||||||
|
prompt_style = get_prompt_style(settings.llm.prompt_style)
|
||||||
|
openai_settings = settings.openai
|
||||||
|
self.llm = OpenAILike(
|
||||||
|
api_base=openai_settings.api_base,
|
||||||
|
api_key=openai_settings.api_key,
|
||||||
|
model=openai_settings.model,
|
||||||
|
is_chat_model=True,
|
||||||
|
max_tokens=settings.llm.max_new_tokens,
|
||||||
|
api_version="",
|
||||||
|
temperature=settings.llm.temperature,
|
||||||
|
context_window=settings.llm.context_window,
|
||||||
|
messages_to_prompt=prompt_style.messages_to_prompt,
|
||||||
|
completion_to_prompt=prompt_style.completion_to_prompt,
|
||||||
|
tokenizer=settings.llm.tokenizer,
|
||||||
|
timeout=openai_settings.request_timeout,
|
||||||
|
reuse_client=False,
|
||||||
|
)
|
||||||
|
case "ollama":
|
||||||
|
try:
|
||||||
|
from llama_index.llms.ollama import Ollama # type: ignore
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Ollama dependencies not found, install with `poetry install --extras llms-ollama`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
ollama_settings = settings.ollama
|
||||||
|
|
||||||
|
settings_kwargs = {
|
||||||
|
"tfs_z": ollama_settings.tfs_z, # ollama and llama-cpp
|
||||||
|
"num_predict": ollama_settings.num_predict, # ollama only
|
||||||
|
"top_k": ollama_settings.top_k, # ollama and llama-cpp
|
||||||
|
"top_p": ollama_settings.top_p, # ollama and llama-cpp
|
||||||
|
"repeat_last_n": ollama_settings.repeat_last_n, # ollama
|
||||||
|
"repeat_penalty": ollama_settings.repeat_penalty, # ollama llama-cpp
|
||||||
|
}
|
||||||
|
|
||||||
|
# calculate llm model. If not provided tag, it will be use latest
|
||||||
|
model_name = (
|
||||||
|
ollama_settings.llm_model + ":latest"
|
||||||
|
if ":" not in ollama_settings.llm_model
|
||||||
|
else ollama_settings.llm_model
|
||||||
|
)
|
||||||
|
|
||||||
|
llm = Ollama(
|
||||||
|
model=model_name,
|
||||||
|
base_url=ollama_settings.api_base,
|
||||||
|
temperature=settings.llm.temperature,
|
||||||
|
context_window=settings.llm.context_window,
|
||||||
|
additional_kwargs=settings_kwargs,
|
||||||
|
request_timeout=ollama_settings.request_timeout,
|
||||||
|
)
|
||||||
|
|
||||||
|
if ollama_settings.autopull_models:
|
||||||
|
from private_gpt.utils.ollama import check_connection, pull_model
|
||||||
|
|
||||||
|
if not check_connection(llm.client):
|
||||||
|
raise ValueError(
|
||||||
|
f"Failed to connect to Ollama, "
|
||||||
|
f"check if Ollama server is running on {ollama_settings.api_base}"
|
||||||
|
)
|
||||||
|
pull_model(llm.client, model_name)
|
||||||
|
|
||||||
|
if (
|
||||||
|
ollama_settings.keep_alive
|
||||||
|
!= ollama_settings.model_fields["keep_alive"].default
|
||||||
|
):
|
||||||
|
# Modify Ollama methods to use the "keep_alive" field.
|
||||||
|
def add_keep_alive(func: Callable[..., Any]) -> Callable[..., Any]:
|
||||||
|
def wrapper(*args: Any, **kwargs: Any) -> Any:
|
||||||
|
kwargs["keep_alive"] = ollama_settings.keep_alive
|
||||||
|
return func(*args, **kwargs)
|
||||||
|
|
||||||
|
return wrapper
|
||||||
|
|
||||||
|
Ollama.chat = add_keep_alive(Ollama.chat) # type: ignore
|
||||||
|
Ollama.stream_chat = add_keep_alive(Ollama.stream_chat) # type: ignore
|
||||||
|
Ollama.complete = add_keep_alive(Ollama.complete) # type: ignore
|
||||||
|
Ollama.stream_complete = add_keep_alive(Ollama.stream_complete) # type: ignore
|
||||||
|
|
||||||
|
self.llm = llm
|
||||||
|
|
||||||
|
case "azopenai":
|
||||||
|
try:
|
||||||
|
from llama_index.llms.azure_openai import ( # type: ignore
|
||||||
|
AzureOpenAI,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Azure OpenAI dependencies not found, install with `poetry install --extras llms-azopenai`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
azopenai_settings = settings.azopenai
|
||||||
|
self.llm = AzureOpenAI(
|
||||||
|
model=azopenai_settings.llm_model,
|
||||||
|
deployment_name=azopenai_settings.llm_deployment_name,
|
||||||
|
api_key=azopenai_settings.api_key,
|
||||||
|
azure_endpoint=azopenai_settings.azure_endpoint,
|
||||||
|
api_version=azopenai_settings.api_version,
|
||||||
|
)
|
||||||
|
case "gemini":
|
||||||
|
try:
|
||||||
|
from llama_index.llms.gemini import ( # type: ignore
|
||||||
|
Gemini,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Google Gemini dependencies not found, install with `poetry install --extras llms-gemini`"
|
||||||
|
) from e
|
||||||
|
gemini_settings = settings.gemini
|
||||||
|
self.llm = Gemini(
|
||||||
|
model_name=gemini_settings.model, api_key=gemini_settings.api_key
|
||||||
|
)
|
||||||
|
case "mock":
|
||||||
|
self.llm = MockLLM()
|
||||||
310
pgpt/private_gpt/components/llm/prompt_helper.py
Normal file
310
pgpt/private_gpt/components/llm/prompt_helper.py
Normal file
@ -0,0 +1,310 @@
|
|||||||
|
import abc
|
||||||
|
import logging
|
||||||
|
from collections.abc import Sequence
|
||||||
|
from typing import Any, Literal
|
||||||
|
|
||||||
|
from llama_index.core.llms import ChatMessage, MessageRole
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class AbstractPromptStyle(abc.ABC):
|
||||||
|
"""Abstract class for prompt styles.
|
||||||
|
|
||||||
|
This class is used to format a series of messages into a prompt that can be
|
||||||
|
understood by the models. A series of messages represents the interaction(s)
|
||||||
|
between a user and an assistant. This series of messages can be considered as a
|
||||||
|
session between a user X and an assistant Y.This session holds, through the
|
||||||
|
messages, the state of the conversation. This session, to be understood by the
|
||||||
|
model, needs to be formatted into a prompt (i.e. a string that the models
|
||||||
|
can understand). Prompts can be formatted in different ways,
|
||||||
|
depending on the model.
|
||||||
|
|
||||||
|
The implementations of this class represent the different ways to format a
|
||||||
|
series of messages into a prompt.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
||||||
|
logger.debug("Initializing prompt_style=%s", self.__class__.__name__)
|
||||||
|
|
||||||
|
@abc.abstractmethod
|
||||||
|
def _messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
pass
|
||||||
|
|
||||||
|
@abc.abstractmethod
|
||||||
|
def _completion_to_prompt(self, completion: str) -> str:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
prompt = self._messages_to_prompt(messages)
|
||||||
|
logger.debug("Got for messages='%s' the prompt='%s'", messages, prompt)
|
||||||
|
return prompt
|
||||||
|
|
||||||
|
def completion_to_prompt(self, prompt: str) -> str:
|
||||||
|
completion = prompt # Fix: Llama-index parameter has to be named as prompt
|
||||||
|
prompt = self._completion_to_prompt(completion)
|
||||||
|
logger.debug("Got for completion='%s' the prompt='%s'", completion, prompt)
|
||||||
|
return prompt
|
||||||
|
|
||||||
|
|
||||||
|
class DefaultPromptStyle(AbstractPromptStyle):
|
||||||
|
"""Default prompt style that uses the defaults from llama_utils.
|
||||||
|
|
||||||
|
It basically passes None to the LLM, indicating it should use
|
||||||
|
the default functions.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
||||||
|
super().__init__(*args, **kwargs)
|
||||||
|
|
||||||
|
# Hacky way to override the functions
|
||||||
|
# Override the functions to be None, and pass None to the LLM.
|
||||||
|
self.messages_to_prompt = None # type: ignore[method-assign, assignment]
|
||||||
|
self.completion_to_prompt = None # type: ignore[method-assign, assignment]
|
||||||
|
|
||||||
|
def _messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
def _completion_to_prompt(self, completion: str) -> str:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
class Llama2PromptStyle(AbstractPromptStyle):
|
||||||
|
"""Simple prompt style that uses llama 2 prompt style.
|
||||||
|
|
||||||
|
Inspired by llama_index/legacy/llms/llama_utils.py
|
||||||
|
|
||||||
|
It transforms the sequence of messages into a prompt that should look like:
|
||||||
|
```text
|
||||||
|
<s> [INST] <<SYS>> your system prompt here. <</SYS>>
|
||||||
|
|
||||||
|
user message here [/INST] assistant (model) response here </s>
|
||||||
|
```
|
||||||
|
"""
|
||||||
|
|
||||||
|
BOS, EOS = "<s>", "</s>"
|
||||||
|
B_INST, E_INST = "[INST]", "[/INST]"
|
||||||
|
B_SYS, E_SYS = "<<SYS>>\n", "\n<</SYS>>\n\n"
|
||||||
|
DEFAULT_SYSTEM_PROMPT = """\
|
||||||
|
You are a helpful, respectful and honest assistant. \
|
||||||
|
Always answer as helpfully as possible and follow ALL given instructions. \
|
||||||
|
Do not speculate or make up information. \
|
||||||
|
Do not reference any given instructions or context. \
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
string_messages: list[str] = []
|
||||||
|
if messages[0].role == MessageRole.SYSTEM:
|
||||||
|
# pull out the system message (if it exists in messages)
|
||||||
|
system_message_str = messages[0].content or ""
|
||||||
|
messages = messages[1:]
|
||||||
|
else:
|
||||||
|
system_message_str = self.DEFAULT_SYSTEM_PROMPT
|
||||||
|
|
||||||
|
system_message_str = f"{self.B_SYS} {system_message_str.strip()} {self.E_SYS}"
|
||||||
|
|
||||||
|
for i in range(0, len(messages), 2):
|
||||||
|
# first message should always be a user
|
||||||
|
user_message = messages[i]
|
||||||
|
assert user_message.role == MessageRole.USER
|
||||||
|
|
||||||
|
if i == 0:
|
||||||
|
# make sure system prompt is included at the start
|
||||||
|
str_message = f"{self.BOS} {self.B_INST} {system_message_str} "
|
||||||
|
else:
|
||||||
|
# end previous user-assistant interaction
|
||||||
|
string_messages[-1] += f" {self.EOS}"
|
||||||
|
# no need to include system prompt
|
||||||
|
str_message = f"{self.BOS} {self.B_INST} "
|
||||||
|
|
||||||
|
# include user message content
|
||||||
|
str_message += f"{user_message.content} {self.E_INST}"
|
||||||
|
|
||||||
|
if len(messages) > (i + 1):
|
||||||
|
# if assistant message exists, add to str_message
|
||||||
|
assistant_message = messages[i + 1]
|
||||||
|
assert assistant_message.role == MessageRole.ASSISTANT
|
||||||
|
str_message += f" {assistant_message.content}"
|
||||||
|
|
||||||
|
string_messages.append(str_message)
|
||||||
|
|
||||||
|
return "".join(string_messages)
|
||||||
|
|
||||||
|
def _completion_to_prompt(self, completion: str) -> str:
|
||||||
|
system_prompt_str = self.DEFAULT_SYSTEM_PROMPT
|
||||||
|
|
||||||
|
return (
|
||||||
|
f"{self.BOS} {self.B_INST} {self.B_SYS} {system_prompt_str.strip()} {self.E_SYS} "
|
||||||
|
f"{completion.strip()} {self.E_INST}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class Llama3PromptStyle(AbstractPromptStyle):
|
||||||
|
r"""Template for Meta's Llama 3.1.
|
||||||
|
|
||||||
|
The format follows this structure:
|
||||||
|
<|begin_of_text|>
|
||||||
|
<|start_header_id|>system<|end_header_id|>
|
||||||
|
|
||||||
|
[System message content]<|eot_id|>
|
||||||
|
<|start_header_id|>user<|end_header_id|>
|
||||||
|
|
||||||
|
[User message content]<|eot_id|>
|
||||||
|
<|start_header_id|>assistant<|end_header_id|>
|
||||||
|
|
||||||
|
[Assistant message content]<|eot_id|>
|
||||||
|
...
|
||||||
|
(Repeat for each message, including possible 'ipython' role)
|
||||||
|
"""
|
||||||
|
|
||||||
|
BOS, EOS = "<|begin_of_text|>", "<|end_of_text|>"
|
||||||
|
B_INST, E_INST = "<|start_header_id|>", "<|end_header_id|>"
|
||||||
|
EOT = "<|eot_id|>"
|
||||||
|
B_SYS, E_SYS = "<|start_header_id|>system<|end_header_id|>", "<|eot_id|>"
|
||||||
|
ASSISTANT_INST = "<|start_header_id|>assistant<|end_header_id|>"
|
||||||
|
DEFAULT_SYSTEM_PROMPT = """\
|
||||||
|
You are a helpful, respectful and honest assistant. \
|
||||||
|
Always answer as helpfully as possible and follow ALL given instructions. \
|
||||||
|
Do not speculate or make up information. \
|
||||||
|
Do not reference any given instructions or context. \
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
prompt = ""
|
||||||
|
has_system_message = False
|
||||||
|
|
||||||
|
for i, message in enumerate(messages):
|
||||||
|
if not message or message.content is None:
|
||||||
|
continue
|
||||||
|
if message.role == MessageRole.SYSTEM:
|
||||||
|
prompt += f"{self.B_SYS}\n\n{message.content.strip()}{self.E_SYS}"
|
||||||
|
has_system_message = True
|
||||||
|
else:
|
||||||
|
role_header = f"{self.B_INST}{message.role.value}{self.E_INST}"
|
||||||
|
prompt += f"{role_header}\n\n{message.content.strip()}{self.EOT}"
|
||||||
|
|
||||||
|
# Add assistant header if the last message is not from the assistant
|
||||||
|
if i == len(messages) - 1 and message.role != MessageRole.ASSISTANT:
|
||||||
|
prompt += f"{self.ASSISTANT_INST}\n\n"
|
||||||
|
|
||||||
|
# Add default system prompt if no system message was provided
|
||||||
|
if not has_system_message:
|
||||||
|
prompt = (
|
||||||
|
f"{self.B_SYS}\n\n{self.DEFAULT_SYSTEM_PROMPT}{self.E_SYS}" + prompt
|
||||||
|
)
|
||||||
|
|
||||||
|
# TODO: Implement tool handling logic
|
||||||
|
|
||||||
|
return prompt
|
||||||
|
|
||||||
|
def _completion_to_prompt(self, completion: str) -> str:
|
||||||
|
return (
|
||||||
|
f"{self.B_SYS}\n\n{self.DEFAULT_SYSTEM_PROMPT}{self.E_SYS}"
|
||||||
|
f"{self.B_INST}user{self.E_INST}\n\n{completion.strip()}{self.EOT}"
|
||||||
|
f"{self.ASSISTANT_INST}\n\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TagPromptStyle(AbstractPromptStyle):
|
||||||
|
"""Tag prompt style (used by Vigogne) that uses the prompt style `<|ROLE|>`.
|
||||||
|
|
||||||
|
It transforms the sequence of messages into a prompt that should look like:
|
||||||
|
```text
|
||||||
|
<|system|>: your system prompt here.
|
||||||
|
<|user|>: user message here
|
||||||
|
(possibly with context and question)
|
||||||
|
<|assistant|>: assistant (model) response here.
|
||||||
|
```
|
||||||
|
|
||||||
|
FIXME: should we add surrounding `<s>` and `</s>` tags, like in llama2?
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
"""Format message to prompt with `<|ROLE|>: MSG` style."""
|
||||||
|
prompt = ""
|
||||||
|
for message in messages:
|
||||||
|
role = message.role
|
||||||
|
content = message.content or ""
|
||||||
|
message_from_user = f"<|{role.lower()}|>: {content.strip()}"
|
||||||
|
message_from_user += "\n"
|
||||||
|
prompt += message_from_user
|
||||||
|
# we are missing the last <|assistant|> tag that will trigger a completion
|
||||||
|
prompt += "<|assistant|>: "
|
||||||
|
return prompt
|
||||||
|
|
||||||
|
def _completion_to_prompt(self, completion: str) -> str:
|
||||||
|
return self._messages_to_prompt(
|
||||||
|
[ChatMessage(content=completion, role=MessageRole.USER)]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class MistralPromptStyle(AbstractPromptStyle):
|
||||||
|
def _messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
inst_buffer = []
|
||||||
|
text = ""
|
||||||
|
for message in messages:
|
||||||
|
if message.role == MessageRole.SYSTEM or message.role == MessageRole.USER:
|
||||||
|
inst_buffer.append(str(message.content).strip())
|
||||||
|
elif message.role == MessageRole.ASSISTANT:
|
||||||
|
text += "<s>[INST] " + "\n".join(inst_buffer) + " [/INST]"
|
||||||
|
text += " " + str(message.content).strip() + "</s>"
|
||||||
|
inst_buffer.clear()
|
||||||
|
else:
|
||||||
|
raise ValueError(f"Unknown message role {message.role}")
|
||||||
|
|
||||||
|
if len(inst_buffer) > 0:
|
||||||
|
text += "<s>[INST] " + "\n".join(inst_buffer) + " [/INST]"
|
||||||
|
|
||||||
|
return text
|
||||||
|
|
||||||
|
def _completion_to_prompt(self, completion: str) -> str:
|
||||||
|
return self._messages_to_prompt(
|
||||||
|
[ChatMessage(content=completion, role=MessageRole.USER)]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ChatMLPromptStyle(AbstractPromptStyle):
|
||||||
|
def _messages_to_prompt(self, messages: Sequence[ChatMessage]) -> str:
|
||||||
|
prompt = "<|im_start|>system\n"
|
||||||
|
for message in messages:
|
||||||
|
role = message.role
|
||||||
|
content = message.content or ""
|
||||||
|
if role.lower() == "system":
|
||||||
|
message_from_user = f"{content.strip()}"
|
||||||
|
prompt += message_from_user
|
||||||
|
elif role.lower() == "user":
|
||||||
|
prompt += "<|im_end|>\n<|im_start|>user\n"
|
||||||
|
message_from_user = f"{content.strip()}<|im_end|>\n"
|
||||||
|
prompt += message_from_user
|
||||||
|
prompt += "<|im_start|>assistant\n"
|
||||||
|
return prompt
|
||||||
|
|
||||||
|
def _completion_to_prompt(self, completion: str) -> str:
|
||||||
|
return self._messages_to_prompt(
|
||||||
|
[ChatMessage(content=completion, role=MessageRole.USER)]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_prompt_style(
|
||||||
|
prompt_style: (
|
||||||
|
Literal["default", "llama2", "llama3", "tag", "mistral", "chatml"] | None
|
||||||
|
)
|
||||||
|
) -> AbstractPromptStyle:
|
||||||
|
"""Get the prompt style to use from the given string.
|
||||||
|
|
||||||
|
:param prompt_style: The prompt style to use.
|
||||||
|
:return: The prompt style to use.
|
||||||
|
"""
|
||||||
|
if prompt_style is None or prompt_style == "default":
|
||||||
|
return DefaultPromptStyle()
|
||||||
|
elif prompt_style == "llama2":
|
||||||
|
return Llama2PromptStyle()
|
||||||
|
elif prompt_style == "llama3":
|
||||||
|
return Llama3PromptStyle()
|
||||||
|
elif prompt_style == "tag":
|
||||||
|
return TagPromptStyle()
|
||||||
|
elif prompt_style == "mistral":
|
||||||
|
return MistralPromptStyle()
|
||||||
|
elif prompt_style == "chatml":
|
||||||
|
return ChatMLPromptStyle()
|
||||||
|
raise ValueError(f"Unknown prompt_style='{prompt_style}'")
|
||||||
0
pgpt/private_gpt/components/node_store/__init__.py
Normal file
0
pgpt/private_gpt/components/node_store/__init__.py
Normal file
@ -0,0 +1,68 @@
|
|||||||
|
import logging
|
||||||
|
|
||||||
|
from injector import inject, singleton
|
||||||
|
from llama_index.core.storage.docstore import BaseDocumentStore, SimpleDocumentStore
|
||||||
|
from llama_index.core.storage.index_store import SimpleIndexStore
|
||||||
|
from llama_index.core.storage.index_store.types import BaseIndexStore
|
||||||
|
|
||||||
|
from private_gpt.paths import local_data_path
|
||||||
|
from private_gpt.settings.settings import Settings
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@singleton
|
||||||
|
class NodeStoreComponent:
|
||||||
|
index_store: BaseIndexStore
|
||||||
|
doc_store: BaseDocumentStore
|
||||||
|
|
||||||
|
@inject
|
||||||
|
def __init__(self, settings: Settings) -> None:
|
||||||
|
match settings.nodestore.database:
|
||||||
|
case "simple":
|
||||||
|
try:
|
||||||
|
self.index_store = SimpleIndexStore.from_persist_dir(
|
||||||
|
persist_dir=str(local_data_path)
|
||||||
|
)
|
||||||
|
except FileNotFoundError:
|
||||||
|
logger.debug("Local index store not found, creating a new one")
|
||||||
|
self.index_store = SimpleIndexStore()
|
||||||
|
|
||||||
|
try:
|
||||||
|
self.doc_store = SimpleDocumentStore.from_persist_dir(
|
||||||
|
persist_dir=str(local_data_path)
|
||||||
|
)
|
||||||
|
except FileNotFoundError:
|
||||||
|
logger.debug("Local document store not found, creating a new one")
|
||||||
|
self.doc_store = SimpleDocumentStore()
|
||||||
|
|
||||||
|
case "postgres":
|
||||||
|
try:
|
||||||
|
from llama_index.storage.docstore.postgres import ( # type: ignore
|
||||||
|
PostgresDocumentStore,
|
||||||
|
)
|
||||||
|
from llama_index.storage.index_store.postgres import ( # type: ignore
|
||||||
|
PostgresIndexStore,
|
||||||
|
)
|
||||||
|
except ImportError:
|
||||||
|
raise ImportError(
|
||||||
|
"Postgres dependencies not found, install with `poetry install --extras storage-nodestore-postgres`"
|
||||||
|
) from None
|
||||||
|
|
||||||
|
if settings.postgres is None:
|
||||||
|
raise ValueError("Postgres index/doc store settings not found.")
|
||||||
|
|
||||||
|
self.index_store = PostgresIndexStore.from_params(
|
||||||
|
**settings.postgres.model_dump(exclude_none=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
self.doc_store = PostgresDocumentStore.from_params(
|
||||||
|
**settings.postgres.model_dump(exclude_none=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
case _:
|
||||||
|
# Should be unreachable
|
||||||
|
# The settings validator should have caught this
|
||||||
|
raise ValueError(
|
||||||
|
f"Database {settings.nodestore.database} not supported"
|
||||||
|
)
|
||||||
106
pgpt/private_gpt/components/vector_store/batched_chroma.py
Normal file
106
pgpt/private_gpt/components/vector_store/batched_chroma.py
Normal file
@ -0,0 +1,106 @@
|
|||||||
|
from collections.abc import Generator, Sequence
|
||||||
|
from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
|
from llama_index.core.schema import BaseNode, MetadataMode
|
||||||
|
from llama_index.core.vector_stores.utils import node_to_metadata_dict
|
||||||
|
from llama_index.vector_stores.chroma import ChromaVectorStore # type: ignore
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Mapping
|
||||||
|
|
||||||
|
|
||||||
|
def chunk_list(
|
||||||
|
lst: Sequence[BaseNode], max_chunk_size: int
|
||||||
|
) -> Generator[Sequence[BaseNode], None, None]:
|
||||||
|
"""Yield successive max_chunk_size-sized chunks from lst.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
lst (List[BaseNode]): list of nodes with embeddings
|
||||||
|
max_chunk_size (int): max chunk size
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
Generator[List[BaseNode], None, None]: list of nodes with embeddings
|
||||||
|
"""
|
||||||
|
for i in range(0, len(lst), max_chunk_size):
|
||||||
|
yield lst[i : i + max_chunk_size]
|
||||||
|
|
||||||
|
|
||||||
|
class BatchedChromaVectorStore(ChromaVectorStore): # type: ignore
|
||||||
|
"""Chroma vector store, batching additions to avoid reaching the max batch limit.
|
||||||
|
|
||||||
|
In this vector store, embeddings are stored within a ChromaDB collection.
|
||||||
|
|
||||||
|
During query time, the index uses ChromaDB to query for the top
|
||||||
|
k most similar nodes.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
chroma_client (from chromadb.api.API):
|
||||||
|
API instance
|
||||||
|
chroma_collection (chromadb.api.models.Collection.Collection):
|
||||||
|
ChromaDB collection instance
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
chroma_client: Any | None
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
chroma_client: Any,
|
||||||
|
chroma_collection: Any,
|
||||||
|
host: str | None = None,
|
||||||
|
port: str | None = None,
|
||||||
|
ssl: bool = False,
|
||||||
|
headers: dict[str, str] | None = None,
|
||||||
|
collection_kwargs: dict[Any, Any] | None = None,
|
||||||
|
) -> None:
|
||||||
|
super().__init__(
|
||||||
|
chroma_collection=chroma_collection,
|
||||||
|
host=host,
|
||||||
|
port=port,
|
||||||
|
ssl=ssl,
|
||||||
|
headers=headers,
|
||||||
|
collection_kwargs=collection_kwargs or {},
|
||||||
|
)
|
||||||
|
self.chroma_client = chroma_client
|
||||||
|
|
||||||
|
def add(self, nodes: Sequence[BaseNode], **add_kwargs: Any) -> list[str]:
|
||||||
|
"""Add nodes to index, batching the insertion to avoid issues.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
nodes: List[BaseNode]: list of nodes with embeddings
|
||||||
|
add_kwargs: _
|
||||||
|
"""
|
||||||
|
if not self.chroma_client:
|
||||||
|
raise ValueError("Client not initialized")
|
||||||
|
|
||||||
|
if not self._collection:
|
||||||
|
raise ValueError("Collection not initialized")
|
||||||
|
|
||||||
|
max_chunk_size = self.chroma_client.max_batch_size
|
||||||
|
node_chunks = chunk_list(nodes, max_chunk_size)
|
||||||
|
|
||||||
|
all_ids = []
|
||||||
|
for node_chunk in node_chunks:
|
||||||
|
embeddings: list[Sequence[float]] = []
|
||||||
|
metadatas: list[Mapping[str, Any]] = []
|
||||||
|
ids = []
|
||||||
|
documents = []
|
||||||
|
for node in node_chunk:
|
||||||
|
embeddings.append(node.get_embedding())
|
||||||
|
metadatas.append(
|
||||||
|
node_to_metadata_dict(
|
||||||
|
node, remove_text=True, flat_metadata=self.flat_metadata
|
||||||
|
)
|
||||||
|
)
|
||||||
|
ids.append(node.node_id)
|
||||||
|
documents.append(node.get_content(metadata_mode=MetadataMode.NONE))
|
||||||
|
|
||||||
|
self._collection.add(
|
||||||
|
embeddings=embeddings,
|
||||||
|
ids=ids,
|
||||||
|
metadatas=metadatas,
|
||||||
|
documents=documents,
|
||||||
|
)
|
||||||
|
all_ids.extend(ids)
|
||||||
|
|
||||||
|
return all_ids
|
||||||
@ -0,0 +1,217 @@
|
|||||||
|
import logging
|
||||||
|
import typing
|
||||||
|
|
||||||
|
from injector import inject, singleton
|
||||||
|
from llama_index.core.indices.vector_store import VectorIndexRetriever, VectorStoreIndex
|
||||||
|
from llama_index.core.vector_stores.types import (
|
||||||
|
BasePydanticVectorStore,
|
||||||
|
FilterCondition,
|
||||||
|
MetadataFilter,
|
||||||
|
MetadataFilters,
|
||||||
|
)
|
||||||
|
|
||||||
|
from private_gpt.open_ai.extensions.context_filter import ContextFilter
|
||||||
|
from private_gpt.paths import local_data_path
|
||||||
|
from private_gpt.settings.settings import Settings
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _doc_id_metadata_filter(
|
||||||
|
context_filter: ContextFilter | None,
|
||||||
|
) -> MetadataFilters:
|
||||||
|
filters = MetadataFilters(filters=[], condition=FilterCondition.OR)
|
||||||
|
|
||||||
|
if context_filter is not None and context_filter.docs_ids is not None:
|
||||||
|
for doc_id in context_filter.docs_ids:
|
||||||
|
filters.filters.append(MetadataFilter(key="doc_id", value=doc_id))
|
||||||
|
|
||||||
|
return filters
|
||||||
|
|
||||||
|
|
||||||
|
@singleton
|
||||||
|
class VectorStoreComponent:
|
||||||
|
settings: Settings
|
||||||
|
vector_store: BasePydanticVectorStore
|
||||||
|
|
||||||
|
@inject
|
||||||
|
def __init__(self, settings: Settings) -> None:
|
||||||
|
self.settings = settings
|
||||||
|
match settings.vectorstore.database:
|
||||||
|
case "postgres":
|
||||||
|
try:
|
||||||
|
from llama_index.vector_stores.postgres import ( # type: ignore
|
||||||
|
PGVectorStore,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Postgres dependencies not found, install with `poetry install --extras vector-stores-postgres`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
if settings.postgres is None:
|
||||||
|
raise ValueError(
|
||||||
|
"Postgres settings not found. Please provide settings."
|
||||||
|
)
|
||||||
|
|
||||||
|
self.vector_store = typing.cast(
|
||||||
|
BasePydanticVectorStore,
|
||||||
|
PGVectorStore.from_params(
|
||||||
|
**settings.postgres.model_dump(exclude_none=True),
|
||||||
|
table_name="embeddings",
|
||||||
|
embed_dim=settings.embedding.embed_dim,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
case "chroma":
|
||||||
|
try:
|
||||||
|
import chromadb # type: ignore
|
||||||
|
from chromadb.config import ( # type: ignore
|
||||||
|
Settings as ChromaSettings,
|
||||||
|
)
|
||||||
|
|
||||||
|
from private_gpt.components.vector_store.batched_chroma import (
|
||||||
|
BatchedChromaVectorStore,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"ChromaDB dependencies not found, install with `poetry install --extras vector-stores-chroma`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
chroma_settings = ChromaSettings(anonymized_telemetry=False)
|
||||||
|
chroma_client = chromadb.PersistentClient(
|
||||||
|
path=str((local_data_path / "chroma_db").absolute()),
|
||||||
|
settings=chroma_settings,
|
||||||
|
)
|
||||||
|
chroma_collection = chroma_client.get_or_create_collection(
|
||||||
|
"make_this_parameterizable_per_api_call"
|
||||||
|
) # TODO
|
||||||
|
|
||||||
|
self.vector_store = typing.cast(
|
||||||
|
BasePydanticVectorStore,
|
||||||
|
BatchedChromaVectorStore(
|
||||||
|
chroma_client=chroma_client, chroma_collection=chroma_collection
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
case "qdrant":
|
||||||
|
try:
|
||||||
|
from llama_index.vector_stores.qdrant import ( # type: ignore
|
||||||
|
QdrantVectorStore,
|
||||||
|
)
|
||||||
|
from qdrant_client import QdrantClient # type: ignore
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Qdrant dependencies not found, install with `poetry install --extras vector-stores-qdrant`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
if settings.qdrant is None:
|
||||||
|
logger.info(
|
||||||
|
"Qdrant config not found. Using default settings."
|
||||||
|
"Trying to connect to Qdrant at localhost:6333."
|
||||||
|
)
|
||||||
|
client = QdrantClient()
|
||||||
|
else:
|
||||||
|
client = QdrantClient(
|
||||||
|
**settings.qdrant.model_dump(exclude_none=True)
|
||||||
|
)
|
||||||
|
self.vector_store = typing.cast(
|
||||||
|
BasePydanticVectorStore,
|
||||||
|
QdrantVectorStore(
|
||||||
|
client=client,
|
||||||
|
collection_name="make_this_parameterizable_per_api_call",
|
||||||
|
), # TODO
|
||||||
|
)
|
||||||
|
|
||||||
|
case "milvus":
|
||||||
|
try:
|
||||||
|
from llama_index.vector_stores.milvus import ( # type: ignore
|
||||||
|
MilvusVectorStore,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"Milvus dependencies not found, install with `poetry install --extras vector-stores-milvus`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
if settings.milvus is None:
|
||||||
|
logger.info(
|
||||||
|
"Milvus config not found. Using default settings.\n"
|
||||||
|
"Trying to connect to Milvus at local_data/private_gpt/milvus/milvus_local.db "
|
||||||
|
"with collection 'make_this_parameterizable_per_api_call'."
|
||||||
|
)
|
||||||
|
|
||||||
|
self.vector_store = typing.cast(
|
||||||
|
BasePydanticVectorStore,
|
||||||
|
MilvusVectorStore(
|
||||||
|
dim=settings.embedding.embed_dim,
|
||||||
|
collection_name="make_this_parameterizable_per_api_call",
|
||||||
|
overwrite=True,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
else:
|
||||||
|
self.vector_store = typing.cast(
|
||||||
|
BasePydanticVectorStore,
|
||||||
|
MilvusVectorStore(
|
||||||
|
dim=settings.embedding.embed_dim,
|
||||||
|
uri=settings.milvus.uri,
|
||||||
|
token=settings.milvus.token,
|
||||||
|
collection_name=settings.milvus.collection_name,
|
||||||
|
overwrite=settings.milvus.overwrite,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
case "clickhouse":
|
||||||
|
try:
|
||||||
|
from clickhouse_connect import ( # type: ignore
|
||||||
|
get_client,
|
||||||
|
)
|
||||||
|
from llama_index.vector_stores.clickhouse import ( # type: ignore
|
||||||
|
ClickHouseVectorStore,
|
||||||
|
)
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"ClickHouse dependencies not found, install with `poetry install --extras vector-stores-clickhouse`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
if settings.clickhouse is None:
|
||||||
|
raise ValueError(
|
||||||
|
"ClickHouse settings not found. Please provide settings."
|
||||||
|
)
|
||||||
|
|
||||||
|
clickhouse_client = get_client(
|
||||||
|
host=settings.clickhouse.host,
|
||||||
|
port=settings.clickhouse.port,
|
||||||
|
username=settings.clickhouse.username,
|
||||||
|
password=settings.clickhouse.password,
|
||||||
|
)
|
||||||
|
self.vector_store = ClickHouseVectorStore(
|
||||||
|
clickhouse_client=clickhouse_client
|
||||||
|
)
|
||||||
|
case _:
|
||||||
|
# Should be unreachable
|
||||||
|
# The settings validator should have caught this
|
||||||
|
raise ValueError(
|
||||||
|
f"Vectorstore database {settings.vectorstore.database} not supported"
|
||||||
|
)
|
||||||
|
|
||||||
|
def get_retriever(
|
||||||
|
self,
|
||||||
|
index: VectorStoreIndex,
|
||||||
|
context_filter: ContextFilter | None = None,
|
||||||
|
similarity_top_k: int = 2,
|
||||||
|
) -> VectorIndexRetriever:
|
||||||
|
# This way we support qdrant (using doc_ids) and the rest (using filters)
|
||||||
|
return VectorIndexRetriever(
|
||||||
|
index=index,
|
||||||
|
similarity_top_k=similarity_top_k,
|
||||||
|
doc_ids=context_filter.docs_ids if context_filter else None,
|
||||||
|
filters=(
|
||||||
|
_doc_id_metadata_filter(context_filter)
|
||||||
|
if self.settings.vectorstore.database != "qdrant"
|
||||||
|
else None
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
def close(self) -> None:
|
||||||
|
if hasattr(self.vector_store.client, "close"):
|
||||||
|
self.vector_store.client.close()
|
||||||
3
pgpt/private_gpt/constants.py
Normal file
3
pgpt/private_gpt/constants.py
Normal file
@ -0,0 +1,3 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
PROJECT_ROOT_PATH: Path = Path(__file__).parents[1]
|
||||||
19
pgpt/private_gpt/di.py
Normal file
19
pgpt/private_gpt/di.py
Normal file
@ -0,0 +1,19 @@
|
|||||||
|
from injector import Injector
|
||||||
|
|
||||||
|
from private_gpt.settings.settings import Settings, unsafe_typed_settings
|
||||||
|
|
||||||
|
|
||||||
|
def create_application_injector() -> Injector:
|
||||||
|
_injector = Injector(auto_bind=True)
|
||||||
|
_injector.binder.bind(Settings, to=unsafe_typed_settings)
|
||||||
|
return _injector
|
||||||
|
|
||||||
|
|
||||||
|
"""
|
||||||
|
Global injector for the application.
|
||||||
|
|
||||||
|
Avoid using this reference, it will make your code harder to test.
|
||||||
|
|
||||||
|
Instead, use the `request.state.injector` reference, which is bound to every request
|
||||||
|
"""
|
||||||
|
global_injector: Injector = create_application_injector()
|
||||||
69
pgpt/private_gpt/launcher.py
Normal file
69
pgpt/private_gpt/launcher.py
Normal file
@ -0,0 +1,69 @@
|
|||||||
|
"""FastAPI app creation, logger configuration and main API routes."""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from fastapi import Depends, FastAPI, Request
|
||||||
|
from fastapi.middleware.cors import CORSMiddleware
|
||||||
|
from injector import Injector
|
||||||
|
from llama_index.core.callbacks import CallbackManager
|
||||||
|
from llama_index.core.callbacks.global_handlers import create_global_handler
|
||||||
|
from llama_index.core.settings import Settings as LlamaIndexSettings
|
||||||
|
|
||||||
|
from private_gpt.server.chat.chat_router import chat_router
|
||||||
|
from private_gpt.server.chunks.chunks_router import chunks_router
|
||||||
|
from private_gpt.server.completions.completions_router import completions_router
|
||||||
|
from private_gpt.server.embeddings.embeddings_router import embeddings_router
|
||||||
|
from private_gpt.server.health.health_router import health_router
|
||||||
|
from private_gpt.server.ingest.ingest_router import ingest_router
|
||||||
|
from private_gpt.server.recipes.summarize.summarize_router import summarize_router
|
||||||
|
from private_gpt.settings.settings import Settings
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def create_app(root_injector: Injector) -> FastAPI:
|
||||||
|
|
||||||
|
# Start the API
|
||||||
|
async def bind_injector_to_request(request: Request) -> None:
|
||||||
|
request.state.injector = root_injector
|
||||||
|
|
||||||
|
app = FastAPI(dependencies=[Depends(bind_injector_to_request)])
|
||||||
|
|
||||||
|
app.include_router(completions_router)
|
||||||
|
app.include_router(chat_router)
|
||||||
|
app.include_router(chunks_router)
|
||||||
|
app.include_router(ingest_router)
|
||||||
|
app.include_router(summarize_router)
|
||||||
|
app.include_router(embeddings_router)
|
||||||
|
app.include_router(health_router)
|
||||||
|
|
||||||
|
# Add LlamaIndex simple observability
|
||||||
|
global_handler = create_global_handler("simple")
|
||||||
|
if global_handler:
|
||||||
|
LlamaIndexSettings.callback_manager = CallbackManager([global_handler])
|
||||||
|
|
||||||
|
settings = root_injector.get(Settings)
|
||||||
|
if settings.server.cors.enabled:
|
||||||
|
logger.debug("Setting up CORS middleware")
|
||||||
|
app.add_middleware(
|
||||||
|
CORSMiddleware,
|
||||||
|
allow_credentials=settings.server.cors.allow_credentials,
|
||||||
|
allow_origins=settings.server.cors.allow_origins,
|
||||||
|
allow_origin_regex=settings.server.cors.allow_origin_regex,
|
||||||
|
allow_methods=settings.server.cors.allow_methods,
|
||||||
|
allow_headers=settings.server.cors.allow_headers,
|
||||||
|
)
|
||||||
|
|
||||||
|
if settings.ui.enabled:
|
||||||
|
logger.debug("Importing the UI module")
|
||||||
|
try:
|
||||||
|
from private_gpt.ui.ui import PrivateGptUi
|
||||||
|
except ImportError as e:
|
||||||
|
raise ImportError(
|
||||||
|
"UI dependencies not found, install with `poetry install --extras ui`"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
ui = root_injector.get(PrivateGptUi)
|
||||||
|
ui.mount_in_app(app, settings.ui.path)
|
||||||
|
|
||||||
|
return app
|
||||||
6
pgpt/private_gpt/main.py
Normal file
6
pgpt/private_gpt/main.py
Normal file
@ -0,0 +1,6 @@
|
|||||||
|
"""FastAPI app creation, logger configuration and main API routes."""
|
||||||
|
|
||||||
|
from private_gpt.di import global_injector
|
||||||
|
from private_gpt.launcher import create_app
|
||||||
|
|
||||||
|
app = create_app(global_injector)
|
||||||
1
pgpt/private_gpt/open_ai/__init__.py
Normal file
1
pgpt/private_gpt/open_ai/__init__.py
Normal file
@ -0,0 +1 @@
|
|||||||
|
"""OpenAI compatibility utilities."""
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Loading…
x
Reference in New Issue
Block a user