YELY_AI_Module / tools /check_seen_image.py
danielxdata's picture
Module IA YELY - CRNN fine-tune, API FastAPI, interface demo
b510add
Raw
History Blame Contribute Delete
4.26 kB
"""Vérifie si une image a déjà été utilisée pour l'entraînement/validation
du CRNN (annotator/annotations/annotations.json), afin de choisir des photos
réellement inédites pour les tests de démonstration.
Compare par hash perceptuel (pHash) pour détecter aussi les quasi-doublons
(même écran re-photographié, recadré, recompressé) — pas seulement les noms
de fichiers identiques.
Usage :
python tools/check_seen_image.py chemin/vers/photo.jpg [autre.jpg ...]
"""
import argparse
import json
import os
import sys
import numpy as np
from PIL import Image
REPO_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
ANNOTATIONS_FILE = os.path.join(REPO_ROOT, "annotator", "annotations", "annotations.json")
HASH_SIZE = 16
SIMILARITY_THRESHOLD = 10 # distance de Hamming ; en dessous = quasi-identique
def phash(image_path: str, hash_size: int = HASH_SIZE) -> np.ndarray:
img = Image.open(image_path).convert("L").resize(
(hash_size + 1, hash_size), Image.LANCZOS
)
pixels = np.asarray(img, dtype=np.float32)
diff = pixels[:, 1:] > pixels[:, :-1]
return diff.flatten()
def hamming_distance(a: np.ndarray, b: np.ndarray) -> int:
return int(np.count_nonzero(a != b))
def _index_repo_filenames() -> dict:
"""Indexe une fois tous les fichiers image du dépôt par nom de fichier,
pour retrouver les sources annotées même quand `image_path` ne pointe
plus vers un chemin valide (dossiers déplacés/renommés)."""
exts = {".jpg", ".jpeg", ".png", ".bmp", ".webp", ".tiff"}
index: dict = {}
skip_dirs = {".git", ".env", "__pycache__", "node_modules", "dataset_lines",
"dataset_doctr", "crops", "outputs", "outputs_test", "photos"}
for root, dirs, files in os.walk(REPO_ROOT):
dirs[:] = [d for d in dirs if d not in skip_dirs]
for f in files:
if os.path.splitext(f)[1].lower() in exts:
index.setdefault(f, os.path.join(root, f))
return index
def build_reference_hashes() -> dict:
"""Calcule les hash de toutes les images sources annotées (introuvables
localement sont ignorées silencieusement)."""
if not os.path.exists(ANNOTATIONS_FILE):
raise FileNotFoundError(f"Introuvable : {ANNOTATIONS_FILE}")
ann = json.load(open(ANNOTATIONS_FILE, encoding="utf-8"))
filename_index = _index_repo_filenames()
ann_dir = os.path.dirname(ANNOTATIONS_FILE)
refs = {}
for name, entry in ann.items():
rel_path = entry.get("image_path", "")
candidates = [
os.path.normpath(os.path.join(ann_dir, rel_path)) if rel_path else None,
os.path.normpath(os.path.join(REPO_ROOT, "images", name)),
filename_index.get(name),
]
for c in candidates:
if c and os.path.exists(c):
try:
refs[name] = phash(c)
except Exception:
pass
break
return refs
def check_image(path: str, refs: dict) -> None:
try:
h = phash(path)
except Exception as e:
print(f"{path} : impossible de lire l'image ({e})")
return
best_name, best_dist = None, None
for name, ref_hash in refs.items():
d = hamming_distance(h, ref_hash)
if best_dist is None or d < best_dist:
best_dist, best_name = d, name
if best_dist is not None and best_dist <= SIMILARITY_THRESHOLD:
print(f"{os.path.basename(path)} : DEJA VU (proche de '{best_name}', "
f"distance={best_dist}/{len(h)}) -> ne pas utiliser pour tester la generalisation.")
else:
dist_info = f"(plus proche : '{best_name}', distance={best_dist})" if best_name else ""
print(f"{os.path.basename(path)} : INEDITE {dist_info} -> bon candidat pour le test.")
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("images", nargs="+", help="Chemins des images à vérifier")
args = parser.parse_args()
refs = build_reference_hashes()
print(f"{len(refs)} images de référence indexées (annotations.json).\n")
for path in args.images:
check_image(path, refs)
if __name__ == "__main__":
main()