first commit

This commit is contained in:
julien
2026-08-05 13:53:18 +02:00
commit 91772b907c
70 changed files with 3762 additions and 0 deletions
+61
View File
@@ -0,0 +1,61 @@
"""Définitions communes au générateur et au validateur de contenus."""
SECTION_ALIASES = {
"article": "articles",
"articles": "articles",
"grammaire": "grammaire",
"signe": "signes",
"signes": "signes",
"texte": "textes",
"textes": "textes",
"vocabulaire": "vocabulaire",
}
VOCABULARY_FIELDS = {
"nom": ("meaning", "logograms", "gender", "bound", "plural"),
"adjectif": (
"meaning",
"logograms",
"bound",
"feminine",
"masculine_plural",
"feminine_plural",
"predicative",
),
"verbe": (
"meaning",
"logograms",
"root",
"stem",
"verb_class",
"vowel_class",
"preterite",
"durative",
"perfect",
"imperative",
"participle",
"verbal_adjective",
),
"pronom": (
"meaning",
"logograms",
"pronoun_type",
"person",
"gender",
"number",
),
"préposition": ("meaning", "logograms", "governs"),
"adverbe": ("meaning", "logograms", "function"),
"conjonction": ("meaning", "logograms", "function"),
"particule": ("meaning", "logograms", "function"),
"numéral": (
"meaning",
"logograms",
"numeral_type",
"value",
"gender",
"feminine",
),
}
KNOWN_SECTIONS = frozenset(SECTION_ALIASES.values())
+286
View File
@@ -0,0 +1,286 @@
#!/usr/bin/env python3
"""Vérifie la structure des contenus, le vocabulaire et les signes."""
import re
import sys
import tomllib
import unicodedata
from datetime import date
from pathlib import Path
from typing import Any
from _schema import KNOWN_SECTIONS, VOCABULARY_FIELDS
ROOT_DIR = Path(__file__).resolve().parent.parent
CONTENT_DIR = ROOT_DIR / "src" / "content"
MZL_FILENAME_RE = re.compile(r"mzl-([0-9]{3})\.md")
MZL_VALUE_RE = re.compile(r"[0-9]{3}")
class Validator:
def __init__(self) -> None:
self.errors = 0
def error(self, path: Path | None, message: str) -> None:
if path is None:
label = ""
else:
label = f"{path.relative_to(ROOT_DIR).as_posix()} : "
print(f"ERREUR : {label}{message}", file=sys.stderr)
self.errors += 1
def finish(self) -> int:
print(f"\nRésultat : {self.errors} erreur(s).")
return 1 if self.errors else 0
def is_non_empty_string(value: Any) -> bool:
return isinstance(value, str) and bool(value.strip())
def read_front_matter(path: Path, validator: Validator) -> dict[str, Any] | None:
try:
text = path.read_text(encoding="utf-8")
except UnicodeDecodeError as exc:
validator.error(path, f"fichier non valide en UTF-8 ({exc})")
return None
except OSError as exc:
validator.error(path, f"lecture impossible ({exc})")
return None
if not unicodedata.is_normalized("NFC", text):
validator.error(path, "contenu non normalisé en Unicode NFC")
lines = text.splitlines()
if not lines or lines[0] != "+++":
validator.error(path, "front matter TOML initial absent")
return None
try:
closing_index = lines.index("+++", 1)
except ValueError:
validator.error(path, "délimiteur final +++ absent")
return None
try:
return tomllib.loads("\n".join(lines[1:closing_index]))
except tomllib.TOMLDecodeError as exc:
validator.error(path, f"front matter TOML invalide ({exc})")
return None
def require_table(
metadata: dict[str, Any],
key: str,
path: Path,
validator: Validator,
) -> dict[str, Any] | None:
value = metadata.get(key)
if not isinstance(value, dict):
validator.error(path, f"{key} doit être une table TOML")
return None
return value
def require_string(
metadata: dict[str, Any],
key: str,
path: Path,
validator: Validator,
) -> None:
if not is_non_empty_string(metadata.get(key)):
validator.error(path, f"{key} doit être une chaîne non vide")
def require_date(
metadata: dict[str, Any],
path: Path,
validator: Validator,
) -> None:
if not isinstance(metadata.get("date"), date):
validator.error(path, "date doit être une date TOML valide")
def check_structure(
path: Path,
metadata: dict[str, Any],
validator: Validator,
) -> tuple[str, bool] | None:
relative_path = path.relative_to(CONTENT_DIR)
section = relative_path.parts[0] if len(relative_path.parts) > 1 else ""
if not unicodedata.is_normalized("NFC", relative_path.as_posix()):
validator.error(path, "chemin non normalisé en Unicode NFC")
if section and section not in KNOWN_SECTIONS:
validator.error(path, f"section inconnue « {section} »")
if path.name == "_index.md":
require_string(metadata, "title", path, validator)
return None
draft = metadata.get("draft")
if not isinstance(draft, bool):
validator.error(path, "draft doit être un booléen")
return None
if draft:
return section, True
require_string(metadata, "title", path, validator)
if section:
require_date(metadata, path, validator)
return section, False
def vocabulary_nature(
metadata: dict[str, Any],
path: Path,
validator: Validator,
) -> str | None:
taxonomies = require_table(metadata, "taxonomies", path, validator)
if taxonomies is None:
return None
natures = taxonomies.get("natures")
if not isinstance(natures, list) or len(natures) != 1:
validator.error(path, "taxonomies.natures doit contenir exactement une nature")
return None
nature = natures[0]
if not isinstance(nature, str):
validator.error(path, "taxonomies.natures doit contenir une chaîne")
return None
if nature not in VOCABULARY_FIELDS:
validator.error(path, f"nature grammaticale inconnue « {nature} »")
return None
return nature
def check_vocabulary(
path: Path,
metadata: dict[str, Any],
validator: Validator,
) -> None:
nature = vocabulary_nature(metadata, path, validator)
extra = require_table(metadata, "extra", path, validator)
if nature is None or extra is None:
return
expected_fields = VOCABULARY_FIELDS[nature]
expected_keys = set(expected_fields)
actual_keys = set(extra)
missing_fields = [field for field in expected_fields if field not in actual_keys]
if missing_fields:
validator.error(
path,
"champs extra manquants pour "
f"la nature « {nature} » : {', '.join(missing_fields)}",
)
unexpected_fields = sorted(actual_keys - expected_keys)
if unexpected_fields:
validator.error(
path,
"champs extra inattendus pour "
f"la nature « {nature} » : {', '.join(unexpected_fields)}",
)
for field in expected_fields:
if field not in extra:
continue
value = extra[field]
if field == "logograms":
if not isinstance(value, list) or not all(
is_non_empty_string(item) for item in value
):
validator.error(
path,
"extra.logograms doit être un tableau de chaînes non vides",
)
elif not isinstance(value, str):
validator.error(path, f"extra.{field} doit être une chaîne")
required_fields = ("meaning", "stem") if nature == "verbe" else ("meaning",)
for field in required_fields:
value = extra.get(field)
if isinstance(value, str) and not value.strip():
validator.error(path, f"extra.{field} doit être une chaîne non vide")
def filename_mzl(path: Path, validator: Validator) -> str | None:
match = MZL_FILENAME_RE.fullmatch(path.name)
if match is None:
validator.error(path, "le nom du fichier doit suivre le format mzl-XXX.md")
return None
value = match.group(1)
if int(value) == 0:
validator.error(path, "le numéro MZL doit être compris entre 001 et 999")
return None
return value
def check_sign(
path: Path,
metadata: dict[str, Any],
expected_mzl: str | None,
validator: Validator,
) -> None:
extra = require_table(metadata, "extra", path, validator)
if extra is None:
return
require_string(extra, "sign", path, validator)
declared_mzl = extra.get("mzl")
if not isinstance(declared_mzl, str) or MZL_VALUE_RE.fullmatch(declared_mzl) is None:
validator.error(path, "extra.mzl doit être une chaîne de trois chiffres ASCII")
return
if int(declared_mzl) == 0:
validator.error(path, "extra.mzl doit être compris entre 001 et 999")
return
if expected_mzl is not None and declared_mzl != expected_mzl:
validator.error(
path,
f'extra.mzl vaut "{declared_mzl}", mais le fichier indique MZL {expected_mzl}',
)
def main() -> int:
validator = Validator()
if not CONTENT_DIR.is_dir():
validator.error(None, f"répertoire introuvable : {CONTENT_DIR}")
return validator.finish()
print("Vérification des contenus…", flush=True)
for path in sorted(CONTENT_DIR.rglob("*.md")):
metadata = read_front_matter(path, validator)
if metadata is None:
continue
structure = check_structure(path, metadata, validator)
if structure is None:
continue
section, is_draft = structure
expected_mzl = filename_mzl(path, validator) if section == "signes" else None
if is_draft:
continue
if section == "vocabulaire":
check_vocabulary(path, metadata, validator)
elif section == "signes":
check_sign(path, metadata, expected_mzl, validator)
return validator.finish()
if __name__ == "__main__":
raise SystemExit(main())
+268
View File
@@ -0,0 +1,268 @@
#!/usr/bin/env python3
"""Crée un brouillon Markdown adapté à une section du site."""
import argparse
import json
import re
import sys
import unicodedata
from datetime import datetime
from pathlib import Path
from zoneinfo import ZoneInfo, ZoneInfoNotFoundError
from _schema import SECTION_ALIASES, VOCABULARY_FIELDS
ROOT_DIR = Path(__file__).resolve().parent.parent
CONTENT_DIR = ROOT_DIR / "src" / "content"
TIMEZONE_NAME = "Europe/Paris"
MZL_RE = re.compile(r"(?:mzl-)?([0-9]{1,3})")
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Créer un brouillon Markdown horodaté et prêt à compléter."
)
parser.add_argument(
"section",
choices=sorted(SECTION_ALIASES),
help="section de destination",
)
parser.add_argument(
"identifiant",
help="nom du fichier sans .md ; pour un signe, numéro MZL ou mzl-XXX",
)
parser.add_argument(
"--title",
help="titre initial ; déduit de lidentifiant par défaut",
)
parser.add_argument(
"--nature",
choices=tuple(VOCABULARY_FIELDS),
help="nature grammaticale dune fiche de vocabulaire",
)
return parser.parse_args()
def normalize(value: str) -> str:
return unicodedata.normalize("NFC", value.strip())
def toml_string(value: str) -> str:
return json.dumps(value, ensure_ascii=False)
def current_timestamp() -> str:
try:
timezone = ZoneInfo(TIMEZONE_NAME)
except ZoneInfoNotFoundError as exc:
raise RuntimeError(
f"fuseau {TIMEZONE_NAME} indisponible ; installer les données de fuseaux horaires"
) from exc
return datetime.now(timezone).isoformat(timespec="seconds")
def validate_identifier(identifier: str, *, allow_leading_hyphen: bool = False) -> str:
"""Valide un identifiant déjà écrit sous sa forme de nom de fichier."""
value = normalize(identifier)
if value.startswith("-") and not allow_leading_hyphen:
raise ValueError("seul le vocabulaire accepte un identifiant commençant par un tiret")
body = value[1:] if value.startswith("-") else value
if not body or value in {".", "..", "_index"}:
raise ValueError("identifiant de fichier invalide")
if value != value.lower():
raise ValueError("lidentifiant doit être écrit en minuscules")
if value.endswith("-") or "--" in value:
raise ValueError(
"lidentifiant ne doit pas finir par un tiret ni en contenir deux de suite"
)
if not all(character.isalnum() or character == "-" for character in body):
raise ValueError(
"lidentifiant ne doit contenir que des lettres, des chiffres et des tirets"
)
return value
def parse_mzl(identifier: str) -> tuple[str, str]:
value = normalize(identifier)
match = MZL_RE.fullmatch(value)
if match is None:
raise ValueError(
"un signe doit être identifié par un numéro MZL de un à trois chiffres ASCII"
)
number = int(match.group(1))
if not 1 <= number <= 999:
raise ValueError("le numéro MZL doit être compris entre 1 et 999")
mzl = f"{number:03d}"
return f"mzl-{mzl}", mzl
def choose_nature(nature: str | None) -> str:
if nature:
return nature
if not sys.stdin.isatty():
raise ValueError("--nature est requis pour une fiche de vocabulaire")
choices = tuple(VOCABULARY_FIELDS)
print("Nature grammaticale :")
for index, choice in enumerate(choices, start=1):
print(f" {index}. {choice}")
try:
answer = input("Choix : ").strip()
except (EOFError, KeyboardInterrupt) as exc:
raise ValueError("sélection interrompue") from exc
if not answer.isdigit() or not 1 <= int(answer) <= len(choices):
raise ValueError("nature grammaticale invalide")
return choices[int(answer) - 1]
def title_from_identifier(identifier: str) -> str:
title = identifier.replace("-", " ")
return title[:1].upper() + title[1:]
def build_vocabulary(title: str, timestamp: str, nature: str) -> str:
fields = "\n".join(
f"{field} = []" if field == "logograms" else f'{field} = ""'
for field in VOCABULARY_FIELDS[nature]
)
return f"""+++
title = {toml_string(title)}
date = {timestamp}
description = ""
draft = true
[extra]
{fields}
[taxonomies]
natures = [{toml_string(nature)}]
+++
"""
def build_sign(title: str, timestamp: str, mzl: str) -> str:
return f"""+++
title = {toml_string(title)}
date = {timestamp}
description = ""
draft = true
[extra]
sign = ""
mzl = "{mzl}"
[taxonomies]
lectures = []
+++
"""
def build_page(section: str, title: str, timestamp: str) -> str:
if section == "articles":
return f"""+++
title = {toml_string(title)}
date = {timestamp}
description = ""
draft = true
[extra]
banner = ""
banner_alt = ""
banner_credit = ""
+++
"""
if section == "grammaire":
return f"""+++
title = {toml_string(title)}
date = {timestamp}
description = ""
draft = true
[taxonomies]
themes = []
+++
"""
if section == "textes":
return f"""+++
title = {toml_string(title)}
date = {timestamp}
description = ""
draft = true
[taxonomies]
genres = []
+++
"""
raise ValueError(f"section non prise en charge : {section}")
def main() -> int:
args = parse_args()
section = SECTION_ALIASES[args.section]
try:
if section != "vocabulaire" and args.nature is not None:
raise ValueError("--nature est réservé aux fiches de vocabulaire")
timestamp = current_timestamp()
custom_title = normalize(args.title) if args.title is not None else None
if section == "signes":
filename, mzl = parse_mzl(args.identifiant)
title = custom_title if custom_title is not None else ""
content = build_sign(title, timestamp, mzl)
else:
filename = validate_identifier(
args.identifiant,
allow_leading_hyphen=section == "vocabulaire",
)
if section == "vocabulaire":
nature = choose_nature(args.nature)
title = custom_title if custom_title is not None else filename
content = build_vocabulary(title, timestamp, nature)
else:
title = (
custom_title
if custom_title is not None
else title_from_identifier(filename)
)
content = build_page(section, title, timestamp)
except (ValueError, RuntimeError) as exc:
print(f"Erreur : {exc}", file=sys.stderr)
return 2
destination = CONTENT_DIR / section / f"{filename}.md"
try:
destination.parent.mkdir(parents=True, exist_ok=True)
with destination.open("x", encoding="utf-8") as stream:
stream.write(content)
except FileExistsError:
print(
f"Erreur : le fichier existe déjà : {destination.relative_to(ROOT_DIR)}",
file=sys.stderr,
)
return 1
except OSError as exc:
print(
f"Erreur : impossible de créer {destination.relative_to(ROOT_DIR)} ({exc})",
file=sys.stderr,
)
return 1
print(f"Brouillon créé : {destination.relative_to(ROOT_DIR)}")
print("Compléter la fiche, passer draft à false, puis lancer :")
print(" ./scripts/check-content.py")
return 0
if __name__ == "__main__":
raise SystemExit(main())