first commit
This commit is contained in:
@@ -0,0 +1,61 @@
|
||||
"""Définitions communes au générateur et au validateur de contenus."""
|
||||
|
||||
SECTION_ALIASES = {
|
||||
"article": "articles",
|
||||
"articles": "articles",
|
||||
"grammaire": "grammaire",
|
||||
"signe": "signes",
|
||||
"signes": "signes",
|
||||
"texte": "textes",
|
||||
"textes": "textes",
|
||||
"vocabulaire": "vocabulaire",
|
||||
}
|
||||
|
||||
VOCABULARY_FIELDS = {
|
||||
"nom": ("meaning", "logograms", "gender", "bound", "plural"),
|
||||
"adjectif": (
|
||||
"meaning",
|
||||
"logograms",
|
||||
"bound",
|
||||
"feminine",
|
||||
"masculine_plural",
|
||||
"feminine_plural",
|
||||
"predicative",
|
||||
),
|
||||
"verbe": (
|
||||
"meaning",
|
||||
"logograms",
|
||||
"root",
|
||||
"stem",
|
||||
"verb_class",
|
||||
"vowel_class",
|
||||
"preterite",
|
||||
"durative",
|
||||
"perfect",
|
||||
"imperative",
|
||||
"participle",
|
||||
"verbal_adjective",
|
||||
),
|
||||
"pronom": (
|
||||
"meaning",
|
||||
"logograms",
|
||||
"pronoun_type",
|
||||
"person",
|
||||
"gender",
|
||||
"number",
|
||||
),
|
||||
"préposition": ("meaning", "logograms", "governs"),
|
||||
"adverbe": ("meaning", "logograms", "function"),
|
||||
"conjonction": ("meaning", "logograms", "function"),
|
||||
"particule": ("meaning", "logograms", "function"),
|
||||
"numéral": (
|
||||
"meaning",
|
||||
"logograms",
|
||||
"numeral_type",
|
||||
"value",
|
||||
"gender",
|
||||
"feminine",
|
||||
),
|
||||
}
|
||||
|
||||
KNOWN_SECTIONS = frozenset(SECTION_ALIASES.values())
|
||||
Executable
+286
@@ -0,0 +1,286 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Vérifie la structure des contenus, le vocabulaire et les signes."""
|
||||
|
||||
import re
|
||||
import sys
|
||||
import tomllib
|
||||
import unicodedata
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from _schema import KNOWN_SECTIONS, VOCABULARY_FIELDS
|
||||
|
||||
ROOT_DIR = Path(__file__).resolve().parent.parent
|
||||
CONTENT_DIR = ROOT_DIR / "src" / "content"
|
||||
|
||||
MZL_FILENAME_RE = re.compile(r"mzl-([0-9]{3})\.md")
|
||||
MZL_VALUE_RE = re.compile(r"[0-9]{3}")
|
||||
|
||||
|
||||
class Validator:
|
||||
def __init__(self) -> None:
|
||||
self.errors = 0
|
||||
|
||||
def error(self, path: Path | None, message: str) -> None:
|
||||
if path is None:
|
||||
label = ""
|
||||
else:
|
||||
label = f"{path.relative_to(ROOT_DIR).as_posix()} : "
|
||||
print(f"ERREUR : {label}{message}", file=sys.stderr)
|
||||
self.errors += 1
|
||||
|
||||
def finish(self) -> int:
|
||||
print(f"\nRésultat : {self.errors} erreur(s).")
|
||||
return 1 if self.errors else 0
|
||||
|
||||
|
||||
def is_non_empty_string(value: Any) -> bool:
|
||||
return isinstance(value, str) and bool(value.strip())
|
||||
|
||||
|
||||
def read_front_matter(path: Path, validator: Validator) -> dict[str, Any] | None:
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8")
|
||||
except UnicodeDecodeError as exc:
|
||||
validator.error(path, f"fichier non valide en UTF-8 ({exc})")
|
||||
return None
|
||||
except OSError as exc:
|
||||
validator.error(path, f"lecture impossible ({exc})")
|
||||
return None
|
||||
|
||||
if not unicodedata.is_normalized("NFC", text):
|
||||
validator.error(path, "contenu non normalisé en Unicode NFC")
|
||||
|
||||
lines = text.splitlines()
|
||||
if not lines or lines[0] != "+++":
|
||||
validator.error(path, "front matter TOML initial absent")
|
||||
return None
|
||||
|
||||
try:
|
||||
closing_index = lines.index("+++", 1)
|
||||
except ValueError:
|
||||
validator.error(path, "délimiteur final +++ absent")
|
||||
return None
|
||||
|
||||
try:
|
||||
return tomllib.loads("\n".join(lines[1:closing_index]))
|
||||
except tomllib.TOMLDecodeError as exc:
|
||||
validator.error(path, f"front matter TOML invalide ({exc})")
|
||||
return None
|
||||
|
||||
|
||||
def require_table(
|
||||
metadata: dict[str, Any],
|
||||
key: str,
|
||||
path: Path,
|
||||
validator: Validator,
|
||||
) -> dict[str, Any] | None:
|
||||
value = metadata.get(key)
|
||||
if not isinstance(value, dict):
|
||||
validator.error(path, f"{key} doit être une table TOML")
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def require_string(
|
||||
metadata: dict[str, Any],
|
||||
key: str,
|
||||
path: Path,
|
||||
validator: Validator,
|
||||
) -> None:
|
||||
if not is_non_empty_string(metadata.get(key)):
|
||||
validator.error(path, f"{key} doit être une chaîne non vide")
|
||||
|
||||
|
||||
def require_date(
|
||||
metadata: dict[str, Any],
|
||||
path: Path,
|
||||
validator: Validator,
|
||||
) -> None:
|
||||
if not isinstance(metadata.get("date"), date):
|
||||
validator.error(path, "date doit être une date TOML valide")
|
||||
|
||||
|
||||
def check_structure(
|
||||
path: Path,
|
||||
metadata: dict[str, Any],
|
||||
validator: Validator,
|
||||
) -> tuple[str, bool] | None:
|
||||
relative_path = path.relative_to(CONTENT_DIR)
|
||||
section = relative_path.parts[0] if len(relative_path.parts) > 1 else ""
|
||||
|
||||
if not unicodedata.is_normalized("NFC", relative_path.as_posix()):
|
||||
validator.error(path, "chemin non normalisé en Unicode NFC")
|
||||
|
||||
if section and section not in KNOWN_SECTIONS:
|
||||
validator.error(path, f"section inconnue « {section} »")
|
||||
|
||||
if path.name == "_index.md":
|
||||
require_string(metadata, "title", path, validator)
|
||||
return None
|
||||
|
||||
draft = metadata.get("draft")
|
||||
if not isinstance(draft, bool):
|
||||
validator.error(path, "draft doit être un booléen")
|
||||
return None
|
||||
|
||||
if draft:
|
||||
return section, True
|
||||
|
||||
require_string(metadata, "title", path, validator)
|
||||
if section:
|
||||
require_date(metadata, path, validator)
|
||||
return section, False
|
||||
|
||||
|
||||
def vocabulary_nature(
|
||||
metadata: dict[str, Any],
|
||||
path: Path,
|
||||
validator: Validator,
|
||||
) -> str | None:
|
||||
taxonomies = require_table(metadata, "taxonomies", path, validator)
|
||||
if taxonomies is None:
|
||||
return None
|
||||
|
||||
natures = taxonomies.get("natures")
|
||||
if not isinstance(natures, list) or len(natures) != 1:
|
||||
validator.error(path, "taxonomies.natures doit contenir exactement une nature")
|
||||
return None
|
||||
|
||||
nature = natures[0]
|
||||
if not isinstance(nature, str):
|
||||
validator.error(path, "taxonomies.natures doit contenir une chaîne")
|
||||
return None
|
||||
|
||||
if nature not in VOCABULARY_FIELDS:
|
||||
validator.error(path, f"nature grammaticale inconnue « {nature} »")
|
||||
return None
|
||||
|
||||
return nature
|
||||
|
||||
|
||||
def check_vocabulary(
|
||||
path: Path,
|
||||
metadata: dict[str, Any],
|
||||
validator: Validator,
|
||||
) -> None:
|
||||
nature = vocabulary_nature(metadata, path, validator)
|
||||
extra = require_table(metadata, "extra", path, validator)
|
||||
if nature is None or extra is None:
|
||||
return
|
||||
|
||||
expected_fields = VOCABULARY_FIELDS[nature]
|
||||
expected_keys = set(expected_fields)
|
||||
actual_keys = set(extra)
|
||||
|
||||
missing_fields = [field for field in expected_fields if field not in actual_keys]
|
||||
if missing_fields:
|
||||
validator.error(
|
||||
path,
|
||||
"champs extra manquants pour "
|
||||
f"la nature « {nature} » : {', '.join(missing_fields)}",
|
||||
)
|
||||
|
||||
unexpected_fields = sorted(actual_keys - expected_keys)
|
||||
if unexpected_fields:
|
||||
validator.error(
|
||||
path,
|
||||
"champs extra inattendus pour "
|
||||
f"la nature « {nature} » : {', '.join(unexpected_fields)}",
|
||||
)
|
||||
|
||||
for field in expected_fields:
|
||||
if field not in extra:
|
||||
continue
|
||||
|
||||
value = extra[field]
|
||||
if field == "logograms":
|
||||
if not isinstance(value, list) or not all(
|
||||
is_non_empty_string(item) for item in value
|
||||
):
|
||||
validator.error(
|
||||
path,
|
||||
"extra.logograms doit être un tableau de chaînes non vides",
|
||||
)
|
||||
elif not isinstance(value, str):
|
||||
validator.error(path, f"extra.{field} doit être une chaîne")
|
||||
|
||||
required_fields = ("meaning", "stem") if nature == "verbe" else ("meaning",)
|
||||
for field in required_fields:
|
||||
value = extra.get(field)
|
||||
if isinstance(value, str) and not value.strip():
|
||||
validator.error(path, f"extra.{field} doit être une chaîne non vide")
|
||||
|
||||
|
||||
def filename_mzl(path: Path, validator: Validator) -> str | None:
|
||||
match = MZL_FILENAME_RE.fullmatch(path.name)
|
||||
if match is None:
|
||||
validator.error(path, "le nom du fichier doit suivre le format mzl-XXX.md")
|
||||
return None
|
||||
|
||||
value = match.group(1)
|
||||
if int(value) == 0:
|
||||
validator.error(path, "le numéro MZL doit être compris entre 001 et 999")
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def check_sign(
|
||||
path: Path,
|
||||
metadata: dict[str, Any],
|
||||
expected_mzl: str | None,
|
||||
validator: Validator,
|
||||
) -> None:
|
||||
extra = require_table(metadata, "extra", path, validator)
|
||||
if extra is None:
|
||||
return
|
||||
|
||||
require_string(extra, "sign", path, validator)
|
||||
|
||||
declared_mzl = extra.get("mzl")
|
||||
if not isinstance(declared_mzl, str) or MZL_VALUE_RE.fullmatch(declared_mzl) is None:
|
||||
validator.error(path, "extra.mzl doit être une chaîne de trois chiffres ASCII")
|
||||
return
|
||||
if int(declared_mzl) == 0:
|
||||
validator.error(path, "extra.mzl doit être compris entre 001 et 999")
|
||||
return
|
||||
if expected_mzl is not None and declared_mzl != expected_mzl:
|
||||
validator.error(
|
||||
path,
|
||||
f'extra.mzl vaut "{declared_mzl}", mais le fichier indique MZL {expected_mzl}',
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
validator = Validator()
|
||||
|
||||
if not CONTENT_DIR.is_dir():
|
||||
validator.error(None, f"répertoire introuvable : {CONTENT_DIR}")
|
||||
return validator.finish()
|
||||
|
||||
print("Vérification des contenus…", flush=True)
|
||||
for path in sorted(CONTENT_DIR.rglob("*.md")):
|
||||
metadata = read_front_matter(path, validator)
|
||||
if metadata is None:
|
||||
continue
|
||||
|
||||
structure = check_structure(path, metadata, validator)
|
||||
if structure is None:
|
||||
continue
|
||||
|
||||
section, is_draft = structure
|
||||
expected_mzl = filename_mzl(path, validator) if section == "signes" else None
|
||||
if is_draft:
|
||||
continue
|
||||
|
||||
if section == "vocabulaire":
|
||||
check_vocabulary(path, metadata, validator)
|
||||
elif section == "signes":
|
||||
check_sign(path, metadata, expected_mzl, validator)
|
||||
|
||||
return validator.finish()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Executable
+268
@@ -0,0 +1,268 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Crée un brouillon Markdown adapté à une section du site."""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import unicodedata
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from zoneinfo import ZoneInfo, ZoneInfoNotFoundError
|
||||
|
||||
from _schema import SECTION_ALIASES, VOCABULARY_FIELDS
|
||||
|
||||
ROOT_DIR = Path(__file__).resolve().parent.parent
|
||||
CONTENT_DIR = ROOT_DIR / "src" / "content"
|
||||
TIMEZONE_NAME = "Europe/Paris"
|
||||
|
||||
MZL_RE = re.compile(r"(?:mzl-)?([0-9]{1,3})")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Créer un brouillon Markdown horodaté et prêt à compléter."
|
||||
)
|
||||
parser.add_argument(
|
||||
"section",
|
||||
choices=sorted(SECTION_ALIASES),
|
||||
help="section de destination",
|
||||
)
|
||||
parser.add_argument(
|
||||
"identifiant",
|
||||
help="nom du fichier sans .md ; pour un signe, numéro MZL ou mzl-XXX",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--title",
|
||||
help="titre initial ; déduit de l’identifiant par défaut",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--nature",
|
||||
choices=tuple(VOCABULARY_FIELDS),
|
||||
help="nature grammaticale d’une fiche de vocabulaire",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def normalize(value: str) -> str:
|
||||
return unicodedata.normalize("NFC", value.strip())
|
||||
|
||||
|
||||
def toml_string(value: str) -> str:
|
||||
return json.dumps(value, ensure_ascii=False)
|
||||
|
||||
|
||||
def current_timestamp() -> str:
|
||||
try:
|
||||
timezone = ZoneInfo(TIMEZONE_NAME)
|
||||
except ZoneInfoNotFoundError as exc:
|
||||
raise RuntimeError(
|
||||
f"fuseau {TIMEZONE_NAME} indisponible ; installer les données de fuseaux horaires"
|
||||
) from exc
|
||||
return datetime.now(timezone).isoformat(timespec="seconds")
|
||||
|
||||
|
||||
def validate_identifier(identifier: str, *, allow_leading_hyphen: bool = False) -> str:
|
||||
"""Valide un identifiant déjà écrit sous sa forme de nom de fichier."""
|
||||
value = normalize(identifier)
|
||||
if value.startswith("-") and not allow_leading_hyphen:
|
||||
raise ValueError("seul le vocabulaire accepte un identifiant commençant par un tiret")
|
||||
|
||||
body = value[1:] if value.startswith("-") else value
|
||||
if not body or value in {".", "..", "_index"}:
|
||||
raise ValueError("identifiant de fichier invalide")
|
||||
if value != value.lower():
|
||||
raise ValueError("l’identifiant doit être écrit en minuscules")
|
||||
if value.endswith("-") or "--" in value:
|
||||
raise ValueError(
|
||||
"l’identifiant ne doit pas finir par un tiret ni en contenir deux de suite"
|
||||
)
|
||||
if not all(character.isalnum() or character == "-" for character in body):
|
||||
raise ValueError(
|
||||
"l’identifiant ne doit contenir que des lettres, des chiffres et des tirets"
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
def parse_mzl(identifier: str) -> tuple[str, str]:
|
||||
value = normalize(identifier)
|
||||
match = MZL_RE.fullmatch(value)
|
||||
if match is None:
|
||||
raise ValueError(
|
||||
"un signe doit être identifié par un numéro MZL de un à trois chiffres ASCII"
|
||||
)
|
||||
|
||||
number = int(match.group(1))
|
||||
if not 1 <= number <= 999:
|
||||
raise ValueError("le numéro MZL doit être compris entre 1 et 999")
|
||||
|
||||
mzl = f"{number:03d}"
|
||||
return f"mzl-{mzl}", mzl
|
||||
|
||||
|
||||
def choose_nature(nature: str | None) -> str:
|
||||
if nature:
|
||||
return nature
|
||||
if not sys.stdin.isatty():
|
||||
raise ValueError("--nature est requis pour une fiche de vocabulaire")
|
||||
|
||||
choices = tuple(VOCABULARY_FIELDS)
|
||||
print("Nature grammaticale :")
|
||||
for index, choice in enumerate(choices, start=1):
|
||||
print(f" {index}. {choice}")
|
||||
|
||||
try:
|
||||
answer = input("Choix : ").strip()
|
||||
except (EOFError, KeyboardInterrupt) as exc:
|
||||
raise ValueError("sélection interrompue") from exc
|
||||
|
||||
if not answer.isdigit() or not 1 <= int(answer) <= len(choices):
|
||||
raise ValueError("nature grammaticale invalide")
|
||||
return choices[int(answer) - 1]
|
||||
|
||||
|
||||
def title_from_identifier(identifier: str) -> str:
|
||||
title = identifier.replace("-", " ")
|
||||
return title[:1].upper() + title[1:]
|
||||
|
||||
|
||||
def build_vocabulary(title: str, timestamp: str, nature: str) -> str:
|
||||
fields = "\n".join(
|
||||
f"{field} = []" if field == "logograms" else f'{field} = ""'
|
||||
for field in VOCABULARY_FIELDS[nature]
|
||||
)
|
||||
return f"""+++
|
||||
title = {toml_string(title)}
|
||||
date = {timestamp}
|
||||
description = ""
|
||||
draft = true
|
||||
|
||||
[extra]
|
||||
{fields}
|
||||
|
||||
[taxonomies]
|
||||
natures = [{toml_string(nature)}]
|
||||
+++
|
||||
"""
|
||||
|
||||
|
||||
def build_sign(title: str, timestamp: str, mzl: str) -> str:
|
||||
return f"""+++
|
||||
title = {toml_string(title)}
|
||||
date = {timestamp}
|
||||
description = ""
|
||||
draft = true
|
||||
|
||||
[extra]
|
||||
sign = ""
|
||||
mzl = "{mzl}"
|
||||
|
||||
[taxonomies]
|
||||
lectures = []
|
||||
+++
|
||||
"""
|
||||
|
||||
|
||||
def build_page(section: str, title: str, timestamp: str) -> str:
|
||||
if section == "articles":
|
||||
return f"""+++
|
||||
title = {toml_string(title)}
|
||||
date = {timestamp}
|
||||
description = ""
|
||||
draft = true
|
||||
|
||||
[extra]
|
||||
banner = ""
|
||||
banner_alt = ""
|
||||
banner_credit = ""
|
||||
+++
|
||||
"""
|
||||
|
||||
if section == "grammaire":
|
||||
return f"""+++
|
||||
title = {toml_string(title)}
|
||||
date = {timestamp}
|
||||
description = ""
|
||||
draft = true
|
||||
|
||||
[taxonomies]
|
||||
themes = []
|
||||
+++
|
||||
"""
|
||||
|
||||
if section == "textes":
|
||||
return f"""+++
|
||||
title = {toml_string(title)}
|
||||
date = {timestamp}
|
||||
description = ""
|
||||
draft = true
|
||||
|
||||
[taxonomies]
|
||||
genres = []
|
||||
+++
|
||||
"""
|
||||
|
||||
raise ValueError(f"section non prise en charge : {section}")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
section = SECTION_ALIASES[args.section]
|
||||
|
||||
try:
|
||||
if section != "vocabulaire" and args.nature is not None:
|
||||
raise ValueError("--nature est réservé aux fiches de vocabulaire")
|
||||
|
||||
timestamp = current_timestamp()
|
||||
custom_title = normalize(args.title) if args.title is not None else None
|
||||
|
||||
if section == "signes":
|
||||
filename, mzl = parse_mzl(args.identifiant)
|
||||
title = custom_title if custom_title is not None else ""
|
||||
content = build_sign(title, timestamp, mzl)
|
||||
else:
|
||||
filename = validate_identifier(
|
||||
args.identifiant,
|
||||
allow_leading_hyphen=section == "vocabulaire",
|
||||
)
|
||||
if section == "vocabulaire":
|
||||
nature = choose_nature(args.nature)
|
||||
title = custom_title if custom_title is not None else filename
|
||||
content = build_vocabulary(title, timestamp, nature)
|
||||
else:
|
||||
title = (
|
||||
custom_title
|
||||
if custom_title is not None
|
||||
else title_from_identifier(filename)
|
||||
)
|
||||
content = build_page(section, title, timestamp)
|
||||
except (ValueError, RuntimeError) as exc:
|
||||
print(f"Erreur : {exc}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
destination = CONTENT_DIR / section / f"{filename}.md"
|
||||
try:
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
with destination.open("x", encoding="utf-8") as stream:
|
||||
stream.write(content)
|
||||
except FileExistsError:
|
||||
print(
|
||||
f"Erreur : le fichier existe déjà : {destination.relative_to(ROOT_DIR)}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
except OSError as exc:
|
||||
print(
|
||||
f"Erreur : impossible de créer {destination.relative_to(ROOT_DIR)} ({exc})",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
|
||||
print(f"Brouillon créé : {destination.relative_to(ROOT_DIR)}")
|
||||
print("Compléter la fiche, passer draft à false, puis lancer :")
|
||||
print(" ./scripts/check-content.py")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user