Files
tessera-ctl/.planning/quick/261009-p0m-sicherheitsprotokoll-und-ci-sicherheitsp/checks/link-check.py
T
schalli d62e6c2dbe
Tessera CI/CD / Lint & Type Check (push) Successful in 53s
Tessera CI/CD / Tests (push) Failing after 2m14s
Tessera CI/CD / Desktop-Pakete bauen (push) Has been skipped
Tessera CI/CD / Build & Publish Images (push) Has been skipped
Tessera CI/CD / Sicherheitspruefung (nur Bericht) (push) Has been skipped
docs(quick-261009-p0m): Sicherheitsprotokoll und CI-Pruefungen
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-09 21:07:46 +02:00

105 lines
3.5 KiB
Python

"""Linkpruefung fuer das Sicherheitsprotokoll und seine Verweise (quick-261009-p0m).
Aufruf vom Hauptordner des Repositorys: python3 -I <Pfad>/link-check.py
Geprueft werden in docs/sicherheitsprotokoll.md, docs/README.md,
docs/anleitung-betrieb.md, docs/anleitung-entwicklung.md und docs/ci-cd-setup.md
alle relativen Markdown-Links, die
- im Sicherheitsprotokoll stehen, oder
- auf sicherheitsprotokoll.md zeigen, oder
- der Anker #sicherheitsprüfungen sind.
Das Linkziel (Datei) muss existieren, und ein Anker muss zu einer Ueberschrift des
Ziels passen (GitHub-Regel: Kleinbuchstaben, alles ausser Wortzeichen, Bindestrich
und Leerzeichen entfernen, Leerzeichen zu Bindestrichen). Exit 1 bei Fehlern.
"""
import os
import re
import sys
from urllib.parse import unquote
DOCS = [
"docs/sicherheitsprotokoll.md",
"docs/README.md",
"docs/anleitung-betrieb.md",
"docs/anleitung-entwicklung.md",
"docs/ci-cd-setup.md",
]
PROTOCOL = "docs/sicherheitsprotokoll.md"
LINK = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")
def slug(heading):
text = re.sub(r"\[([^\]]*)\]\([^)]*\)", r"\1", heading) # Link -> Text
text = text.replace("`", "").replace("*", "")
text = text.strip().lower()
text = re.sub(r"[^\w\- ]", "", text)
return text.replace(" ", "-")
def anchors(path):
found = set()
in_fence = False
with open(path, encoding="utf-8") as handle:
for line in handle:
if line.lstrip().startswith("```"):
in_fence = not in_fence
continue
if in_fence:
continue
match = re.match(r"^(#{1,6})\s+(.*?)\s*#*\s*$", line)
if match:
found.add(slug(match.group(2)))
return found
def main():
broken = []
checked = 0
cache = {}
for doc in DOCS:
if not os.path.exists(doc):
broken.append(f"{doc}: Datei fehlt")
continue
with open(doc, encoding="utf-8") as handle:
text = handle.read()
in_fence = False
for number, line in enumerate(text.splitlines(), 1):
if line.lstrip().startswith("```"):
in_fence = not in_fence
continue
if in_fence:
continue
for target in LINK.findall(line):
if re.match(r"^[a-z][a-z0-9+.-]*:", target):
continue # http:, mailto: usw.
file_part, _, anchor = target.partition("#")
anchor = unquote(anchor)
relevant = (
doc == PROTOCOL
or file_part.endswith("sicherheitsprotokoll.md")
or anchor == "sicherheitsprüfungen"
)
if not relevant:
continue
checked += 1
dest = doc if file_part == "" else os.path.normpath(
os.path.join(os.path.dirname(doc), file_part)
)
if not os.path.exists(dest):
broken.append(f"{doc}:{number}: Ziel fehlt: {target}")
continue
if anchor:
if dest not in cache:
cache[dest] = anchors(dest)
if anchor not in cache[dest]:
broken.append(f"{doc}:{number}: Anker fehlt in {dest}: #{anchor}")
for line in broken:
print(line)
print(f"{checked} Links geprueft, {len(broken)} kaputt")
return 1 if broken else 0
if __name__ == "__main__":
sys.exit(main())