d62e6c2dbe
Tessera CI/CD / Lint & Type Check (push) Successful in 53s
Tessera CI/CD / Tests (push) Failing after 2m14s
Tessera CI/CD / Desktop-Pakete bauen (push) Has been skipped
Tessera CI/CD / Build & Publish Images (push) Has been skipped
Tessera CI/CD / Sicherheitspruefung (nur Bericht) (push) Has been skipped
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
105 lines
3.5 KiB
Python
105 lines
3.5 KiB
Python
"""Linkpruefung fuer das Sicherheitsprotokoll und seine Verweise (quick-261009-p0m).
|
|
|
|
Aufruf vom Hauptordner des Repositorys: python3 -I <Pfad>/link-check.py
|
|
|
|
Geprueft werden in docs/sicherheitsprotokoll.md, docs/README.md,
|
|
docs/anleitung-betrieb.md, docs/anleitung-entwicklung.md und docs/ci-cd-setup.md
|
|
alle relativen Markdown-Links, die
|
|
- im Sicherheitsprotokoll stehen, oder
|
|
- auf sicherheitsprotokoll.md zeigen, oder
|
|
- der Anker #sicherheitsprüfungen sind.
|
|
Das Linkziel (Datei) muss existieren, und ein Anker muss zu einer Ueberschrift des
|
|
Ziels passen (GitHub-Regel: Kleinbuchstaben, alles ausser Wortzeichen, Bindestrich
|
|
und Leerzeichen entfernen, Leerzeichen zu Bindestrichen). Exit 1 bei Fehlern.
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
from urllib.parse import unquote
|
|
|
|
DOCS = [
|
|
"docs/sicherheitsprotokoll.md",
|
|
"docs/README.md",
|
|
"docs/anleitung-betrieb.md",
|
|
"docs/anleitung-entwicklung.md",
|
|
"docs/ci-cd-setup.md",
|
|
]
|
|
PROTOCOL = "docs/sicherheitsprotokoll.md"
|
|
LINK = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")
|
|
|
|
|
|
def slug(heading):
|
|
text = re.sub(r"\[([^\]]*)\]\([^)]*\)", r"\1", heading) # Link -> Text
|
|
text = text.replace("`", "").replace("*", "")
|
|
text = text.strip().lower()
|
|
text = re.sub(r"[^\w\- ]", "", text)
|
|
return text.replace(" ", "-")
|
|
|
|
|
|
def anchors(path):
|
|
found = set()
|
|
in_fence = False
|
|
with open(path, encoding="utf-8") as handle:
|
|
for line in handle:
|
|
if line.lstrip().startswith("```"):
|
|
in_fence = not in_fence
|
|
continue
|
|
if in_fence:
|
|
continue
|
|
match = re.match(r"^(#{1,6})\s+(.*?)\s*#*\s*$", line)
|
|
if match:
|
|
found.add(slug(match.group(2)))
|
|
return found
|
|
|
|
|
|
def main():
|
|
broken = []
|
|
checked = 0
|
|
cache = {}
|
|
for doc in DOCS:
|
|
if not os.path.exists(doc):
|
|
broken.append(f"{doc}: Datei fehlt")
|
|
continue
|
|
with open(doc, encoding="utf-8") as handle:
|
|
text = handle.read()
|
|
in_fence = False
|
|
for number, line in enumerate(text.splitlines(), 1):
|
|
if line.lstrip().startswith("```"):
|
|
in_fence = not in_fence
|
|
continue
|
|
if in_fence:
|
|
continue
|
|
for target in LINK.findall(line):
|
|
if re.match(r"^[a-z][a-z0-9+.-]*:", target):
|
|
continue # http:, mailto: usw.
|
|
file_part, _, anchor = target.partition("#")
|
|
anchor = unquote(anchor)
|
|
relevant = (
|
|
doc == PROTOCOL
|
|
or file_part.endswith("sicherheitsprotokoll.md")
|
|
or anchor == "sicherheitsprüfungen"
|
|
)
|
|
if not relevant:
|
|
continue
|
|
checked += 1
|
|
dest = doc if file_part == "" else os.path.normpath(
|
|
os.path.join(os.path.dirname(doc), file_part)
|
|
)
|
|
if not os.path.exists(dest):
|
|
broken.append(f"{doc}:{number}: Ziel fehlt: {target}")
|
|
continue
|
|
if anchor:
|
|
if dest not in cache:
|
|
cache[dest] = anchors(dest)
|
|
if anchor not in cache[dest]:
|
|
broken.append(f"{doc}:{number}: Anker fehlt in {dest}: #{anchor}")
|
|
for line in broken:
|
|
print(line)
|
|
print(f"{checked} Links geprueft, {len(broken)} kaputt")
|
|
return 1 if broken else 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|