Files
artex/scripts/check-doc-links.py
T
dela 0335d572de
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
detections / detections (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
First Commit
2026-10-09 08:38:16 +08:00

205 lines
8.5 KiB
Python
Executable File
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""저장소 안 마크다운 문서의 내부 링크·이미지·앵커 참조 무결성을 검사한다.
추적되는 모든 `.md` 문서를 훑어 두 가지를 확인한다.
1. 저장소 안 다른 파일을 가리키는 상대 경로 링크(`[text](path)`)와 이미지
참조(`![alt](path)`·`<img src="path">`)가 실제로 존재하는 파일을 가리키는지.
2. 문서 앵커 링크(`[text](#heading)`·`[text](other.md#heading)`)가 대상 `.md`
문서에 실제로 있는 헤딩을 가리키는지. 앵커 slug 는 GitHub 의 규칙
(소문자화 → `\\p{Word}`·하이픈·공백만 남기고 제거 → 공백을 하이픈으로,
같은 slug 가 겹치면 등장 순서대로 `-1`·`-2` 접미)을 표준 라이브러리로
재현해 생성한다.
깨진 참조가 하나라도 있으면 종료 코드 1 로 끝나므로, CI 머지 게이트
(`.github/workflows/docs.yml`)와 기여자의 로컬 검증에 그대로 쓸 수 있다.
검사 대상이 아닌 것:
- 외부 URL(`http://`·`https://`·`mailto:`·`tel:`·`data:`): 네트워크에 의존해
flaky 하므로 이 결정론적 게이트에서는 다루지 않는다. 외부 링크 상태는 별도로
점검한다.
- `.md` 가 아닌 파일을 가리키는 링크의 `#` 뒤 조각(예: 소스 코드 줄 번호): 헤딩
앵커가 아니므로 파일 실존만 확인하고 앵커는 검사하지 않는다.
- 이미지 참조의 `#` 조각: 이미지에는 헤딩 앵커가 의미가 없다.
- fenced code block(``` 또는 ~~~ 로 감싼 블록) 안의 예시 링크·헤딩 꼴 주석:
실제 참조·헤딩이 아니라 코드 예시이므로 건너뛴다.
표준 라이브러리만 쓰고 네트워크에 접속하지 않는다. 실행: `python3 -I scripts/check-doc-links.py`
"""
import os
import re
import subprocess
import sys
import unicodedata
from collections import defaultdict
# [text](path) 중 이미지(`![...]`)가 아닌 링크. 경로는 공백 전까지(제목 "..." 제외).
MD_LINK = re.compile(r"(?<!\!)\[[^\]]*\]\(([^)\s]+)")
# ![alt](path) 이미지 링크.
MD_IMAGE = re.compile(r"!\[[^\]]*\]\(([^)\s]+)")
# <img ... src="path" ...> HTML 이미지.
HTML_IMAGE = re.compile(r"<img[^>]*\bsrc=[\"']([^\"']+)[\"']", re.IGNORECASE)
# 네트워크·비파일 스킴. 앵커(`#`)는 같은 문서 참조이므로 여기 넣지 않는다.
EXTERNAL_PREFIXES = ("http://", "https://", "mailto:", "tel:", "data:")
ATX_HEADING = re.compile(r"^(#{1,6})\s+(.*?)\s*#*\s*$")
# 헤딩 텍스트 안의 인라인 마크다운을 걷어내 순수 텍스트만 남긴다.
MD_INLINE_LINK = re.compile(r"\[([^\]]*)\]\([^)]*\)")
def is_external(target: str) -> bool:
return target.startswith(EXTERNAL_PREFIXES)
def _is_word_char(ch: str) -> bool:
"""GitHub 의 앵커 slug 규칙이 남기는 `\\p{Word}` 문자인지 판정한다.
Ruby 정규식 `\\p{Word}` = Letter(L*) + Mark(M*) + Decimal_Number(Nd)
+ Connector_Punctuation(Pc) + Join_Control(U+200C·U+200D). 한글·한자·
언더스코어·이모지의 variation selector(Mn) 등이 그대로 유지된다.
"""
if ch == "_":
return True
if ch in ("‌", "‍"):
return True
category = unicodedata.category(ch)
if category[0] in ("L", "M"):
return True
return category in ("Nd", "Pc")
def slugify(text: str) -> str:
"""헤딩 텍스트를 GitHub 의 앵커 slug 로 변환한다(중복 접미 처리 제외)."""
text = text.strip().lower()
out = []
for ch in text:
if ch == " ":
out.append("-")
elif ch == "-" or _is_word_char(ch):
out.append(ch)
return "".join(out)
def heading_slugs(root: str, md: str) -> set:
"""파일의 모든 ATX 헤딩에서 GitHub 앵커 slug 집합을 만든다.
같은 slug 가 겹치면 GitHub 처럼 등장 순서대로 `-1`·`-2` 접미를 붙인다.
"""
slugs = set()
counts = defaultdict(int)
in_fence = False
with open(os.path.join(root, md), encoding="utf-8", errors="ignore") as fh:
for line in fh:
stripped = line.lstrip()
if stripped.startswith("```") or stripped.startswith("~~~"):
in_fence = not in_fence
continue
if in_fence:
continue
match = ATX_HEADING.match(line.rstrip("\n"))
if not match:
continue
raw = match.group(2)
raw = MD_INLINE_LINK.sub(r"\1", raw)
raw = raw.replace("`", "").replace("*", "").replace("_", "")
base = slugify(raw)
seen = counts[base]
counts[base] += 1
slugs.add(base if seen == 0 else f"{base}-{seen}")
return slugs
def main() -> int:
root = subprocess.check_output(
["git", "rev-parse", "--show-toplevel"], text=True
).strip()
tracked = subprocess.check_output(
["git", "ls-files"], cwd=root, text=True
).splitlines()
md_files = [f for f in tracked if f.endswith(".md")]
slug_cache: dict = {}
def slugs_of(md_path: str) -> set:
if md_path not in slug_cache:
slug_cache[md_path] = heading_slugs(root, md_path)
return slug_cache[md_path]
checked_files = 0
checked_anchors = 0
broken_files = []
broken_anchors = []
for md in md_files:
with open(os.path.join(root, md), encoding="utf-8", errors="ignore") as fh:
lines = fh.readlines()
base_dir = os.path.dirname(md)
in_fence = False
for lineno, line in enumerate(lines, 1):
stripped = line.lstrip()
if stripped.startswith("```") or stripped.startswith("~~~"):
in_fence = not in_fence
continue
if in_fence:
continue
for pattern in (MD_LINK, MD_IMAGE, HTML_IMAGE):
for match in pattern.finditer(line):
target = match.group(1).strip()
if is_external(target):
continue
if target.startswith("#"):
file_part, anchor = "", target[1:]
elif "#" in target:
file_part, anchor = target.split("#", 1)
else:
file_part, anchor = target, ""
file_part = file_part.split("?", 1)[0]
anchor = anchor.strip()
# 가리키는 .md 문서(앵커 검사 대상). 다른 파일 링크면 그 파일,
# 파일 조각이 없는 `#앵커`면 자기 문서.
target_md = None
if file_part:
checked_files += 1
resolved = os.path.normpath(
os.path.join(root, base_dir, file_part)
)
if not os.path.exists(resolved):
broken_files.append((md, lineno, target))
continue
if file_part.endswith(".md"):
target_md = os.path.normpath(
os.path.join(base_dir, file_part)
)
else:
target_md = md
# 앵커는 본문 링크(MD_LINK)에만 의미가 있다. 이미지 참조의
# `#` 조각은 건너뛴다.
if anchor and target_md is not None and pattern is MD_LINK:
checked_anchors += 1
if anchor not in slugs_of(target_md):
broken_anchors.append((md, lineno, target, target_md))
print(
f"추적 마크다운 {len(md_files)}개 · 파일 참조 {checked_files}개 · "
f"앵커 참조 {checked_anchors}개 검사"
)
if broken_files:
print(f"깨진 파일 참조 {len(broken_files)}개 — 가리키는 파일이 저장소에 없습니다:")
for md, lineno, target in broken_files:
print(f" {md}:{lineno} -> {target}")
if broken_anchors:
print(f"깨진 앵커 참조 {len(broken_anchors)}개 — 대상 문서에 그 헤딩이 없습니다:")
for md, lineno, target, target_md in broken_anchors:
print(f" {md}:{lineno} -> {target} (대상: {target_md})")
if broken_files or broken_anchors:
return 1
print("깨진 참조 0 — 모든 내부 링크·이미지·앵커가 실존 대상을 가리킵니다.")
return 0
if __name__ == "__main__":
sys.exit(main())