ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s
205 lines
8.5 KiB
Python
Executable File
205 lines
8.5 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
"""저장소 안 마크다운 문서의 내부 링크·이미지·앵커 참조 무결성을 검사한다.
|
||
|
||
추적되는 모든 `.md` 문서를 훑어 두 가지를 확인한다.
|
||
|
||
1. 저장소 안 다른 파일을 가리키는 상대 경로 링크(`[text](path)`)와 이미지
|
||
참조(``·`<img src="path">`)가 실제로 존재하는 파일을 가리키는지.
|
||
2. 문서 앵커 링크(`[text](#heading)`·`[text](other.md#heading)`)가 대상 `.md`
|
||
문서에 실제로 있는 헤딩을 가리키는지. 앵커 slug 는 GitHub 의 규칙
|
||
(소문자화 → `\\p{Word}`·하이픈·공백만 남기고 제거 → 공백을 하이픈으로,
|
||
같은 slug 가 겹치면 등장 순서대로 `-1`·`-2` 접미)을 표준 라이브러리로
|
||
재현해 생성한다.
|
||
|
||
깨진 참조가 하나라도 있으면 종료 코드 1 로 끝나므로, CI 머지 게이트
|
||
(`.github/workflows/docs.yml`)와 기여자의 로컬 검증에 그대로 쓸 수 있다.
|
||
|
||
검사 대상이 아닌 것:
|
||
- 외부 URL(`http://`·`https://`·`mailto:`·`tel:`·`data:`): 네트워크에 의존해
|
||
flaky 하므로 이 결정론적 게이트에서는 다루지 않는다. 외부 링크 상태는 별도로
|
||
점검한다.
|
||
- `.md` 가 아닌 파일을 가리키는 링크의 `#` 뒤 조각(예: 소스 코드 줄 번호): 헤딩
|
||
앵커가 아니므로 파일 실존만 확인하고 앵커는 검사하지 않는다.
|
||
- 이미지 참조의 `#` 조각: 이미지에는 헤딩 앵커가 의미가 없다.
|
||
- fenced code block(``` 또는 ~~~ 로 감싼 블록) 안의 예시 링크·헤딩 꼴 주석:
|
||
실제 참조·헤딩이 아니라 코드 예시이므로 건너뛴다.
|
||
|
||
표준 라이브러리만 쓰고 네트워크에 접속하지 않는다. 실행: `python3 -I scripts/check-doc-links.py`
|
||
"""
|
||
import os
|
||
import re
|
||
import subprocess
|
||
import sys
|
||
import unicodedata
|
||
from collections import defaultdict
|
||
|
||
# [text](path) 중 이미지(`![...]`)가 아닌 링크. 경로는 공백 전까지(제목 "..." 제외).
|
||
MD_LINK = re.compile(r"(?<!\!)\[[^\]]*\]\(([^)\s]+)")
|
||
#  이미지 링크.
|
||
MD_IMAGE = re.compile(r"!\[[^\]]*\]\(([^)\s]+)")
|
||
# <img ... src="path" ...> HTML 이미지.
|
||
HTML_IMAGE = re.compile(r"<img[^>]*\bsrc=[\"']([^\"']+)[\"']", re.IGNORECASE)
|
||
|
||
# 네트워크·비파일 스킴. 앵커(`#`)는 같은 문서 참조이므로 여기 넣지 않는다.
|
||
EXTERNAL_PREFIXES = ("http://", "https://", "mailto:", "tel:", "data:")
|
||
|
||
ATX_HEADING = re.compile(r"^(#{1,6})\s+(.*?)\s*#*\s*$")
|
||
# 헤딩 텍스트 안의 인라인 마크다운을 걷어내 순수 텍스트만 남긴다.
|
||
MD_INLINE_LINK = re.compile(r"\[([^\]]*)\]\([^)]*\)")
|
||
|
||
|
||
def is_external(target: str) -> bool:
|
||
return target.startswith(EXTERNAL_PREFIXES)
|
||
|
||
|
||
def _is_word_char(ch: str) -> bool:
|
||
"""GitHub 의 앵커 slug 규칙이 남기는 `\\p{Word}` 문자인지 판정한다.
|
||
|
||
Ruby 정규식 `\\p{Word}` = Letter(L*) + Mark(M*) + Decimal_Number(Nd)
|
||
+ Connector_Punctuation(Pc) + Join_Control(U+200C·U+200D). 한글·한자·
|
||
언더스코어·이모지의 variation selector(Mn) 등이 그대로 유지된다.
|
||
"""
|
||
if ch == "_":
|
||
return True
|
||
if ch in ("", ""):
|
||
return True
|
||
category = unicodedata.category(ch)
|
||
if category[0] in ("L", "M"):
|
||
return True
|
||
return category in ("Nd", "Pc")
|
||
|
||
|
||
def slugify(text: str) -> str:
|
||
"""헤딩 텍스트를 GitHub 의 앵커 slug 로 변환한다(중복 접미 처리 제외)."""
|
||
text = text.strip().lower()
|
||
out = []
|
||
for ch in text:
|
||
if ch == " ":
|
||
out.append("-")
|
||
elif ch == "-" or _is_word_char(ch):
|
||
out.append(ch)
|
||
return "".join(out)
|
||
|
||
|
||
def heading_slugs(root: str, md: str) -> set:
|
||
"""파일의 모든 ATX 헤딩에서 GitHub 앵커 slug 집합을 만든다.
|
||
|
||
같은 slug 가 겹치면 GitHub 처럼 등장 순서대로 `-1`·`-2` 접미를 붙인다.
|
||
"""
|
||
slugs = set()
|
||
counts = defaultdict(int)
|
||
in_fence = False
|
||
with open(os.path.join(root, md), encoding="utf-8", errors="ignore") as fh:
|
||
for line in fh:
|
||
stripped = line.lstrip()
|
||
if stripped.startswith("```") or stripped.startswith("~~~"):
|
||
in_fence = not in_fence
|
||
continue
|
||
if in_fence:
|
||
continue
|
||
match = ATX_HEADING.match(line.rstrip("\n"))
|
||
if not match:
|
||
continue
|
||
raw = match.group(2)
|
||
raw = MD_INLINE_LINK.sub(r"\1", raw)
|
||
raw = raw.replace("`", "").replace("*", "").replace("_", "")
|
||
base = slugify(raw)
|
||
seen = counts[base]
|
||
counts[base] += 1
|
||
slugs.add(base if seen == 0 else f"{base}-{seen}")
|
||
return slugs
|
||
|
||
|
||
def main() -> int:
|
||
root = subprocess.check_output(
|
||
["git", "rev-parse", "--show-toplevel"], text=True
|
||
).strip()
|
||
tracked = subprocess.check_output(
|
||
["git", "ls-files"], cwd=root, text=True
|
||
).splitlines()
|
||
md_files = [f for f in tracked if f.endswith(".md")]
|
||
|
||
slug_cache: dict = {}
|
||
|
||
def slugs_of(md_path: str) -> set:
|
||
if md_path not in slug_cache:
|
||
slug_cache[md_path] = heading_slugs(root, md_path)
|
||
return slug_cache[md_path]
|
||
|
||
checked_files = 0
|
||
checked_anchors = 0
|
||
broken_files = []
|
||
broken_anchors = []
|
||
for md in md_files:
|
||
with open(os.path.join(root, md), encoding="utf-8", errors="ignore") as fh:
|
||
lines = fh.readlines()
|
||
base_dir = os.path.dirname(md)
|
||
in_fence = False
|
||
for lineno, line in enumerate(lines, 1):
|
||
stripped = line.lstrip()
|
||
if stripped.startswith("```") or stripped.startswith("~~~"):
|
||
in_fence = not in_fence
|
||
continue
|
||
if in_fence:
|
||
continue
|
||
for pattern in (MD_LINK, MD_IMAGE, HTML_IMAGE):
|
||
for match in pattern.finditer(line):
|
||
target = match.group(1).strip()
|
||
if is_external(target):
|
||
continue
|
||
|
||
if target.startswith("#"):
|
||
file_part, anchor = "", target[1:]
|
||
elif "#" in target:
|
||
file_part, anchor = target.split("#", 1)
|
||
else:
|
||
file_part, anchor = target, ""
|
||
file_part = file_part.split("?", 1)[0]
|
||
anchor = anchor.strip()
|
||
|
||
# 가리키는 .md 문서(앵커 검사 대상). 다른 파일 링크면 그 파일,
|
||
# 파일 조각이 없는 `#앵커`면 자기 문서.
|
||
target_md = None
|
||
if file_part:
|
||
checked_files += 1
|
||
resolved = os.path.normpath(
|
||
os.path.join(root, base_dir, file_part)
|
||
)
|
||
if not os.path.exists(resolved):
|
||
broken_files.append((md, lineno, target))
|
||
continue
|
||
if file_part.endswith(".md"):
|
||
target_md = os.path.normpath(
|
||
os.path.join(base_dir, file_part)
|
||
)
|
||
else:
|
||
target_md = md
|
||
|
||
# 앵커는 본문 링크(MD_LINK)에만 의미가 있다. 이미지 참조의
|
||
# `#` 조각은 건너뛴다.
|
||
if anchor and target_md is not None and pattern is MD_LINK:
|
||
checked_anchors += 1
|
||
if anchor not in slugs_of(target_md):
|
||
broken_anchors.append((md, lineno, target, target_md))
|
||
|
||
print(
|
||
f"추적 마크다운 {len(md_files)}개 · 파일 참조 {checked_files}개 · "
|
||
f"앵커 참조 {checked_anchors}개 검사"
|
||
)
|
||
if broken_files:
|
||
print(f"깨진 파일 참조 {len(broken_files)}개 — 가리키는 파일이 저장소에 없습니다:")
|
||
for md, lineno, target in broken_files:
|
||
print(f" {md}:{lineno} -> {target}")
|
||
if broken_anchors:
|
||
print(f"깨진 앵커 참조 {len(broken_anchors)}개 — 대상 문서에 그 헤딩이 없습니다:")
|
||
for md, lineno, target, target_md in broken_anchors:
|
||
print(f" {md}:{lineno} -> {target} (대상: {target_md})")
|
||
if broken_files or broken_anchors:
|
||
return 1
|
||
print("깨진 참조 0 — 모든 내부 링크·이미지·앵커가 실존 대상을 가리킵니다.")
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|