First Commit
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s

This commit is contained in:
dela
2026-10-09 08:38:16 +08:00
commit 0335d572de
756 changed files with 201663 additions and 0 deletions
+204
View File
@@ -0,0 +1,204 @@
#!/usr/bin/env python3
"""저장소 안 마크다운 문서의 내부 링크·이미지·앵커 참조 무결성을 검사한다.
추적되는 모든 `.md` 문서를 훑어 두 가지를 확인한다.
1. 저장소 안 다른 파일을 가리키는 상대 경로 링크(`[text](path)`)와 이미지
참조(`![alt](path)`·`<img src="path">`)가 실제로 존재하는 파일을 가리키는지.
2. 문서 앵커 링크(`[text](#heading)`·`[text](other.md#heading)`)가 대상 `.md`
문서에 실제로 있는 헤딩을 가리키는지. 앵커 slug 는 GitHub 의 규칙
(소문자화 → `\\p{Word}`·하이픈·공백만 남기고 제거 → 공백을 하이픈으로,
같은 slug 가 겹치면 등장 순서대로 `-1`·`-2` 접미)을 표준 라이브러리로
재현해 생성한다.
깨진 참조가 하나라도 있으면 종료 코드 1 로 끝나므로, CI 머지 게이트
(`.github/workflows/docs.yml`)와 기여자의 로컬 검증에 그대로 쓸 수 있다.
검사 대상이 아닌 것:
- 외부 URL(`http://`·`https://`·`mailto:`·`tel:`·`data:`): 네트워크에 의존해
flaky 하므로 이 결정론적 게이트에서는 다루지 않는다. 외부 링크 상태는 별도로
점검한다.
- `.md` 가 아닌 파일을 가리키는 링크의 `#` 뒤 조각(예: 소스 코드 줄 번호): 헤딩
앵커가 아니므로 파일 실존만 확인하고 앵커는 검사하지 않는다.
- 이미지 참조의 `#` 조각: 이미지에는 헤딩 앵커가 의미가 없다.
- fenced code block(``` 또는 ~~~ 로 감싼 블록) 안의 예시 링크·헤딩 꼴 주석:
실제 참조·헤딩이 아니라 코드 예시이므로 건너뛴다.
표준 라이브러리만 쓰고 네트워크에 접속하지 않는다. 실행: `python3 -I scripts/check-doc-links.py`
"""
import os
import re
import subprocess
import sys
import unicodedata
from collections import defaultdict
# [text](path) 중 이미지(`![...]`)가 아닌 링크. 경로는 공백 전까지(제목 "..." 제외).
MD_LINK = re.compile(r"(?<!\!)\[[^\]]*\]\(([^)\s]+)")
# ![alt](path) 이미지 링크.
MD_IMAGE = re.compile(r"!\[[^\]]*\]\(([^)\s]+)")
# <img ... src="path" ...> HTML 이미지.
HTML_IMAGE = re.compile(r"<img[^>]*\bsrc=[\"']([^\"']+)[\"']", re.IGNORECASE)
# 네트워크·비파일 스킴. 앵커(`#`)는 같은 문서 참조이므로 여기 넣지 않는다.
EXTERNAL_PREFIXES = ("http://", "https://", "mailto:", "tel:", "data:")
ATX_HEADING = re.compile(r"^(#{1,6})\s+(.*?)\s*#*\s*$")
# 헤딩 텍스트 안의 인라인 마크다운을 걷어내 순수 텍스트만 남긴다.
MD_INLINE_LINK = re.compile(r"\[([^\]]*)\]\([^)]*\)")
def is_external(target: str) -> bool:
return target.startswith(EXTERNAL_PREFIXES)
def _is_word_char(ch: str) -> bool:
"""GitHub 의 앵커 slug 규칙이 남기는 `\\p{Word}` 문자인지 판정한다.
Ruby 정규식 `\\p{Word}` = Letter(L*) + Mark(M*) + Decimal_Number(Nd)
+ Connector_Punctuation(Pc) + Join_Control(U+200C·U+200D). 한글·한자·
언더스코어·이모지의 variation selector(Mn) 등이 그대로 유지된다.
"""
if ch == "_":
return True
if ch in ("‌", "‍"):
return True
category = unicodedata.category(ch)
if category[0] in ("L", "M"):
return True
return category in ("Nd", "Pc")
def slugify(text: str) -> str:
"""헤딩 텍스트를 GitHub 의 앵커 slug 로 변환한다(중복 접미 처리 제외)."""
text = text.strip().lower()
out = []
for ch in text:
if ch == " ":
out.append("-")
elif ch == "-" or _is_word_char(ch):
out.append(ch)
return "".join(out)
def heading_slugs(root: str, md: str) -> set:
"""파일의 모든 ATX 헤딩에서 GitHub 앵커 slug 집합을 만든다.
같은 slug 가 겹치면 GitHub 처럼 등장 순서대로 `-1`·`-2` 접미를 붙인다.
"""
slugs = set()
counts = defaultdict(int)
in_fence = False
with open(os.path.join(root, md), encoding="utf-8", errors="ignore") as fh:
for line in fh:
stripped = line.lstrip()
if stripped.startswith("```") or stripped.startswith("~~~"):
in_fence = not in_fence
continue
if in_fence:
continue
match = ATX_HEADING.match(line.rstrip("\n"))
if not match:
continue
raw = match.group(2)
raw = MD_INLINE_LINK.sub(r"\1", raw)
raw = raw.replace("`", "").replace("*", "").replace("_", "")
base = slugify(raw)
seen = counts[base]
counts[base] += 1
slugs.add(base if seen == 0 else f"{base}-{seen}")
return slugs
def main() -> int:
root = subprocess.check_output(
["git", "rev-parse", "--show-toplevel"], text=True
).strip()
tracked = subprocess.check_output(
["git", "ls-files"], cwd=root, text=True
).splitlines()
md_files = [f for f in tracked if f.endswith(".md")]
slug_cache: dict = {}
def slugs_of(md_path: str) -> set:
if md_path not in slug_cache:
slug_cache[md_path] = heading_slugs(root, md_path)
return slug_cache[md_path]
checked_files = 0
checked_anchors = 0
broken_files = []
broken_anchors = []
for md in md_files:
with open(os.path.join(root, md), encoding="utf-8", errors="ignore") as fh:
lines = fh.readlines()
base_dir = os.path.dirname(md)
in_fence = False
for lineno, line in enumerate(lines, 1):
stripped = line.lstrip()
if stripped.startswith("```") or stripped.startswith("~~~"):
in_fence = not in_fence
continue
if in_fence:
continue
for pattern in (MD_LINK, MD_IMAGE, HTML_IMAGE):
for match in pattern.finditer(line):
target = match.group(1).strip()
if is_external(target):
continue
if target.startswith("#"):
file_part, anchor = "", target[1:]
elif "#" in target:
file_part, anchor = target.split("#", 1)
else:
file_part, anchor = target, ""
file_part = file_part.split("?", 1)[0]
anchor = anchor.strip()
# 가리키는 .md 문서(앵커 검사 대상). 다른 파일 링크면 그 파일,
# 파일 조각이 없는 `#앵커`면 자기 문서.
target_md = None
if file_part:
checked_files += 1
resolved = os.path.normpath(
os.path.join(root, base_dir, file_part)
)
if not os.path.exists(resolved):
broken_files.append((md, lineno, target))
continue
if file_part.endswith(".md"):
target_md = os.path.normpath(
os.path.join(base_dir, file_part)
)
else:
target_md = md
# 앵커는 본문 링크(MD_LINK)에만 의미가 있다. 이미지 참조의
# `#` 조각은 건너뛴다.
if anchor and target_md is not None and pattern is MD_LINK:
checked_anchors += 1
if anchor not in slugs_of(target_md):
broken_anchors.append((md, lineno, target, target_md))
print(
f"추적 마크다운 {len(md_files)}개 · 파일 참조 {checked_files}개 · "
f"앵커 참조 {checked_anchors}개 검사"
)
if broken_files:
print(f"깨진 파일 참조 {len(broken_files)}개 — 가리키는 파일이 저장소에 없습니다:")
for md, lineno, target in broken_files:
print(f" {md}:{lineno} -> {target}")
if broken_anchors:
print(f"깨진 앵커 참조 {len(broken_anchors)}개 — 대상 문서에 그 헤딩이 없습니다:")
for md, lineno, target, target_md in broken_anchors:
print(f" {md}:{lineno} -> {target} (대상: {target_md})")
if broken_files or broken_anchors:
return 1
print("깨진 참조 0 — 모든 내부 링크·이미지·앵커가 실존 대상을 가리킵니다.")
return 0
if __name__ == "__main__":
sys.exit(main())