First Commit
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s
This commit is contained in:
@@ -0,0 +1,210 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
build_perm_tree.py <JSDIR> <OUTDIR> [--config recon/config.json]
|
||||
|
||||
Rebuild a permissive permission/menu stub from frontend auth modules.
|
||||
|
||||
Typical consumer chain (TDP / enterprise admin SPAs):
|
||||
POST /api/web/user/role_permissions -> { permissions: [...], role_type }
|
||||
POST /api/web/permissions/all -> tree[{ code, position, children }]
|
||||
getResultTree(tree, permissions) -> menu include lists
|
||||
userRouteAuth[code].url -> frontend route path
|
||||
|
||||
This script:
|
||||
1. Locates userRouteAuth={MONITOR:{url:...},...} in js/
|
||||
2. Resolves webpack alias refs (He=o.DASHBOARD) via route_map.json
|
||||
3. Infers hierarchy from code prefixes (MONITOR_ -> MONITOR)
|
||||
4. Writes permissions_tree.json, permissions_all_stub.json,
|
||||
role_permissions_stub.json, userRouteAuth.json
|
||||
5. Optionally patches config.json stubs (permissions/all + role_permissions)
|
||||
|
||||
Adjust ROOTS / PREFIX_PARENT / EXTRA_PARENT per target if heuristics miss nodes.
|
||||
"""
|
||||
import re, json, os, sys, glob, argparse
|
||||
|
||||
DEFAULT_ROOTS = [
|
||||
'MONITOR', 'THREAT', 'ASSETS_RISK', 'INVESTIGATION', 'MANAGEMENT',
|
||||
'AGENT_EVIDENCE', 'MDR', 'PLATFORM',
|
||||
]
|
||||
DEFAULT_PREFIX_PARENT = {
|
||||
'MONITOR_': 'MONITOR', 'THREAT_': 'THREAT', 'ASSETS_RISK_': 'ASSETS_RISK',
|
||||
'INVESTIGATION_': 'INVESTIGATION', 'MANAGEMENT_': 'MANAGEMENT', 'MDR_': 'MDR',
|
||||
'PLATFORM_': 'PLATFORM', 'CUSTOM_': 'PLATFORM', 'FALSE_': 'PLATFORM',
|
||||
}
|
||||
DEFAULT_EXTRA_PARENT = {
|
||||
'LINKAGE_DISPOSAL': 'MANAGEMENT',
|
||||
'ASSETS_RISK_API': 'ASSETS_RISK',
|
||||
'ASSETS_RISK_WEAK_PWD': 'ASSETS_RISK',
|
||||
}
|
||||
|
||||
|
||||
def find_auth_file(jsdir):
|
||||
best_fp, best_len = None, 0
|
||||
for fp in glob.glob(os.path.join(jsdir, '*.js')):
|
||||
try:
|
||||
txt = open(fp, encoding='utf-8', errors='ignore').read()
|
||||
except Exception:
|
||||
continue
|
||||
if 'userRouteAuth' not in txt:
|
||||
continue
|
||||
m = re.search(r'userRouteAuth=\{MONITOR:', txt)
|
||||
if m and len(txt) > best_len:
|
||||
best_fp, best_len = fp, len(m.group(0))
|
||||
elif 'userRouteAuth' in txt and best_fp is None:
|
||||
best_fp = fp
|
||||
return best_fp
|
||||
|
||||
|
||||
def parse_aliases(chunk):
|
||||
aliases = {}
|
||||
for m in re.finditer(r'([a-zA-Z_$][\w$]*)=o\.([A-Z_0-9]+)\b', chunk[:8000]):
|
||||
aliases[m.group(1)] = m.group(2)
|
||||
return aliases
|
||||
|
||||
|
||||
def resolve_ref(token, aliases, route_map):
|
||||
token = token.strip()
|
||||
if token.startswith('"'):
|
||||
return json.loads(token)
|
||||
if token in aliases:
|
||||
key = aliases[token]
|
||||
return route_map.get(key, {}).get('link', key)
|
||||
return token
|
||||
|
||||
|
||||
def parse_user_route_auth(txt, route_map):
|
||||
m = re.search(r'(?:t\.)?userRouteAuth=(\{MONITOR:.*?\})\},', txt, re.S)
|
||||
if not m:
|
||||
m = re.search(r'userRouteAuth=(\{[A-Z_0-9]+:\{url:', txt)
|
||||
if not m:
|
||||
return None, {}
|
||||
# greedy fallback — trim at next webpack module
|
||||
body = m.group(1)
|
||||
end = body.rfind('}')
|
||||
obj_src = body[: end + 1] if end > 0 else body
|
||||
else:
|
||||
obj_src = m.group(1)
|
||||
|
||||
chunk_start = txt.find('userRouteAuth')
|
||||
chunk = txt[chunk_start:chunk_start + 35000]
|
||||
aliases = parse_aliases(chunk)
|
||||
|
||||
entries = {}
|
||||
for em in re.finditer(
|
||||
r'([A-Z_0-9]+):\{url:([^,}]+)(?:,control:(\[.*?\]|[^,}]+))?\}', obj_src
|
||||
):
|
||||
key = em.group(1)
|
||||
url = resolve_ref(em.group(2).strip(), aliases, route_map)
|
||||
controls = []
|
||||
if em.group(3):
|
||||
raw = em.group(3).strip()
|
||||
vars_ = re.findall(r'([A-Za-z_$][\w$]*)', raw) if raw.startswith('[') else [raw]
|
||||
controls = [aliases.get(v, v) for v in vars_]
|
||||
entries[key] = {'code': key, 'url': url, 'control': controls}
|
||||
return obj_src, entries
|
||||
|
||||
|
||||
def parent_of(code, roots, prefix_parent, extra_parent):
|
||||
if code in extra_parent:
|
||||
return extra_parent[code]
|
||||
if code in roots:
|
||||
return None
|
||||
for pref, par in prefix_parent.items():
|
||||
if code.startswith(pref):
|
||||
return par
|
||||
return None
|
||||
|
||||
|
||||
def build_tree(entries, roots, prefix_parent, extra_parent):
|
||||
children_of = {k: [] for k in entries}
|
||||
for code in entries:
|
||||
p = parent_of(code, roots, prefix_parent, extra_parent)
|
||||
if p:
|
||||
children_of.setdefault(p, []).append(code)
|
||||
|
||||
def make_node(code):
|
||||
node = {
|
||||
'code': code,
|
||||
'position': 'top' if code in roots else 'left',
|
||||
'name': code,
|
||||
}
|
||||
kids = sorted(children_of.get(code, []))
|
||||
if kids:
|
||||
node['children'] = [make_node(c) for c in kids]
|
||||
return node
|
||||
|
||||
tree = [make_node(r) for r in roots if r in entries or children_of.get(r)]
|
||||
orphans = [c for c in entries if parent_of(c, roots, prefix_parent, extra_parent) is None and c not in roots]
|
||||
for code in sorted(orphans):
|
||||
tree.append(make_node(code))
|
||||
return tree
|
||||
|
||||
|
||||
def flat_codes(nodes):
|
||||
out = []
|
||||
for n in nodes:
|
||||
out.append(n['code'])
|
||||
out.extend(flat_codes(n.get('children', [])))
|
||||
return out
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description='Build permission tree stubs from JS auth modules')
|
||||
ap.add_argument('jsdir')
|
||||
ap.add_argument('outdir')
|
||||
ap.add_argument('--config', help='patch stubs into config.json')
|
||||
ap.add_argument('--role', default='SUPER_ADMIN', help='role_type in role_permissions stub')
|
||||
args = ap.parse_args()
|
||||
os.makedirs(args.outdir, exist_ok=True)
|
||||
|
||||
route_map_path = os.path.join(args.outdir, 'route_map.json')
|
||||
if not os.path.exists(route_map_path):
|
||||
print('[*] route_map.json missing — run extract_route_map.py first')
|
||||
route_map = {}
|
||||
else:
|
||||
route_map = json.load(open(route_map_path, encoding='utf-8'))
|
||||
|
||||
auth_fp = find_auth_file(args.jsdir)
|
||||
if not auth_fp:
|
||||
print('[!] userRouteAuth module not found in js/')
|
||||
sys.exit(1)
|
||||
print(f'[*] auth module: {os.path.basename(auth_fp)}')
|
||||
txt = open(auth_fp, encoding='utf-8', errors='ignore').read()
|
||||
_, entries = parse_user_route_auth(txt, route_map)
|
||||
if not entries:
|
||||
print('[!] failed to parse userRouteAuth object — adjust regex in script')
|
||||
sys.exit(1)
|
||||
|
||||
tree = build_tree(entries, DEFAULT_ROOTS, DEFAULT_PREFIX_PARENT, DEFAULT_EXTRA_PARENT)
|
||||
all_codes = sorted(set(flat_codes(tree) + [c for e in entries.values() for c in e.get('control', [])]))
|
||||
|
||||
perm_all = {'response_code': 0, 'verbose_msg': 'ok', 'data': tree}
|
||||
role_perm = {
|
||||
'response_code': 0,
|
||||
'verbose_msg': 'ok',
|
||||
'data': {'role_type': args.role, 'permissions': all_codes},
|
||||
}
|
||||
|
||||
json.dump(entries, open(os.path.join(args.outdir, 'userRouteAuth.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
|
||||
json.dump(tree, open(os.path.join(args.outdir, 'permissions_tree.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
|
||||
open(os.path.join(args.outdir, 'perm_codes_all.txt'), 'w', encoding='utf-8').write('\n'.join(all_codes))
|
||||
json.dump(perm_all, open(os.path.join(args.outdir, 'permissions_all_stub.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
|
||||
json.dump(role_perm, open(os.path.join(args.outdir, 'role_permissions_stub.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
|
||||
|
||||
print(f'[+] entries={len(entries)} codes={len(all_codes)} tree_roots={len(tree)}')
|
||||
|
||||
cfg_path = args.config or os.path.join(args.outdir, 'config.json')
|
||||
if os.path.exists(cfg_path):
|
||||
cfg = json.load(open(cfg_path, encoding='utf-8'))
|
||||
stubs = [s for s in cfg.get('stubs', []) if not re.search(r'permissions/all|role_permissions', s.get('match', ''))]
|
||||
stubs = [
|
||||
{'match': 'permissions/all', 'body': perm_all},
|
||||
{'match': 'role_permissions', 'body': role_perm},
|
||||
] + stubs
|
||||
cfg['stubs'] = stubs
|
||||
json.dump(cfg, open(cfg_path, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
|
||||
print(f'[+] patched stubs -> {cfg_path}')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
extract_route_map.py <JSDIR> <OUTDIR>
|
||||
|
||||
Scan downloaded JS chunks for routeMap / routeLink style objects:
|
||||
KEY:{name:"...",link:"/path"}
|
||||
|
||||
Writes route_map.json — used by build_perm_tree.py to resolve alias refs.
|
||||
"""
|
||||
import re, json, os, sys, glob
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 3:
|
||||
print("usage: extract_route_map.py <JSDIR> <OUTDIR>")
|
||||
sys.exit(1)
|
||||
jsdir, outdir = sys.argv[1], sys.argv[2]
|
||||
os.makedirs(outdir, exist_ok=True)
|
||||
|
||||
best = {}
|
||||
best_file = None
|
||||
pat = re.compile(r'([A-Z_][A-Z0-9_]*):\{name:"([^"]*)",link:"([^"]+)"')
|
||||
|
||||
for fp in glob.glob(os.path.join(jsdir, '*.js')):
|
||||
try:
|
||||
txt = open(fp, encoding='utf-8', errors='ignore').read()
|
||||
except Exception:
|
||||
continue
|
||||
hits = pat.findall(txt)
|
||||
if len(hits) > len(best):
|
||||
best = {k: {'name': n, 'link': l} for k, n, l in hits}
|
||||
best_file = fp
|
||||
|
||||
if not best:
|
||||
print('[!] no routeMap pattern found — widen regex or grep manually')
|
||||
sys.exit(1)
|
||||
|
||||
out = os.path.join(outdir, 'route_map.json')
|
||||
json.dump(best, open(out, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
|
||||
print(f'[+] {len(best)} routes from {os.path.basename(best_file)} -> {out}')
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,207 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
harvest_static.py <BASE_URL> <OUTDIR>
|
||||
|
||||
参考模板 — 非通用成品。执行前须按目标站点调整,常见改动:
|
||||
- extract_endpoints 正则(endpoint 方言)
|
||||
- webpack/Vite manifest 解析逻辑
|
||||
- 微前端 publicPath、重试策略
|
||||
|
||||
Static SPA bundle harvester. Framework-agnostic; tuned for webpack + Vite.
|
||||
1. Fetch entry HTML, collect script/module references.
|
||||
2. Parse the runtime chunk manifest(s) and download EVERY chunk (not just the
|
||||
ones in HTML), looping until no new chunk ids appear. Handles multiple
|
||||
micro-frontend runtimes, each with its own publicPath.
|
||||
3. Extract API endpoints and route paths from all downloaded JS.
|
||||
|
||||
Stdlib only. TLS verification is disabled (recon against self-signed/internal hosts).
|
||||
"""
|
||||
import sys, os, re, ssl, json, urllib.request, urllib.parse
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
|
||||
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
|
||||
UA = "Mozilla/5.0 (spa-api-recon)"
|
||||
|
||||
def fetch(url, binary=False):
|
||||
try:
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
with urllib.request.urlopen(req, context=CTX, timeout=25) as r:
|
||||
data = r.read()
|
||||
return (data if binary else data.decode("utf-8", "ignore")), r.status
|
||||
except Exception as e:
|
||||
code = getattr(e, "code", 0)
|
||||
return "", code
|
||||
|
||||
def origin_of(u):
|
||||
p = urllib.parse.urlparse(u)
|
||||
return f"{p.scheme}://{p.netloc}"
|
||||
|
||||
# ---- chunk manifest parsing -------------------------------------------------
|
||||
def matched_block(s, end_brace_idx):
|
||||
"""Walk back from a '}' index to its matching '{' and return the object body."""
|
||||
depth = 0; j = end_brace_idx
|
||||
while j >= 0:
|
||||
if s[j] == '}': depth += 1
|
||||
elif s[j] == '{':
|
||||
depth -= 1
|
||||
if depth == 0: return s[j:end_brace_idx+1]
|
||||
j -= 1
|
||||
return ""
|
||||
|
||||
def parse_chunk_maps(js_text):
|
||||
"""Return list of (publicPath_hint, {id: hash}) for every `}[x]+".js"` map
|
||||
(webpack __webpack_require__.u) found in the text."""
|
||||
maps = []
|
||||
# publicPath hints in this file: .p="..." or publicPath="..."
|
||||
pubs = re.findall(r'(?:\.p|publicPath)\s*=\s*"([^"]*)"', js_text)
|
||||
for m in re.finditer(r'\}\[[A-Za-z_$]\]\s*\+\s*"\.js"', js_text):
|
||||
body = matched_block(js_text, m.start())
|
||||
pairs = re.findall(r'(\d+):"([0-9a-fA-F]{6,16})"', body)
|
||||
if pairs:
|
||||
maps.append((pubs, dict(pairs)))
|
||||
# Vite style: __vite__mapDeps map of "assets/xx.js"
|
||||
for fn in re.findall(r'"(assets/[^"]+\.js)"', js_text):
|
||||
maps.append((["/"], {"__vite__": fn}))
|
||||
return maps
|
||||
|
||||
def chunk_urls(base, js_text):
|
||||
"""Yield absolute chunk URLs reconstructable from this file's manifest(s)."""
|
||||
origin = origin_of(base)
|
||||
out = set()
|
||||
for pubs, idmap in parse_chunk_maps(js_text):
|
||||
paths = pubs or ["/assets/", "/"]
|
||||
for cid, h in idmap.items():
|
||||
if cid == "__vite__":
|
||||
fn = h # already "assets/xx.js"
|
||||
for pp in paths:
|
||||
out.add(urllib.parse.urljoin(origin + "/", fn))
|
||||
continue
|
||||
for pp in paths:
|
||||
if pp.startswith("http"): bareorigin = ""; prefix = pp
|
||||
else: bareorigin = origin; prefix = pp if pp.startswith("/") else "/"+pp
|
||||
if not prefix.endswith("/"): prefix += "/"
|
||||
out.add(f"{bareorigin}{prefix}{cid}.{h}.js")
|
||||
return out
|
||||
|
||||
# ---- endpoint / route extraction -------------------------------------------
|
||||
API_HINT = re.compile(r'/(?:api|rest|service|services|gateway|graphql|v\d|web|admin|backend|open)\b', re.I)
|
||||
ASSET_EXT = re.compile(r'\.(js|css|png|jpe?g|svg|gif|woff2?|ttf|ico|map|json|mp4|webp)(\?|$)', re.I)
|
||||
|
||||
def extract_endpoints(js_text):
|
||||
paths = set()
|
||||
# quoted ('/...'), double-quoted, and backtick template paths
|
||||
for m in re.findall(r'''["'`](/[A-Za-z0-9_\-./{}$:]+)["'`]''', js_text):
|
||||
paths.add(m)
|
||||
# concatenation heads: "/api/x/" + var
|
||||
for m in re.findall(r'''["'](/[A-Za-z0-9_\-./]+/)["']\s*\+''', js_text):
|
||||
paths.add(m)
|
||||
api, other = set(), set()
|
||||
for p in paths:
|
||||
if ASSET_EXT.search(p): continue
|
||||
if p.count('/') < 2 and not API_HINT.search(p): continue
|
||||
(api if API_HINT.search(p) else other).add(p)
|
||||
return api, other
|
||||
|
||||
def extract_routes(js_text):
|
||||
r = set()
|
||||
for key in ('path', 'to', 'redirect', 'href'):
|
||||
for m in re.findall(key + r'''\s*:\s*["'](/[A-Za-z0-9_\-/:]*)["']''', js_text):
|
||||
if not ASSET_EXT.search(m) and not API_HINT.search(m):
|
||||
r.add(m)
|
||||
return r
|
||||
|
||||
# ---- main -------------------------------------------------------------------
|
||||
def main():
|
||||
if len(sys.argv) < 3:
|
||||
print("usage: harvest_static.py <BASE_URL> <OUTDIR>"); sys.exit(1)
|
||||
base, outdir = sys.argv[1], sys.argv[2]
|
||||
if not base.startswith("http"): base = "https://" + base
|
||||
jsdir = os.path.join(outdir, "js"); os.makedirs(jsdir, exist_ok=True)
|
||||
|
||||
print(f"[*] entry: {base}")
|
||||
html, status = fetch(base)
|
||||
open(os.path.join(outdir, "index.html"), "w").write(html)
|
||||
origin = origin_of(base)
|
||||
|
||||
# initial scripts from HTML
|
||||
srcs = set(re.findall(r'<script[^>]+src="([^"]+\.js[^"]*)"', html))
|
||||
srcs |= set(re.findall(r'(?:src|href)="([^"]*\.js)"', html))
|
||||
seed = set()
|
||||
for s in srcs:
|
||||
seed.add(s if s.startswith("http") else urllib.parse.urljoin(base, s))
|
||||
print(f"[*] {len(seed)} scripts referenced in HTML")
|
||||
|
||||
have = {} # url -> local path
|
||||
def dl(url):
|
||||
fn = os.path.basename(urllib.parse.urlparse(url).path)
|
||||
if not fn.endswith(".js"): return None
|
||||
lp = os.path.join(jsdir, fn)
|
||||
if url in have: return have[url]
|
||||
data, st = fetch(url, binary=True)
|
||||
if st == 200 and data and not data[:15].lstrip().startswith(b"<"):
|
||||
open(lp, "wb").write(data); have[url] = lp; return lp
|
||||
return None
|
||||
|
||||
with ThreadPoolExecutor(max_workers=20) as ex:
|
||||
list(ex.map(dl, seed))
|
||||
|
||||
# iteratively expand via chunk manifests (chunks reference more chunks)
|
||||
seen_urls = set(have.keys()); frontier = list(have.values())
|
||||
rounds = 0
|
||||
while frontier and rounds < 6:
|
||||
rounds += 1
|
||||
new_urls = set()
|
||||
for lp in frontier:
|
||||
try: txt = open(lp, encoding="utf-8", errors="ignore").read()
|
||||
except: continue
|
||||
for cu in chunk_urls(base, txt):
|
||||
if cu not in seen_urls: new_urls.add(cu)
|
||||
seen_urls |= new_urls
|
||||
if not new_urls: break
|
||||
print(f"[*] round {rounds}: {len(new_urls)} new chunk urls from manifest")
|
||||
before = set(have.values())
|
||||
with ThreadPoolExecutor(max_workers=24) as ex:
|
||||
list(ex.map(dl, new_urls))
|
||||
frontier = [p for p in have.values() if p not in before]
|
||||
|
||||
# retry-once any manifest chunk that 404'd (transient failures are real)
|
||||
all_manifest = set()
|
||||
for lp in list(have.values()):
|
||||
try: all_manifest |= chunk_urls(base, open(lp, encoding="utf-8", errors="ignore").read())
|
||||
except: pass
|
||||
missing = [u for u in all_manifest if u not in have]
|
||||
if missing:
|
||||
with ThreadPoolExecutor(max_workers=24) as ex:
|
||||
list(ex.map(dl, missing))
|
||||
still = [u for u in all_manifest if u not in have]
|
||||
print(f"[*] manifest chunks: {len(all_manifest)} | downloaded {len(have)} | "
|
||||
f"unreachable {len(still)} (CSS-only / undeployed)")
|
||||
|
||||
print(f"[+] total JS downloaded: {len(have)}")
|
||||
|
||||
# extract from everything
|
||||
api, other, routes = set(), set(), set()
|
||||
for lp in have.values():
|
||||
try: txt = open(lp, encoding="utf-8", errors="ignore").read()
|
||||
except: continue
|
||||
a, o = extract_endpoints(txt); api |= a; other |= o
|
||||
routes |= extract_routes(txt)
|
||||
|
||||
def dump(name, items):
|
||||
path = os.path.join(outdir, name)
|
||||
open(path, "w").write("\n".join(sorted(items)))
|
||||
return path
|
||||
dump("api_static.txt", api)
|
||||
dump("paths_other.txt", other)
|
||||
dump("routes.txt", routes)
|
||||
dump("chunkmap.txt", sorted(os.path.basename(u) for u in all_manifest))
|
||||
|
||||
print(f"[+] api endpoints: {len(api)} (api_static.txt)")
|
||||
print(f"[+] other paths: {len(other)} (paths_other.txt)")
|
||||
print(f"[+] route paths: {len(routes)} (routes.txt)")
|
||||
print(f"[+] outdir: {outdir}")
|
||||
print("\n[next] reverse the 3 gate facts (see reference.md), fill config.json, "
|
||||
"then: node runtime_harvest.js config.json")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+1011
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"name": "api-recon-runtime",
|
||||
"version": "1.0.0",
|
||||
"private": true,
|
||||
"description": "Headless runtime API harvester for the api-recon skill",
|
||||
"dependencies": {
|
||||
"puppeteer-core": "^23.11.1"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,348 @@
|
||||
/**
|
||||
* api-recon coverage 模式预加载脚本 — 参考模板
|
||||
*
|
||||
* ⚠ 非通用成品:须按目标站点调整后再注入。
|
||||
* 常见改动:loginPathRe、stubs、neutralize.fields/success、apiPattern、mockTier、forward
|
||||
*
|
||||
* document-start 注入(CDP addScriptToEvaluateOnNewDocument 或 userscript)
|
||||
* CONFIG 字段应与 recon/config.json 保持一致
|
||||
*/
|
||||
(function () {
|
||||
'use strict';
|
||||
|
||||
const CONFIG = {
|
||||
loginPathRe: /\/(login|signin)(\/|$|\?)/i,
|
||||
mockTier: 'L1+L2',
|
||||
forward: true,
|
||||
neutralize: {
|
||||
fields: ['response_code', 'code', 'errno', 'ret', 'status'],
|
||||
success: 0,
|
||||
flags: { success: true, message: 'ok' },
|
||||
},
|
||||
stubs: [],
|
||||
apiPattern: /\/(api|apis|v\d+|dev|internal|graphql)\//i,
|
||||
apiFallbackRe: /^\/(api|apis|v\d+|dev|internal)\//i,
|
||||
// API 发现增强:fetch/XHR Hook、请求头与响应录制
|
||||
recordDetail: true, // 详细录制 method/url/headers/body/响应
|
||||
respMax: 600,
|
||||
extractUrlsFromResponse: true, // 从 JSON 响应里抠嵌套 URL
|
||||
neutralizeVueRouter: true, // Vue beforeEach/push 登录跳转中和
|
||||
observe: {
|
||||
storageReads: false, // 观察 localStorage.getItem(辅助确认会话键名)
|
||||
cookieReads: false, // 观察 document.cookie 读取
|
||||
xhrHeaders: true, // 录制 XHR setRequestHeader
|
||||
},
|
||||
};
|
||||
|
||||
window.__API_RECON_PRELOAD__ = true;
|
||||
window.__API_RECON_LOG__ = window.__API_RECON_LOG__ || new Set();
|
||||
window.__API_RECON_DETAIL__ = window.__API_RECON_DETAIL__ || [];
|
||||
window.__API_RECON_ROUTES__ = window.__API_RECON_ROUTES__ || new Set();
|
||||
window.__API_RECON_OBSERVE__ = window.__API_RECON_OBSERVE__ || { storage: [], headers: [] };
|
||||
|
||||
function trunc(s, n) {
|
||||
n = n || CONFIG.respMax || 600;
|
||||
s = String(s == null ? '' : s);
|
||||
return s.length > n ? s.slice(0, n) + '…' : s;
|
||||
}
|
||||
|
||||
function extractApiUrlsFromText(text) {
|
||||
if (!CONFIG.extractUrlsFromResponse || !text) return;
|
||||
const re = /["'](\/(?:api|apis|v\d+|dev|internal|graphql)[^"'`\\s]*)["'`]/gi;
|
||||
let m;
|
||||
while ((m = re.exec(text))) {
|
||||
const p = m[1].split('?')[0];
|
||||
window.__API_RECON_LOG__.add('GET ' + p);
|
||||
}
|
||||
const broad = /\/api\/[a-zA-Z0-9_./-]+/g;
|
||||
while ((m = broad.exec(text))) {
|
||||
const p = m[0].split('?')[0];
|
||||
if (!/\.(svg|png|jpg|gif|ico)$/i.test(p)) window.__API_RECON_LOG__.add('GET ' + p);
|
||||
}
|
||||
}
|
||||
|
||||
function recordApi(url, method) {
|
||||
const p = String(url || '').split('?')[0];
|
||||
if (CONFIG.apiPattern.test(p)) {
|
||||
window.__API_RECON_LOG__.add((method || 'GET').toUpperCase() + ' ' + p);
|
||||
}
|
||||
}
|
||||
|
||||
function recordApiDetail(entry) {
|
||||
recordApi(entry.url, entry.method);
|
||||
if (!CONFIG.recordDetail) return;
|
||||
window.__API_RECON_DETAIL__.push(entry);
|
||||
if (entry.responseBody) extractApiUrlsFromText(entry.responseBody);
|
||||
}
|
||||
|
||||
function blocked(url) {
|
||||
return url && CONFIG.loginPathRe.test(String(url));
|
||||
}
|
||||
|
||||
// --- 跳转中和 ---
|
||||
(function neutralizeNativeNavigation() {
|
||||
const rawAssign = Location.prototype.assign;
|
||||
const rawReplace = Location.prototype.replace;
|
||||
Location.prototype.assign = function (url) {
|
||||
if (blocked(url)) return;
|
||||
return rawAssign.call(this, url);
|
||||
};
|
||||
Location.prototype.replace = function (url) {
|
||||
if (blocked(url)) return;
|
||||
return rawReplace.call(this, url);
|
||||
};
|
||||
|
||||
const rawPush = history.pushState;
|
||||
const rawRep = history.replaceState;
|
||||
history.pushState = function (s, t, url) {
|
||||
if (blocked(url)) return;
|
||||
return rawPush.apply(this, arguments);
|
||||
};
|
||||
history.replaceState = function (s, t, url) {
|
||||
if (blocked(url)) return;
|
||||
return rawRep.apply(this, arguments);
|
||||
};
|
||||
|
||||
const hrefDesc = Object.getOwnPropertyDescriptor(Location.prototype, 'href');
|
||||
if (hrefDesc && hrefDesc.set) {
|
||||
const nativeSet = hrefDesc.set;
|
||||
Object.defineProperty(Location.prototype, 'href', {
|
||||
configurable: true,
|
||||
enumerable: hrefDesc.enumerable,
|
||||
get: hrefDesc.get,
|
||||
set(url) {
|
||||
if (blocked(url)) return;
|
||||
return nativeSet.call(this, url);
|
||||
},
|
||||
});
|
||||
}
|
||||
window.close = function () {};
|
||||
})();
|
||||
|
||||
// --- Vue Router 登录跳转中和 ---
|
||||
function neutralizeVueRouter() {
|
||||
if (!CONFIG.neutralizeVueRouter) return;
|
||||
try {
|
||||
const el = document.querySelector('[data-v-app]');
|
||||
const app = (el && el.__vue_app__) || window.__VUE__;
|
||||
if (!app) return;
|
||||
const router = app.config && app.config.globalProperties && app.config.globalProperties.$router;
|
||||
if (!router) return;
|
||||
router.beforeEach(function (to, from, next) { next(); });
|
||||
if (router.beforeResolve) router.beforeResolve(function (to, from, next) { next(); });
|
||||
function wrapNav(fn) {
|
||||
return function (loc) {
|
||||
const path = typeof loc === 'string' ? loc : (loc && (loc.path || loc.fullPath)) || '';
|
||||
if (blocked(path)) return Promise.resolve();
|
||||
return fn.apply(this, arguments);
|
||||
};
|
||||
}
|
||||
router.push = wrapNav(router.push.bind(router));
|
||||
router.replace = wrapNav(router.replace.bind(router));
|
||||
if (router.getRoutes) {
|
||||
router.getRoutes().forEach(function (r) {
|
||||
if (r.path) window.__API_RECON_ROUTES__.add(r.path);
|
||||
});
|
||||
}
|
||||
} catch (e) { /* ignore */ }
|
||||
}
|
||||
|
||||
function runPostLoadHooks() {
|
||||
neutralizeVueRouter();
|
||||
// 业务层跳转函数(goPage / navigateTo 等)
|
||||
['goPage', 'navigateTo', 'jumpTo', 'redirectTo'].forEach(function (name) {
|
||||
if (typeof window[name] !== 'function' || window[name].__apiReconWrapped) return;
|
||||
const raw = window[name];
|
||||
window[name] = function () {
|
||||
const arg = arguments[0];
|
||||
const path = typeof arg === 'string' ? arg : (arg && arg.path) || '';
|
||||
if (blocked(path)) return;
|
||||
return raw.apply(this, arguments);
|
||||
};
|
||||
window[name].__apiReconWrapped = true;
|
||||
});
|
||||
}
|
||||
document.addEventListener('DOMContentLoaded', runPostLoadHooks);
|
||||
window.addEventListener('load', runPostLoadHooks);
|
||||
|
||||
// --- 观察 Hook:辅助发现会话键名与请求头(可选,默认关 storage/cookie)---
|
||||
if (CONFIG.observe && CONFIG.observe.storageReads) {
|
||||
const rawGet = Storage.prototype.getItem;
|
||||
Storage.prototype.getItem = function (key) {
|
||||
window.__API_RECON_OBSERVE__.storage.push({ type: 'getItem', key: key, at: location.pathname });
|
||||
return rawGet.apply(this, arguments);
|
||||
};
|
||||
}
|
||||
if (CONFIG.observe && CONFIG.observe.cookieReads) {
|
||||
const cookieDesc = Object.getOwnPropertyDescriptor(Document.prototype, 'cookie');
|
||||
if (cookieDesc && cookieDesc.get) {
|
||||
const nativeGet = cookieDesc.get;
|
||||
Object.defineProperty(document, 'cookie', {
|
||||
configurable: true,
|
||||
get: function () {
|
||||
window.__API_RECON_OBSERVE__.storage.push({ type: 'cookieRead', at: location.pathname });
|
||||
return nativeGet.call(this);
|
||||
},
|
||||
set: cookieDesc.set,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// --- Mock 辅助 ---
|
||||
const NEGATIVE_RE = /未登录|未授权|授权|not\s*login|unauthorized|forbidden/i;
|
||||
const tier = CONFIG.mockTier || 'L1+L2';
|
||||
|
||||
function patchJsonBody(text) {
|
||||
if (!tier.includes('L2')) return text;
|
||||
try {
|
||||
const j = JSON.parse(text);
|
||||
if (j && typeof j === 'object') {
|
||||
const msg = String(j.message || j.msg || '');
|
||||
for (const f of CONFIG.neutralize.fields) {
|
||||
if (f in j && j[f] !== CONFIG.neutralize.success && NEGATIVE_RE.test(msg)) {
|
||||
j[f] = CONFIG.neutralize.success;
|
||||
}
|
||||
}
|
||||
Object.assign(j, CONFIG.neutralize.flags || {});
|
||||
if (j.data == null) j.data = {};
|
||||
return JSON.stringify(j);
|
||||
}
|
||||
} catch (e) { /* ignore */ }
|
||||
return text;
|
||||
}
|
||||
|
||||
function lookupPrecise(url) {
|
||||
if (!tier.includes('L1')) return null;
|
||||
const p = String(url || '');
|
||||
for (const s of CONFIG.stubs) {
|
||||
if (s._re ? s._re.test(p) : new RegExp(s.match).test(p)) return s.body;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function fallbackBody(url) {
|
||||
const p = String(url).split('?')[0];
|
||||
if (/\/(list|search|query|page|all|tree|nodes|options)/i.test(p)) {
|
||||
return { response_code: 0, data: [] };
|
||||
}
|
||||
if (/\/(get|detail|info|config|status)/i.test(p)) {
|
||||
return { response_code: 0, data: {} };
|
||||
}
|
||||
return { response_code: 0, data: {}, success: true };
|
||||
}
|
||||
|
||||
function matchMock(url) {
|
||||
const precise = lookupPrecise(url);
|
||||
if (precise) return precise;
|
||||
if (tier.includes('L3') && CONFIG.apiFallbackRe.test(String(url || '').split('?')[0])) {
|
||||
return fallbackBody(url);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
CONFIG.stubs.forEach(function (s) {
|
||||
if (s.match && !s._re) s._re = new RegExp(s.match);
|
||||
});
|
||||
|
||||
// --- fetch Hook ---
|
||||
const rawFetch = window.fetch;
|
||||
window.fetch = async function (input, init) {
|
||||
const url = typeof input === 'string' ? input : (input && input.url) || '';
|
||||
const method = ((init && init.method) || 'GET').toUpperCase();
|
||||
const reqHeaders = (init && init.headers) || {};
|
||||
const reqBody = init && init.body ? trunc(init.body) : null;
|
||||
|
||||
const mock = matchMock(url);
|
||||
if (mock && !CONFIG.forward) {
|
||||
recordApiDetail({ method: method, url: url, reqHeaders: reqHeaders, reqBody: reqBody, status: 200, responseBody: JSON.stringify(mock), source: 'mock' });
|
||||
return new Response(JSON.stringify(mock), {
|
||||
status: 200,
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
});
|
||||
}
|
||||
|
||||
const resp = await rawFetch.apply(this, arguments);
|
||||
const clone = resp.clone();
|
||||
let text = '';
|
||||
try { text = await clone.text(); } catch (e) { /* ignore */ }
|
||||
|
||||
recordApiDetail({
|
||||
method: method, url: url, reqHeaders: reqHeaders, reqBody: reqBody,
|
||||
status: resp.status, responseBody: trunc(text), source: 'fetch',
|
||||
});
|
||||
|
||||
if (!tier.includes('L2') && !mock) return resp;
|
||||
|
||||
const patched = patchJsonBody(text);
|
||||
if (patched !== text) {
|
||||
return new Response(patched, {
|
||||
status: resp.status,
|
||||
statusText: resp.statusText,
|
||||
headers: resp.headers,
|
||||
});
|
||||
}
|
||||
return resp;
|
||||
};
|
||||
|
||||
// --- XHR Hook ---
|
||||
const rawOpen = XMLHttpRequest.prototype.open;
|
||||
const rawSend = XMLHttpRequest.prototype.send;
|
||||
const rawSetHeader = XMLHttpRequest.prototype.setRequestHeader;
|
||||
|
||||
XMLHttpRequest.prototype.open = function (method, url) {
|
||||
this.__apiReconUrl = url;
|
||||
this.__apiReconMethod = method;
|
||||
this.__apiReconHeaders = {};
|
||||
return rawOpen.apply(this, arguments);
|
||||
};
|
||||
|
||||
if (CONFIG.observe && CONFIG.observe.xhrHeaders) {
|
||||
XMLHttpRequest.prototype.setRequestHeader = function (name, value) {
|
||||
if (!this.__apiReconHeaders) this.__apiReconHeaders = {};
|
||||
this.__apiReconHeaders[name] = value;
|
||||
window.__API_RECON_OBSERVE__.headers.push({ name: name, url: this.__apiReconUrl });
|
||||
return rawSetHeader.apply(this, arguments);
|
||||
};
|
||||
}
|
||||
|
||||
XMLHttpRequest.prototype.send = function (body) {
|
||||
const xhr = this;
|
||||
const url = xhr.__apiReconUrl || '';
|
||||
const method = (xhr.__apiReconMethod || 'GET').toUpperCase();
|
||||
const mock = matchMock(url);
|
||||
|
||||
if (mock && !CONFIG.forward) {
|
||||
const bodyStr = JSON.stringify(mock);
|
||||
recordApiDetail({ method: method, url: url, reqHeaders: xhr.__apiReconHeaders, reqBody: trunc(body), status: 200, responseBody: bodyStr, source: 'mock' });
|
||||
setTimeout(function () {
|
||||
Object.defineProperty(xhr, 'readyState', { configurable: true, get: function () { return 4; } });
|
||||
Object.defineProperty(xhr, 'status', { configurable: true, get: function () { return 200; } });
|
||||
Object.defineProperty(xhr, 'responseText', { configurable: true, get: function () { return bodyStr; } });
|
||||
xhr.onreadystatechange && xhr.onreadystatechange();
|
||||
xhr.onload && xhr.onload();
|
||||
}, 0);
|
||||
return;
|
||||
}
|
||||
|
||||
const orig = xhr.onreadystatechange;
|
||||
xhr.onreadystatechange = function () {
|
||||
if (xhr.readyState === 4) {
|
||||
recordApiDetail({
|
||||
method: method, url: url, reqHeaders: xhr.__apiReconHeaders,
|
||||
reqBody: trunc(body), status: xhr.status,
|
||||
responseBody: trunc(xhr.responseText), source: 'xhr',
|
||||
});
|
||||
if (tier.includes('L2') && xhr.responseText) {
|
||||
try {
|
||||
const patched = patchJsonBody(xhr.responseText);
|
||||
if (patched !== xhr.responseText) {
|
||||
Object.defineProperty(xhr, 'responseText', { configurable: true, get: function () { return patched; } });
|
||||
}
|
||||
} catch (e) { /* ignore */ }
|
||||
}
|
||||
}
|
||||
orig && orig.apply(this, arguments);
|
||||
};
|
||||
return rawSend.apply(this, arguments);
|
||||
};
|
||||
})();
|
||||
@@ -0,0 +1,228 @@
|
||||
#!/usr/bin/env node
|
||||
/*
|
||||
* runtime_harvest.js <config.json>
|
||||
*
|
||||
* 参考模板 — 非通用成品。执行前须按目标站点调整 config.json 及脚本内逻辑:
|
||||
* cookies/localStorage、neutralize 字段、stubs 结构、loginUrlPattern、apiPattern
|
||||
*
|
||||
* Drives a headless browser through an authorized SPA to capture the live API
|
||||
* surface (method + url + body) by defeating three client-side gates:
|
||||
* 1. render gate -> inject fake auth state (cookies / localStorage)
|
||||
* 2. interceptor gate-> rewrite the "unauthorized" code field to success
|
||||
* 3. content gate -> stub the menu/permission endpoint with a full-feature payload
|
||||
*
|
||||
* Requires puppeteer-core + a system Chromium. ( cd scripts && npm install )
|
||||
*
|
||||
* config.json schema (all fields optional except baseUrl):
|
||||
* {
|
||||
* "baseUrl": "https://target/",
|
||||
* "chromium": "/usr/bin/chromium", // or env CHROMIUM
|
||||
* "cookies": [{"name":"auth","value":"b64json:{\"id\":1,\"username\":\"admin\"}"}],
|
||||
* "localStorage": {"token":"x","isLogin":"1"},
|
||||
* "neutralize": { "fields":["response_code","code","errno","ret"], "success":0,
|
||||
* "flags":{"success":true,"message":"ok"} },
|
||||
* "forward": true, // forward real req then rewrite code; false = offline stub
|
||||
* "loginUrlPattern": "/login", // navigations matching this are suppressed as a fallback
|
||||
* "stubs": [ {"match":"permission|menu|role", "body": { ...full-feature menu... }} ],
|
||||
* "routes": ["/dashboard","/device", ...], // from routes.txt or the forged menu
|
||||
* "apiPattern": "/api/|/rest/|/graphql", // what counts as an API call to record
|
||||
* "proxy": "http://127.0.0.1:8080", // optional; also HTTP_PROXY / HTTPS_PROXY
|
||||
* "waitUntil": "domcontentloaded", // prefer over networkidle2 for large SPAs
|
||||
* "routeTimeout": 12000,
|
||||
* "waitMs": 1200, "perRouteMs": 900, "headless": true
|
||||
* }
|
||||
*/
|
||||
process.env.NODE_TLS_REJECT_UNAUTHORIZED = '0';
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
let puppeteer;
|
||||
try { puppeteer = require('puppeteer-core'); }
|
||||
catch (e) { console.error("[!] run `npm install` in the scripts/ dir first (needs puppeteer-core)"); process.exit(1); }
|
||||
|
||||
function loadCfg(p) {
|
||||
const c = JSON.parse(fs.readFileSync(p, 'utf8'));
|
||||
c.baseUrl || (() => { throw new Error("config.baseUrl required"); })();
|
||||
c.chromium = c.chromium || process.env.CHROMIUM || '/usr/bin/chromium';
|
||||
c.neutralize = c.neutralize || { fields: ['response_code', 'code', 'errno', 'ret', 'status'], success: 0, flags: { success: true } };
|
||||
c.forward = c.forward !== false;
|
||||
c.apiPattern = new RegExp(c.apiPattern || '/api/|/rest/|/service/|/graphql|/gateway/');
|
||||
c.loginUrlPattern = c.loginUrlPattern || '/login';
|
||||
c.routes = c.routes || ['/'];
|
||||
c.waitMs = c.waitMs || 1200; c.perRouteMs = c.perRouteMs || 900;
|
||||
c.headless = c.headless !== false;
|
||||
c.waitUntil = c.waitUntil || 'domcontentloaded';
|
||||
c.routeTimeout = c.routeTimeout || 12000;
|
||||
c.captureResponses = c.captureResponses !== false; // record response body samples (forward mode)
|
||||
c.recordWs = c.recordWs !== false; // record WebSocket frames + SSE endpoints
|
||||
c.respMax = c.respMax || 600; // truncation length for captured bodies
|
||||
c.proxy = c.proxy || process.env.HTTP_PROXY || process.env.HTTPS_PROXY || '';
|
||||
(c.stubs || []).forEach(s => s._re = new RegExp(s.match));
|
||||
return c;
|
||||
}
|
||||
|
||||
function cookieValue(v) {
|
||||
if (typeof v === 'string' && v.startsWith('b64json:'))
|
||||
return Buffer.from(v.slice(8)).toString('base64');
|
||||
if (typeof v === 'string' && v.startsWith('json:'))
|
||||
return v.slice(5);
|
||||
return v;
|
||||
}
|
||||
|
||||
function neutralize(txt, n) {
|
||||
try {
|
||||
const j = JSON.parse(txt);
|
||||
if (j && typeof j === 'object') {
|
||||
for (const f of n.fields) if (f in j) j[f] = n.success;
|
||||
Object.assign(j, n.flags || {});
|
||||
return JSON.stringify(j);
|
||||
}
|
||||
} catch (e) {}
|
||||
return txt;
|
||||
}
|
||||
|
||||
(async () => {
|
||||
const cfg = loadCfg(process.argv[2] || 'config.json');
|
||||
const origin = new URL(cfg.baseUrl).origin;
|
||||
const host = new URL(cfg.baseUrl).hostname;
|
||||
const rec = []; // {m,u,b,resp,ct}
|
||||
const chunks = new Set();
|
||||
const ws = []; // {url,dir,data} WebSocket frames
|
||||
const sse = new Set(); // SSE (text/event-stream) endpoints
|
||||
|
||||
if (cfg.proxy) {
|
||||
process.env.HTTP_PROXY = cfg.proxy;
|
||||
process.env.HTTPS_PROXY = cfg.proxy;
|
||||
}
|
||||
const launchArgs = ['--no-sandbox', '--disable-dev-shm-usage', '--ignore-certificate-errors'];
|
||||
if (cfg.proxy) launchArgs.push(`--proxy-server=${cfg.proxy}`);
|
||||
const browser = await puppeteer.launch({
|
||||
executablePath: cfg.chromium, headless: cfg.headless ? 'new' : false,
|
||||
args: launchArgs
|
||||
});
|
||||
const page = await browser.newPage();
|
||||
|
||||
// ---- WebSocket frame capture via CDP (fetch/XHR interception can't see WS) ----
|
||||
if (cfg.recordWs) {
|
||||
try {
|
||||
const cdp = await page.target().createCDPSession();
|
||||
await cdp.send('Network.enable');
|
||||
const wsUrl = {}; // requestId -> url
|
||||
cdp.on('Network.webSocketCreated', e => { wsUrl[e.requestId] = e.url; });
|
||||
const onFrame = dir => e => {
|
||||
const p = e.response && e.response.payloadData;
|
||||
if (p != null) ws.push({ url: (wsUrl[e.requestId] || '').replace(origin, ''), dir, data: String(p).slice(0, cfg.respMax) });
|
||||
};
|
||||
cdp.on('Network.webSocketFrameSent', onFrame('send'));
|
||||
cdp.on('Network.webSocketFrameReceived', onFrame('recv'));
|
||||
} catch (e) { console.log('[ws] CDP capture unavailable:', e.message); }
|
||||
}
|
||||
|
||||
// inject localStorage on every document
|
||||
if (cfg.localStorage) {
|
||||
await page.evaluateOnNewDocument((kv) => {
|
||||
try { for (const k in kv) localStorage.setItem(k, kv[k]); } catch (e) {}
|
||||
}, cfg.localStorage);
|
||||
}
|
||||
// fallback: block full-page navigations to the login url
|
||||
await page.evaluateOnNewDocument((pat) => {
|
||||
const bad = u => { try { return String(u).indexOf(pat) >= 0; } catch (e) { return false; } };
|
||||
try {
|
||||
const d = Object.getOwnPropertyDescriptor(Location.prototype, 'href');
|
||||
Object.defineProperty(Location.prototype, 'href', { configurable: true,
|
||||
get() { return d.get.call(this); },
|
||||
set(v) { if (bad(v)) return; return d.set.call(this, v); } });
|
||||
const a = Location.prototype.assign, r = Location.prototype.replace;
|
||||
Location.prototype.assign = function (v) { if (bad(v)) return; return a.call(this, v); };
|
||||
Location.prototype.replace = function (v) { if (bad(v)) return; return r.call(this, v); };
|
||||
} catch (e) {}
|
||||
}, cfg.loginUrlPattern);
|
||||
|
||||
await page.setRequestInterception(true);
|
||||
page.on('request', async (req) => {
|
||||
const u = req.url(), m = req.method();
|
||||
if (/\.js(\?|$)/.test(u) && /\/assets\/|\/static\/|\/js\//.test(u)) chunks.add(u.split('/').pop());
|
||||
|
||||
// suppress fallback login navigations (after the app has bootstrapped)
|
||||
if (req.isNavigationRequest() && req.frame() === page.mainFrame()
|
||||
&& u.includes(cfg.loginUrlPattern) && rec.length > 3) {
|
||||
return req.respond({ status: 204, body: '' });
|
||||
}
|
||||
if (!cfg.apiPattern.test(u)) return req.continue();
|
||||
|
||||
const entry = { m, u: u.replace(origin, '').split('?')[0], full: u, b: req.postData() ? req.postData().slice(0, 400) : null };
|
||||
rec.push(entry);
|
||||
|
||||
// explicit stubs (menu / permission forgery) win
|
||||
const stub = (cfg.stubs || []).find(s => s._re.test(u));
|
||||
if (stub) return req.respond({ status: 200, contentType: 'application/json', body: JSON.stringify(stub.body) });
|
||||
|
||||
// SSE: forwarding a text/event-stream would hang the handler — record + short-circuit
|
||||
const accept = (req.headers().accept || '');
|
||||
if (/text\/event-stream/.test(accept)) { sse.add(entry.u); return req.respond({ status: 200, contentType: 'application/json', body: '{}' }); }
|
||||
|
||||
if (!cfg.forward) {
|
||||
return req.respond({ status: 200, contentType: 'application/json',
|
||||
body: neutralize('{"data":{},"list":[],"total":0}', cfg.neutralize) });
|
||||
}
|
||||
// forward real request, then rewrite the unauthorized code field
|
||||
try {
|
||||
const headers = Object.assign({}, req.headers());
|
||||
if (cfg.cookies) headers.cookie = cfg.cookies.map(c => `${c.name}=${cookieValue(c.value)}`).join('; ');
|
||||
const r = await fetch(u, { method: m, headers, body: (m !== 'GET' && m !== 'HEAD') ? req.postData() : undefined });
|
||||
const t = await r.text();
|
||||
if (cfg.captureResponses) { entry.resp = t.slice(0, cfg.respMax); entry.ct = r.headers.get('content-type') || ''; }
|
||||
req.respond({ status: 200, contentType: 'application/json', body: neutralize(t, cfg.neutralize) });
|
||||
} catch (e) {
|
||||
req.respond({ status: 200, contentType: 'application/json', body: '{"response_code":0,"code":0,"data":{}}' });
|
||||
}
|
||||
});
|
||||
|
||||
// set auth cookies
|
||||
if (cfg.cookies) {
|
||||
for (const c of cfg.cookies)
|
||||
await page.setCookie({ name: c.name, value: cookieValue(c.value), domain: host, path: c.path || '/' });
|
||||
}
|
||||
|
||||
// boot
|
||||
await page.goto(cfg.baseUrl, { waitUntil: cfg.waitUntil, timeout: 45000 }).catch(e => console.log('[goto]', e.message));
|
||||
await new Promise(r => setTimeout(r, cfg.waitMs));
|
||||
const shell = await page.evaluate(() => ({
|
||||
url: location.href,
|
||||
loginForm: !!document.querySelector('input[type=password]'),
|
||||
links: [...document.querySelectorAll('a[href^="/"]')].map(a => a.getAttribute('href'))
|
||||
}));
|
||||
console.log(`[*] after boot: url=${shell.url} loginForm=${shell.loginForm}`);
|
||||
if (shell.loginForm) console.log('[!] still on login — recheck render-gate facts (cookies/localStorage) in config');
|
||||
|
||||
// discovered menu links extend the route list
|
||||
const routes = [...new Set([...cfg.routes, ...shell.links.filter(h => h && h.length > 1)])];
|
||||
const perRoute = {};
|
||||
for (const rt of routes) {
|
||||
const before = rec.length;
|
||||
try { await page.goto(origin + rt, { waitUntil: cfg.waitUntil, timeout: cfg.routeTimeout }); } catch (e) {}
|
||||
await new Promise(r => setTimeout(r, cfg.perRouteMs));
|
||||
const fresh = [...new Set(rec.slice(before).map(r => r.m + ' ' + r.u))];
|
||||
perRoute[rt] = fresh;
|
||||
}
|
||||
|
||||
const uniq = [...new Set(rec.map(r => r.m + ' ' + r.u))].sort();
|
||||
const wsUniq = [...new Set(ws.map(f => f.url))].filter(Boolean).sort();
|
||||
const outdir = path.dirname(process.argv[2] || '.');
|
||||
fs.writeFileSync(path.join(outdir, 'runtime_api.json'),
|
||||
JSON.stringify({ shell, uniq, perRoute, rec, ws, wsEndpoints: wsUniq, sse: [...sse] }, null, 2));
|
||||
|
||||
// merge with static
|
||||
let merged = new Set(uniq.map(x => x.split(' ')[1]));
|
||||
const staticFile = path.join(outdir, 'api_static.txt');
|
||||
if (fs.existsSync(staticFile)) fs.readFileSync(staticFile, 'utf8').split('\n').filter(Boolean).forEach(p => merged.add(p));
|
||||
fs.writeFileSync(path.join(outdir, 'api_merged.txt'), [...merged].sort().join('\n'));
|
||||
|
||||
console.log(`[+] runtime endpoints (with method): ${uniq.length}`);
|
||||
console.log(`[+] chunks seen at runtime: ${chunks.size}`);
|
||||
if (cfg.recordWs) console.log(`[+] websocket endpoints: ${wsUniq.length} (${ws.length} frames) | SSE endpoints: ${sse.size}`);
|
||||
if (cfg.captureResponses) console.log(`[+] response bodies captured for ${rec.filter(r => r.resp != null).length}/${rec.length} calls`);
|
||||
console.log(`[+] merged unique paths: ${merged.size} -> api_merged.txt`);
|
||||
console.log(`[+] full detail -> runtime_api.json (per-route + bodies + ws frames + sse)`);
|
||||
await browser.close();
|
||||
})();
|
||||
@@ -0,0 +1,123 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
spider_mpa.py <BASE_URL> <OUTDIR> [--cookie "k=v; k2=v2"] [--max 200] [--depth 4]
|
||||
|
||||
参考模板 — 非通用成品。执行前须按目标调整 --exclude、cookie、depth/max 等同域策略。
|
||||
|
||||
Fallback for NON-SPA targets (traditional server-rendered MPAs: Django/Rails/PHP/
|
||||
JSP, classic admin panels). When there is no JS endpoint bundle, the API surface
|
||||
lives in HTML <form action>, <a href>, and inline-JS ajax urls. This BFS-crawls
|
||||
the same origin and extracts:
|
||||
- forms.txt : METHOD action [param1, param2, ...] (the real "endpoints")
|
||||
- links.txt : every same-origin URL reached
|
||||
- api_inline.txt : url-ish strings found in inline <script> / onclick (fetch/ajax)
|
||||
|
||||
Stdlib only. Same-origin, bounded, polite. Provide --cookie for an authed crawl.
|
||||
"""
|
||||
import sys, os, re, ssl, argparse, urllib.request, urllib.parse, time
|
||||
from html.parser import HTMLParser
|
||||
from collections import deque
|
||||
|
||||
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
|
||||
UA = "Mozilla/5.0 (spa-api-recon spider)"
|
||||
SKIP_EXT = re.compile(r'\.(css|png|jpe?g|gif|svg|ico|woff2?|ttf|pdf|zip|mp4|webp|js)(\?|$)', re.I)
|
||||
API_HINT = re.compile(r'/(?:api|rest|service|ajax|action|do|rpc|graphql|v\d|admin|backend)\b', re.I)
|
||||
|
||||
def fetch(url, cookie):
|
||||
h = {"User-Agent": UA}
|
||||
if cookie: h["Cookie"] = cookie
|
||||
try:
|
||||
with urllib.request.urlopen(urllib.request.Request(url, headers=h), context=CTX, timeout=20) as r:
|
||||
ct = r.headers.get("Content-Type", "")
|
||||
if "html" not in ct and "xml" not in ct: return "", ct
|
||||
return r.read().decode("utf-8", "ignore"), ct
|
||||
except Exception:
|
||||
return "", ""
|
||||
|
||||
class Page(HTMLParser):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.links, self.scripts, self.inline = [], [], []
|
||||
self.forms, self._cur = [], None
|
||||
self._in_script = False
|
||||
def handle_starttag(self, tag, attrs):
|
||||
a = dict(attrs)
|
||||
if tag == "a" and a.get("href"): self.links.append(a["href"])
|
||||
elif tag == "script":
|
||||
self._in_script = True
|
||||
if a.get("src"): self.scripts.append(a["src"])
|
||||
elif tag == "form":
|
||||
self._cur = {"action": a.get("action", ""), "method": (a.get("method") or "GET").upper(), "params": []}
|
||||
elif tag in ("input", "select", "textarea", "button") and self._cur is not None:
|
||||
n = a.get("name")
|
||||
if n: self._cur["params"].append(n)
|
||||
# ajax-ish handlers
|
||||
for v in a.values():
|
||||
for m in re.findall(r'''["'](/[^"']{2,120})["']''', v or ""):
|
||||
if API_HINT.search(m): self.inline.append(m)
|
||||
def handle_endtag(self, tag):
|
||||
if tag == "script": self._in_script = False
|
||||
elif tag == "form" and self._cur is not None:
|
||||
self.forms.append(self._cur); self._cur = None
|
||||
def handle_data(self, data):
|
||||
if self._in_script and data:
|
||||
for m in re.findall(r'''["'`](/[A-Za-z0-9_\-./{}$:?=&]{2,120})["'`]''', data):
|
||||
if API_HINT.search(m): self.inline.append(m)
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("base"); ap.add_argument("outdir")
|
||||
ap.add_argument("--cookie", default=""); ap.add_argument("--max", type=int, default=200)
|
||||
ap.add_argument("--depth", type=int, default=4); ap.add_argument("--delay", type=float, default=0.1)
|
||||
ap.add_argument("--include", default=""); ap.add_argument("--exclude", default="logout|signout|delete|remove|destroy")
|
||||
args = ap.parse_args()
|
||||
base = args.base if args.base.startswith("http") else "https://" + args.base
|
||||
os.makedirs(args.outdir, exist_ok=True)
|
||||
origin = "{0.scheme}://{0.netloc}".format(urllib.parse.urlparse(base))
|
||||
inc = re.compile(args.include) if args.include else None
|
||||
exc = re.compile(args.exclude, re.I) if args.exclude else None
|
||||
|
||||
seen, forms, inline, links = set(), {}, set(), set()
|
||||
q = deque([(base, 0)]); seen.add(base.split("#")[0])
|
||||
n = 0
|
||||
while q and n < args.max:
|
||||
url, d = q.popleft(); n += 1
|
||||
html, ct = fetch(url, args.cookie)
|
||||
if args.delay: time.sleep(args.delay)
|
||||
if not html: continue
|
||||
p = Page()
|
||||
try: p.feed(html)
|
||||
except Exception: pass
|
||||
links.add(url.replace(origin, "") or "/")
|
||||
for f in p.forms:
|
||||
act = urllib.parse.urljoin(url, f["action"] or url)
|
||||
key = f["method"] + " " + act.replace(origin, "")
|
||||
forms.setdefault(key, set()).update(f["params"])
|
||||
for s in p.inline: inline.add(s)
|
||||
if d < args.depth:
|
||||
for href in p.links:
|
||||
if href.startswith(("mailto:", "tel:", "javascript:", "#")): continue
|
||||
nxt = urllib.parse.urljoin(url, href).split("#")[0]
|
||||
if not nxt.startswith(origin): continue
|
||||
if SKIP_EXT.search(nxt): continue
|
||||
if exc and exc.search(nxt): continue
|
||||
if inc and not inc.search(nxt): continue
|
||||
if nxt not in seen:
|
||||
seen.add(nxt); q.append((nxt, d + 1))
|
||||
|
||||
with open(os.path.join(args.outdir, "forms.txt"), "w") as fh:
|
||||
for k in sorted(forms):
|
||||
ps = ", ".join(sorted(forms[k]))
|
||||
fh.write(f"{k} [{ps}]\n")
|
||||
open(os.path.join(args.outdir, "links.txt"), "w").write("\n".join(sorted(links)))
|
||||
open(os.path.join(args.outdir, "api_inline.txt"), "w").write("\n".join(sorted(inline)))
|
||||
|
||||
print(f"[+] crawled {n} pages (cap {args.max}, depth {args.depth})")
|
||||
print(f"[+] forms (endpoints): {len(forms)} -> forms.txt")
|
||||
print(f"[+] inline ajax urls: {len(inline)} -> api_inline.txt")
|
||||
print(f"[+] pages reached: {len(links)} -> links.txt")
|
||||
if not args.cookie:
|
||||
print("[i] no --cookie: only public pages crawled. Pass an authed session cookie for the full surface.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user