First Commit
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s

This commit is contained in:
dela
2026-10-09 08:38:16 +08:00
commit 0335d572de
756 changed files with 201663 additions and 0 deletions
+210
View File
@@ -0,0 +1,210 @@
#!/usr/bin/env python3
"""
build_perm_tree.py <JSDIR> <OUTDIR> [--config recon/config.json]
Rebuild a permissive permission/menu stub from frontend auth modules.
Typical consumer chain (TDP / enterprise admin SPAs):
POST /api/web/user/role_permissions -> { permissions: [...], role_type }
POST /api/web/permissions/all -> tree[{ code, position, children }]
getResultTree(tree, permissions) -> menu include lists
userRouteAuth[code].url -> frontend route path
This script:
1. Locates userRouteAuth={MONITOR:{url:...},...} in js/
2. Resolves webpack alias refs (He=o.DASHBOARD) via route_map.json
3. Infers hierarchy from code prefixes (MONITOR_ -> MONITOR)
4. Writes permissions_tree.json, permissions_all_stub.json,
role_permissions_stub.json, userRouteAuth.json
5. Optionally patches config.json stubs (permissions/all + role_permissions)
Adjust ROOTS / PREFIX_PARENT / EXTRA_PARENT per target if heuristics miss nodes.
"""
import re, json, os, sys, glob, argparse
DEFAULT_ROOTS = [
'MONITOR', 'THREAT', 'ASSETS_RISK', 'INVESTIGATION', 'MANAGEMENT',
'AGENT_EVIDENCE', 'MDR', 'PLATFORM',
]
DEFAULT_PREFIX_PARENT = {
'MONITOR_': 'MONITOR', 'THREAT_': 'THREAT', 'ASSETS_RISK_': 'ASSETS_RISK',
'INVESTIGATION_': 'INVESTIGATION', 'MANAGEMENT_': 'MANAGEMENT', 'MDR_': 'MDR',
'PLATFORM_': 'PLATFORM', 'CUSTOM_': 'PLATFORM', 'FALSE_': 'PLATFORM',
}
DEFAULT_EXTRA_PARENT = {
'LINKAGE_DISPOSAL': 'MANAGEMENT',
'ASSETS_RISK_API': 'ASSETS_RISK',
'ASSETS_RISK_WEAK_PWD': 'ASSETS_RISK',
}
def find_auth_file(jsdir):
best_fp, best_len = None, 0
for fp in glob.glob(os.path.join(jsdir, '*.js')):
try:
txt = open(fp, encoding='utf-8', errors='ignore').read()
except Exception:
continue
if 'userRouteAuth' not in txt:
continue
m = re.search(r'userRouteAuth=\{MONITOR:', txt)
if m and len(txt) > best_len:
best_fp, best_len = fp, len(m.group(0))
elif 'userRouteAuth' in txt and best_fp is None:
best_fp = fp
return best_fp
def parse_aliases(chunk):
aliases = {}
for m in re.finditer(r'([a-zA-Z_$][\w$]*)=o\.([A-Z_0-9]+)\b', chunk[:8000]):
aliases[m.group(1)] = m.group(2)
return aliases
def resolve_ref(token, aliases, route_map):
token = token.strip()
if token.startswith('"'):
return json.loads(token)
if token in aliases:
key = aliases[token]
return route_map.get(key, {}).get('link', key)
return token
def parse_user_route_auth(txt, route_map):
m = re.search(r'(?:t\.)?userRouteAuth=(\{MONITOR:.*?\})\},', txt, re.S)
if not m:
m = re.search(r'userRouteAuth=(\{[A-Z_0-9]+:\{url:', txt)
if not m:
return None, {}
# greedy fallback — trim at next webpack module
body = m.group(1)
end = body.rfind('}')
obj_src = body[: end + 1] if end > 0 else body
else:
obj_src = m.group(1)
chunk_start = txt.find('userRouteAuth')
chunk = txt[chunk_start:chunk_start + 35000]
aliases = parse_aliases(chunk)
entries = {}
for em in re.finditer(
r'([A-Z_0-9]+):\{url:([^,}]+)(?:,control:(\[.*?\]|[^,}]+))?\}', obj_src
):
key = em.group(1)
url = resolve_ref(em.group(2).strip(), aliases, route_map)
controls = []
if em.group(3):
raw = em.group(3).strip()
vars_ = re.findall(r'([A-Za-z_$][\w$]*)', raw) if raw.startswith('[') else [raw]
controls = [aliases.get(v, v) for v in vars_]
entries[key] = {'code': key, 'url': url, 'control': controls}
return obj_src, entries
def parent_of(code, roots, prefix_parent, extra_parent):
if code in extra_parent:
return extra_parent[code]
if code in roots:
return None
for pref, par in prefix_parent.items():
if code.startswith(pref):
return par
return None
def build_tree(entries, roots, prefix_parent, extra_parent):
children_of = {k: [] for k in entries}
for code in entries:
p = parent_of(code, roots, prefix_parent, extra_parent)
if p:
children_of.setdefault(p, []).append(code)
def make_node(code):
node = {
'code': code,
'position': 'top' if code in roots else 'left',
'name': code,
}
kids = sorted(children_of.get(code, []))
if kids:
node['children'] = [make_node(c) for c in kids]
return node
tree = [make_node(r) for r in roots if r in entries or children_of.get(r)]
orphans = [c for c in entries if parent_of(c, roots, prefix_parent, extra_parent) is None and c not in roots]
for code in sorted(orphans):
tree.append(make_node(code))
return tree
def flat_codes(nodes):
out = []
for n in nodes:
out.append(n['code'])
out.extend(flat_codes(n.get('children', [])))
return out
def main():
ap = argparse.ArgumentParser(description='Build permission tree stubs from JS auth modules')
ap.add_argument('jsdir')
ap.add_argument('outdir')
ap.add_argument('--config', help='patch stubs into config.json')
ap.add_argument('--role', default='SUPER_ADMIN', help='role_type in role_permissions stub')
args = ap.parse_args()
os.makedirs(args.outdir, exist_ok=True)
route_map_path = os.path.join(args.outdir, 'route_map.json')
if not os.path.exists(route_map_path):
print('[*] route_map.json missing — run extract_route_map.py first')
route_map = {}
else:
route_map = json.load(open(route_map_path, encoding='utf-8'))
auth_fp = find_auth_file(args.jsdir)
if not auth_fp:
print('[!] userRouteAuth module not found in js/')
sys.exit(1)
print(f'[*] auth module: {os.path.basename(auth_fp)}')
txt = open(auth_fp, encoding='utf-8', errors='ignore').read()
_, entries = parse_user_route_auth(txt, route_map)
if not entries:
print('[!] failed to parse userRouteAuth object — adjust regex in script')
sys.exit(1)
tree = build_tree(entries, DEFAULT_ROOTS, DEFAULT_PREFIX_PARENT, DEFAULT_EXTRA_PARENT)
all_codes = sorted(set(flat_codes(tree) + [c for e in entries.values() for c in e.get('control', [])]))
perm_all = {'response_code': 0, 'verbose_msg': 'ok', 'data': tree}
role_perm = {
'response_code': 0,
'verbose_msg': 'ok',
'data': {'role_type': args.role, 'permissions': all_codes},
}
json.dump(entries, open(os.path.join(args.outdir, 'userRouteAuth.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
json.dump(tree, open(os.path.join(args.outdir, 'permissions_tree.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
open(os.path.join(args.outdir, 'perm_codes_all.txt'), 'w', encoding='utf-8').write('\n'.join(all_codes))
json.dump(perm_all, open(os.path.join(args.outdir, 'permissions_all_stub.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
json.dump(role_perm, open(os.path.join(args.outdir, 'role_permissions_stub.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
print(f'[+] entries={len(entries)} codes={len(all_codes)} tree_roots={len(tree)}')
cfg_path = args.config or os.path.join(args.outdir, 'config.json')
if os.path.exists(cfg_path):
cfg = json.load(open(cfg_path, encoding='utf-8'))
stubs = [s for s in cfg.get('stubs', []) if not re.search(r'permissions/all|role_permissions', s.get('match', ''))]
stubs = [
{'match': 'permissions/all', 'body': perm_all},
{'match': 'role_permissions', 'body': role_perm},
] + stubs
cfg['stubs'] = stubs
json.dump(cfg, open(cfg_path, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
print(f'[+] patched stubs -> {cfg_path}')
if __name__ == '__main__':
main()
@@ -0,0 +1,42 @@
#!/usr/bin/env python3
"""
extract_route_map.py <JSDIR> <OUTDIR>
Scan downloaded JS chunks for routeMap / routeLink style objects:
KEY:{name:"...",link:"/path"}
Writes route_map.json — used by build_perm_tree.py to resolve alias refs.
"""
import re, json, os, sys, glob
def main():
if len(sys.argv) < 3:
print("usage: extract_route_map.py <JSDIR> <OUTDIR>")
sys.exit(1)
jsdir, outdir = sys.argv[1], sys.argv[2]
os.makedirs(outdir, exist_ok=True)
best = {}
best_file = None
pat = re.compile(r'([A-Z_][A-Z0-9_]*):\{name:"([^"]*)",link:"([^"]+)"')
for fp in glob.glob(os.path.join(jsdir, '*.js')):
try:
txt = open(fp, encoding='utf-8', errors='ignore').read()
except Exception:
continue
hits = pat.findall(txt)
if len(hits) > len(best):
best = {k: {'name': n, 'link': l} for k, n, l in hits}
best_file = fp
if not best:
print('[!] no routeMap pattern found — widen regex or grep manually')
sys.exit(1)
out = os.path.join(outdir, 'route_map.json')
json.dump(best, open(out, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
print(f'[+] {len(best)} routes from {os.path.basename(best_file)} -> {out}')
if __name__ == '__main__':
main()
+207
View File
@@ -0,0 +1,207 @@
#!/usr/bin/env python3
"""
harvest_static.py <BASE_URL> <OUTDIR>
参考模板 — 非通用成品。执行前须按目标站点调整,常见改动:
- extract_endpoints 正则(endpoint 方言)
- webpack/Vite manifest 解析逻辑
- 微前端 publicPath、重试策略
Static SPA bundle harvester. Framework-agnostic; tuned for webpack + Vite.
1. Fetch entry HTML, collect script/module references.
2. Parse the runtime chunk manifest(s) and download EVERY chunk (not just the
ones in HTML), looping until no new chunk ids appear. Handles multiple
micro-frontend runtimes, each with its own publicPath.
3. Extract API endpoints and route paths from all downloaded JS.
Stdlib only. TLS verification is disabled (recon against self-signed/internal hosts).
"""
import sys, os, re, ssl, json, urllib.request, urllib.parse
from concurrent.futures import ThreadPoolExecutor
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
UA = "Mozilla/5.0 (spa-api-recon)"
def fetch(url, binary=False):
try:
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, context=CTX, timeout=25) as r:
data = r.read()
return (data if binary else data.decode("utf-8", "ignore")), r.status
except Exception as e:
code = getattr(e, "code", 0)
return "", code
def origin_of(u):
p = urllib.parse.urlparse(u)
return f"{p.scheme}://{p.netloc}"
# ---- chunk manifest parsing -------------------------------------------------
def matched_block(s, end_brace_idx):
"""Walk back from a '}' index to its matching '{' and return the object body."""
depth = 0; j = end_brace_idx
while j >= 0:
if s[j] == '}': depth += 1
elif s[j] == '{':
depth -= 1
if depth == 0: return s[j:end_brace_idx+1]
j -= 1
return ""
def parse_chunk_maps(js_text):
"""Return list of (publicPath_hint, {id: hash}) for every `}[x]+".js"` map
(webpack __webpack_require__.u) found in the text."""
maps = []
# publicPath hints in this file: .p="..." or publicPath="..."
pubs = re.findall(r'(?:\.p|publicPath)\s*=\s*"([^"]*)"', js_text)
for m in re.finditer(r'\}\[[A-Za-z_$]\]\s*\+\s*"\.js"', js_text):
body = matched_block(js_text, m.start())
pairs = re.findall(r'(\d+):"([0-9a-fA-F]{6,16})"', body)
if pairs:
maps.append((pubs, dict(pairs)))
# Vite style: __vite__mapDeps map of "assets/xx.js"
for fn in re.findall(r'"(assets/[^"]+\.js)"', js_text):
maps.append((["/"], {"__vite__": fn}))
return maps
def chunk_urls(base, js_text):
"""Yield absolute chunk URLs reconstructable from this file's manifest(s)."""
origin = origin_of(base)
out = set()
for pubs, idmap in parse_chunk_maps(js_text):
paths = pubs or ["/assets/", "/"]
for cid, h in idmap.items():
if cid == "__vite__":
fn = h # already "assets/xx.js"
for pp in paths:
out.add(urllib.parse.urljoin(origin + "/", fn))
continue
for pp in paths:
if pp.startswith("http"): bareorigin = ""; prefix = pp
else: bareorigin = origin; prefix = pp if pp.startswith("/") else "/"+pp
if not prefix.endswith("/"): prefix += "/"
out.add(f"{bareorigin}{prefix}{cid}.{h}.js")
return out
# ---- endpoint / route extraction -------------------------------------------
API_HINT = re.compile(r'/(?:api|rest|service|services|gateway|graphql|v\d|web|admin|backend|open)\b', re.I)
ASSET_EXT = re.compile(r'\.(js|css|png|jpe?g|svg|gif|woff2?|ttf|ico|map|json|mp4|webp)(\?|$)', re.I)
def extract_endpoints(js_text):
paths = set()
# quoted ('/...'), double-quoted, and backtick template paths
for m in re.findall(r'''["'`](/[A-Za-z0-9_\-./{}$:]+)["'`]''', js_text):
paths.add(m)
# concatenation heads: "/api/x/" + var
for m in re.findall(r'''["'](/[A-Za-z0-9_\-./]+/)["']\s*\+''', js_text):
paths.add(m)
api, other = set(), set()
for p in paths:
if ASSET_EXT.search(p): continue
if p.count('/') < 2 and not API_HINT.search(p): continue
(api if API_HINT.search(p) else other).add(p)
return api, other
def extract_routes(js_text):
r = set()
for key in ('path', 'to', 'redirect', 'href'):
for m in re.findall(key + r'''\s*:\s*["'](/[A-Za-z0-9_\-/:]*)["']''', js_text):
if not ASSET_EXT.search(m) and not API_HINT.search(m):
r.add(m)
return r
# ---- main -------------------------------------------------------------------
def main():
if len(sys.argv) < 3:
print("usage: harvest_static.py <BASE_URL> <OUTDIR>"); sys.exit(1)
base, outdir = sys.argv[1], sys.argv[2]
if not base.startswith("http"): base = "https://" + base
jsdir = os.path.join(outdir, "js"); os.makedirs(jsdir, exist_ok=True)
print(f"[*] entry: {base}")
html, status = fetch(base)
open(os.path.join(outdir, "index.html"), "w").write(html)
origin = origin_of(base)
# initial scripts from HTML
srcs = set(re.findall(r'<script[^>]+src="([^"]+\.js[^"]*)"', html))
srcs |= set(re.findall(r'(?:src|href)="([^"]*\.js)"', html))
seed = set()
for s in srcs:
seed.add(s if s.startswith("http") else urllib.parse.urljoin(base, s))
print(f"[*] {len(seed)} scripts referenced in HTML")
have = {} # url -> local path
def dl(url):
fn = os.path.basename(urllib.parse.urlparse(url).path)
if not fn.endswith(".js"): return None
lp = os.path.join(jsdir, fn)
if url in have: return have[url]
data, st = fetch(url, binary=True)
if st == 200 and data and not data[:15].lstrip().startswith(b"<"):
open(lp, "wb").write(data); have[url] = lp; return lp
return None
with ThreadPoolExecutor(max_workers=20) as ex:
list(ex.map(dl, seed))
# iteratively expand via chunk manifests (chunks reference more chunks)
seen_urls = set(have.keys()); frontier = list(have.values())
rounds = 0
while frontier and rounds < 6:
rounds += 1
new_urls = set()
for lp in frontier:
try: txt = open(lp, encoding="utf-8", errors="ignore").read()
except: continue
for cu in chunk_urls(base, txt):
if cu not in seen_urls: new_urls.add(cu)
seen_urls |= new_urls
if not new_urls: break
print(f"[*] round {rounds}: {len(new_urls)} new chunk urls from manifest")
before = set(have.values())
with ThreadPoolExecutor(max_workers=24) as ex:
list(ex.map(dl, new_urls))
frontier = [p for p in have.values() if p not in before]
# retry-once any manifest chunk that 404'd (transient failures are real)
all_manifest = set()
for lp in list(have.values()):
try: all_manifest |= chunk_urls(base, open(lp, encoding="utf-8", errors="ignore").read())
except: pass
missing = [u for u in all_manifest if u not in have]
if missing:
with ThreadPoolExecutor(max_workers=24) as ex:
list(ex.map(dl, missing))
still = [u for u in all_manifest if u not in have]
print(f"[*] manifest chunks: {len(all_manifest)} | downloaded {len(have)} | "
f"unreachable {len(still)} (CSS-only / undeployed)")
print(f"[+] total JS downloaded: {len(have)}")
# extract from everything
api, other, routes = set(), set(), set()
for lp in have.values():
try: txt = open(lp, encoding="utf-8", errors="ignore").read()
except: continue
a, o = extract_endpoints(txt); api |= a; other |= o
routes |= extract_routes(txt)
def dump(name, items):
path = os.path.join(outdir, name)
open(path, "w").write("\n".join(sorted(items)))
return path
dump("api_static.txt", api)
dump("paths_other.txt", other)
dump("routes.txt", routes)
dump("chunkmap.txt", sorted(os.path.basename(u) for u in all_manifest))
print(f"[+] api endpoints: {len(api)} (api_static.txt)")
print(f"[+] other paths: {len(other)} (paths_other.txt)")
print(f"[+] route paths: {len(routes)} (routes.txt)")
print(f"[+] outdir: {outdir}")
print("\n[next] reverse the 3 gate facts (see reference.md), fill config.json, "
"then: node runtime_harvest.js config.json")
if __name__ == "__main__":
main()
File diff suppressed because it is too large Load Diff
+9
View File
@@ -0,0 +1,9 @@
{
"name": "api-recon-runtime",
"version": "1.0.0",
"private": true,
"description": "Headless runtime API harvester for the api-recon skill",
"dependencies": {
"puppeteer-core": "^23.11.1"
}
}
+348
View File
@@ -0,0 +1,348 @@
/**
* api-recon coverage 模式预加载脚本 — 参考模板
*
* ⚠ 非通用成品:须按目标站点调整后再注入。
* 常见改动:loginPathRe、stubs、neutralize.fields/success、apiPattern、mockTier、forward
*
* document-start 注入(CDP addScriptToEvaluateOnNewDocument 或 userscript)
* CONFIG 字段应与 recon/config.json 保持一致
*/
(function () {
'use strict';
const CONFIG = {
loginPathRe: /\/(login|signin)(\/|$|\?)/i,
mockTier: 'L1+L2',
forward: true,
neutralize: {
fields: ['response_code', 'code', 'errno', 'ret', 'status'],
success: 0,
flags: { success: true, message: 'ok' },
},
stubs: [],
apiPattern: /\/(api|apis|v\d+|dev|internal|graphql)\//i,
apiFallbackRe: /^\/(api|apis|v\d+|dev|internal)\//i,
// API 发现增强:fetch/XHR Hook、请求头与响应录制
recordDetail: true, // 详细录制 method/url/headers/body/响应
respMax: 600,
extractUrlsFromResponse: true, // 从 JSON 响应里抠嵌套 URL
neutralizeVueRouter: true, // Vue beforeEach/push 登录跳转中和
observe: {
storageReads: false, // 观察 localStorage.getItem(辅助确认会话键名)
cookieReads: false, // 观察 document.cookie 读取
xhrHeaders: true, // 录制 XHR setRequestHeader
},
};
window.__API_RECON_PRELOAD__ = true;
window.__API_RECON_LOG__ = window.__API_RECON_LOG__ || new Set();
window.__API_RECON_DETAIL__ = window.__API_RECON_DETAIL__ || [];
window.__API_RECON_ROUTES__ = window.__API_RECON_ROUTES__ || new Set();
window.__API_RECON_OBSERVE__ = window.__API_RECON_OBSERVE__ || { storage: [], headers: [] };
function trunc(s, n) {
n = n || CONFIG.respMax || 600;
s = String(s == null ? '' : s);
return s.length > n ? s.slice(0, n) + '…' : s;
}
function extractApiUrlsFromText(text) {
if (!CONFIG.extractUrlsFromResponse || !text) return;
const re = /["'](\/(?:api|apis|v\d+|dev|internal|graphql)[^"'`\\s]*)["'`]/gi;
let m;
while ((m = re.exec(text))) {
const p = m[1].split('?')[0];
window.__API_RECON_LOG__.add('GET ' + p);
}
const broad = /\/api\/[a-zA-Z0-9_./-]+/g;
while ((m = broad.exec(text))) {
const p = m[0].split('?')[0];
if (!/\.(svg|png|jpg|gif|ico)$/i.test(p)) window.__API_RECON_LOG__.add('GET ' + p);
}
}
function recordApi(url, method) {
const p = String(url || '').split('?')[0];
if (CONFIG.apiPattern.test(p)) {
window.__API_RECON_LOG__.add((method || 'GET').toUpperCase() + ' ' + p);
}
}
function recordApiDetail(entry) {
recordApi(entry.url, entry.method);
if (!CONFIG.recordDetail) return;
window.__API_RECON_DETAIL__.push(entry);
if (entry.responseBody) extractApiUrlsFromText(entry.responseBody);
}
function blocked(url) {
return url && CONFIG.loginPathRe.test(String(url));
}
// --- 跳转中和 ---
(function neutralizeNativeNavigation() {
const rawAssign = Location.prototype.assign;
const rawReplace = Location.prototype.replace;
Location.prototype.assign = function (url) {
if (blocked(url)) return;
return rawAssign.call(this, url);
};
Location.prototype.replace = function (url) {
if (blocked(url)) return;
return rawReplace.call(this, url);
};
const rawPush = history.pushState;
const rawRep = history.replaceState;
history.pushState = function (s, t, url) {
if (blocked(url)) return;
return rawPush.apply(this, arguments);
};
history.replaceState = function (s, t, url) {
if (blocked(url)) return;
return rawRep.apply(this, arguments);
};
const hrefDesc = Object.getOwnPropertyDescriptor(Location.prototype, 'href');
if (hrefDesc && hrefDesc.set) {
const nativeSet = hrefDesc.set;
Object.defineProperty(Location.prototype, 'href', {
configurable: true,
enumerable: hrefDesc.enumerable,
get: hrefDesc.get,
set(url) {
if (blocked(url)) return;
return nativeSet.call(this, url);
},
});
}
window.close = function () {};
})();
// --- Vue Router 登录跳转中和 ---
function neutralizeVueRouter() {
if (!CONFIG.neutralizeVueRouter) return;
try {
const el = document.querySelector('[data-v-app]');
const app = (el && el.__vue_app__) || window.__VUE__;
if (!app) return;
const router = app.config && app.config.globalProperties && app.config.globalProperties.$router;
if (!router) return;
router.beforeEach(function (to, from, next) { next(); });
if (router.beforeResolve) router.beforeResolve(function (to, from, next) { next(); });
function wrapNav(fn) {
return function (loc) {
const path = typeof loc === 'string' ? loc : (loc && (loc.path || loc.fullPath)) || '';
if (blocked(path)) return Promise.resolve();
return fn.apply(this, arguments);
};
}
router.push = wrapNav(router.push.bind(router));
router.replace = wrapNav(router.replace.bind(router));
if (router.getRoutes) {
router.getRoutes().forEach(function (r) {
if (r.path) window.__API_RECON_ROUTES__.add(r.path);
});
}
} catch (e) { /* ignore */ }
}
function runPostLoadHooks() {
neutralizeVueRouter();
// 业务层跳转函数(goPage / navigateTo 等)
['goPage', 'navigateTo', 'jumpTo', 'redirectTo'].forEach(function (name) {
if (typeof window[name] !== 'function' || window[name].__apiReconWrapped) return;
const raw = window[name];
window[name] = function () {
const arg = arguments[0];
const path = typeof arg === 'string' ? arg : (arg && arg.path) || '';
if (blocked(path)) return;
return raw.apply(this, arguments);
};
window[name].__apiReconWrapped = true;
});
}
document.addEventListener('DOMContentLoaded', runPostLoadHooks);
window.addEventListener('load', runPostLoadHooks);
// --- 观察 Hook:辅助发现会话键名与请求头(可选,默认关 storage/cookie)---
if (CONFIG.observe && CONFIG.observe.storageReads) {
const rawGet = Storage.prototype.getItem;
Storage.prototype.getItem = function (key) {
window.__API_RECON_OBSERVE__.storage.push({ type: 'getItem', key: key, at: location.pathname });
return rawGet.apply(this, arguments);
};
}
if (CONFIG.observe && CONFIG.observe.cookieReads) {
const cookieDesc = Object.getOwnPropertyDescriptor(Document.prototype, 'cookie');
if (cookieDesc && cookieDesc.get) {
const nativeGet = cookieDesc.get;
Object.defineProperty(document, 'cookie', {
configurable: true,
get: function () {
window.__API_RECON_OBSERVE__.storage.push({ type: 'cookieRead', at: location.pathname });
return nativeGet.call(this);
},
set: cookieDesc.set,
});
}
}
// --- Mock 辅助 ---
const NEGATIVE_RE = /未登录|未授权|授权|not\s*login|unauthorized|forbidden/i;
const tier = CONFIG.mockTier || 'L1+L2';
function patchJsonBody(text) {
if (!tier.includes('L2')) return text;
try {
const j = JSON.parse(text);
if (j && typeof j === 'object') {
const msg = String(j.message || j.msg || '');
for (const f of CONFIG.neutralize.fields) {
if (f in j && j[f] !== CONFIG.neutralize.success && NEGATIVE_RE.test(msg)) {
j[f] = CONFIG.neutralize.success;
}
}
Object.assign(j, CONFIG.neutralize.flags || {});
if (j.data == null) j.data = {};
return JSON.stringify(j);
}
} catch (e) { /* ignore */ }
return text;
}
function lookupPrecise(url) {
if (!tier.includes('L1')) return null;
const p = String(url || '');
for (const s of CONFIG.stubs) {
if (s._re ? s._re.test(p) : new RegExp(s.match).test(p)) return s.body;
}
return null;
}
function fallbackBody(url) {
const p = String(url).split('?')[0];
if (/\/(list|search|query|page|all|tree|nodes|options)/i.test(p)) {
return { response_code: 0, data: [] };
}
if (/\/(get|detail|info|config|status)/i.test(p)) {
return { response_code: 0, data: {} };
}
return { response_code: 0, data: {}, success: true };
}
function matchMock(url) {
const precise = lookupPrecise(url);
if (precise) return precise;
if (tier.includes('L3') && CONFIG.apiFallbackRe.test(String(url || '').split('?')[0])) {
return fallbackBody(url);
}
return null;
}
CONFIG.stubs.forEach(function (s) {
if (s.match && !s._re) s._re = new RegExp(s.match);
});
// --- fetch Hook ---
const rawFetch = window.fetch;
window.fetch = async function (input, init) {
const url = typeof input === 'string' ? input : (input && input.url) || '';
const method = ((init && init.method) || 'GET').toUpperCase();
const reqHeaders = (init && init.headers) || {};
const reqBody = init && init.body ? trunc(init.body) : null;
const mock = matchMock(url);
if (mock && !CONFIG.forward) {
recordApiDetail({ method: method, url: url, reqHeaders: reqHeaders, reqBody: reqBody, status: 200, responseBody: JSON.stringify(mock), source: 'mock' });
return new Response(JSON.stringify(mock), {
status: 200,
headers: { 'Content-Type': 'application/json' },
});
}
const resp = await rawFetch.apply(this, arguments);
const clone = resp.clone();
let text = '';
try { text = await clone.text(); } catch (e) { /* ignore */ }
recordApiDetail({
method: method, url: url, reqHeaders: reqHeaders, reqBody: reqBody,
status: resp.status, responseBody: trunc(text), source: 'fetch',
});
if (!tier.includes('L2') && !mock) return resp;
const patched = patchJsonBody(text);
if (patched !== text) {
return new Response(patched, {
status: resp.status,
statusText: resp.statusText,
headers: resp.headers,
});
}
return resp;
};
// --- XHR Hook ---
const rawOpen = XMLHttpRequest.prototype.open;
const rawSend = XMLHttpRequest.prototype.send;
const rawSetHeader = XMLHttpRequest.prototype.setRequestHeader;
XMLHttpRequest.prototype.open = function (method, url) {
this.__apiReconUrl = url;
this.__apiReconMethod = method;
this.__apiReconHeaders = {};
return rawOpen.apply(this, arguments);
};
if (CONFIG.observe && CONFIG.observe.xhrHeaders) {
XMLHttpRequest.prototype.setRequestHeader = function (name, value) {
if (!this.__apiReconHeaders) this.__apiReconHeaders = {};
this.__apiReconHeaders[name] = value;
window.__API_RECON_OBSERVE__.headers.push({ name: name, url: this.__apiReconUrl });
return rawSetHeader.apply(this, arguments);
};
}
XMLHttpRequest.prototype.send = function (body) {
const xhr = this;
const url = xhr.__apiReconUrl || '';
const method = (xhr.__apiReconMethod || 'GET').toUpperCase();
const mock = matchMock(url);
if (mock && !CONFIG.forward) {
const bodyStr = JSON.stringify(mock);
recordApiDetail({ method: method, url: url, reqHeaders: xhr.__apiReconHeaders, reqBody: trunc(body), status: 200, responseBody: bodyStr, source: 'mock' });
setTimeout(function () {
Object.defineProperty(xhr, 'readyState', { configurable: true, get: function () { return 4; } });
Object.defineProperty(xhr, 'status', { configurable: true, get: function () { return 200; } });
Object.defineProperty(xhr, 'responseText', { configurable: true, get: function () { return bodyStr; } });
xhr.onreadystatechange && xhr.onreadystatechange();
xhr.onload && xhr.onload();
}, 0);
return;
}
const orig = xhr.onreadystatechange;
xhr.onreadystatechange = function () {
if (xhr.readyState === 4) {
recordApiDetail({
method: method, url: url, reqHeaders: xhr.__apiReconHeaders,
reqBody: trunc(body), status: xhr.status,
responseBody: trunc(xhr.responseText), source: 'xhr',
});
if (tier.includes('L2') && xhr.responseText) {
try {
const patched = patchJsonBody(xhr.responseText);
if (patched !== xhr.responseText) {
Object.defineProperty(xhr, 'responseText', { configurable: true, get: function () { return patched; } });
}
} catch (e) { /* ignore */ }
}
}
orig && orig.apply(this, arguments);
};
return rawSend.apply(this, arguments);
};
})();
+228
View File
@@ -0,0 +1,228 @@
#!/usr/bin/env node
/*
* runtime_harvest.js <config.json>
*
* 参考模板 — 非通用成品。执行前须按目标站点调整 config.json 及脚本内逻辑:
* cookies/localStorage、neutralize 字段、stubs 结构、loginUrlPattern、apiPattern
*
* Drives a headless browser through an authorized SPA to capture the live API
* surface (method + url + body) by defeating three client-side gates:
* 1. render gate -> inject fake auth state (cookies / localStorage)
* 2. interceptor gate-> rewrite the "unauthorized" code field to success
* 3. content gate -> stub the menu/permission endpoint with a full-feature payload
*
* Requires puppeteer-core + a system Chromium. ( cd scripts && npm install )
*
* config.json schema (all fields optional except baseUrl):
* {
* "baseUrl": "https://target/",
* "chromium": "/usr/bin/chromium", // or env CHROMIUM
* "cookies": [{"name":"auth","value":"b64json:{\"id\":1,\"username\":\"admin\"}"}],
* "localStorage": {"token":"x","isLogin":"1"},
* "neutralize": { "fields":["response_code","code","errno","ret"], "success":0,
* "flags":{"success":true,"message":"ok"} },
* "forward": true, // forward real req then rewrite code; false = offline stub
* "loginUrlPattern": "/login", // navigations matching this are suppressed as a fallback
* "stubs": [ {"match":"permission|menu|role", "body": { ...full-feature menu... }} ],
* "routes": ["/dashboard","/device", ...], // from routes.txt or the forged menu
* "apiPattern": "/api/|/rest/|/graphql", // what counts as an API call to record
* "proxy": "http://127.0.0.1:8080", // optional; also HTTP_PROXY / HTTPS_PROXY
* "waitUntil": "domcontentloaded", // prefer over networkidle2 for large SPAs
* "routeTimeout": 12000,
* "waitMs": 1200, "perRouteMs": 900, "headless": true
* }
*/
process.env.NODE_TLS_REJECT_UNAUTHORIZED = '0';
const fs = require('fs');
const path = require('path');
let puppeteer;
try { puppeteer = require('puppeteer-core'); }
catch (e) { console.error("[!] run `npm install` in the scripts/ dir first (needs puppeteer-core)"); process.exit(1); }
function loadCfg(p) {
const c = JSON.parse(fs.readFileSync(p, 'utf8'));
c.baseUrl || (() => { throw new Error("config.baseUrl required"); })();
c.chromium = c.chromium || process.env.CHROMIUM || '/usr/bin/chromium';
c.neutralize = c.neutralize || { fields: ['response_code', 'code', 'errno', 'ret', 'status'], success: 0, flags: { success: true } };
c.forward = c.forward !== false;
c.apiPattern = new RegExp(c.apiPattern || '/api/|/rest/|/service/|/graphql|/gateway/');
c.loginUrlPattern = c.loginUrlPattern || '/login';
c.routes = c.routes || ['/'];
c.waitMs = c.waitMs || 1200; c.perRouteMs = c.perRouteMs || 900;
c.headless = c.headless !== false;
c.waitUntil = c.waitUntil || 'domcontentloaded';
c.routeTimeout = c.routeTimeout || 12000;
c.captureResponses = c.captureResponses !== false; // record response body samples (forward mode)
c.recordWs = c.recordWs !== false; // record WebSocket frames + SSE endpoints
c.respMax = c.respMax || 600; // truncation length for captured bodies
c.proxy = c.proxy || process.env.HTTP_PROXY || process.env.HTTPS_PROXY || '';
(c.stubs || []).forEach(s => s._re = new RegExp(s.match));
return c;
}
function cookieValue(v) {
if (typeof v === 'string' && v.startsWith('b64json:'))
return Buffer.from(v.slice(8)).toString('base64');
if (typeof v === 'string' && v.startsWith('json:'))
return v.slice(5);
return v;
}
function neutralize(txt, n) {
try {
const j = JSON.parse(txt);
if (j && typeof j === 'object') {
for (const f of n.fields) if (f in j) j[f] = n.success;
Object.assign(j, n.flags || {});
return JSON.stringify(j);
}
} catch (e) {}
return txt;
}
(async () => {
const cfg = loadCfg(process.argv[2] || 'config.json');
const origin = new URL(cfg.baseUrl).origin;
const host = new URL(cfg.baseUrl).hostname;
const rec = []; // {m,u,b,resp,ct}
const chunks = new Set();
const ws = []; // {url,dir,data} WebSocket frames
const sse = new Set(); // SSE (text/event-stream) endpoints
if (cfg.proxy) {
process.env.HTTP_PROXY = cfg.proxy;
process.env.HTTPS_PROXY = cfg.proxy;
}
const launchArgs = ['--no-sandbox', '--disable-dev-shm-usage', '--ignore-certificate-errors'];
if (cfg.proxy) launchArgs.push(`--proxy-server=${cfg.proxy}`);
const browser = await puppeteer.launch({
executablePath: cfg.chromium, headless: cfg.headless ? 'new' : false,
args: launchArgs
});
const page = await browser.newPage();
// ---- WebSocket frame capture via CDP (fetch/XHR interception can't see WS) ----
if (cfg.recordWs) {
try {
const cdp = await page.target().createCDPSession();
await cdp.send('Network.enable');
const wsUrl = {}; // requestId -> url
cdp.on('Network.webSocketCreated', e => { wsUrl[e.requestId] = e.url; });
const onFrame = dir => e => {
const p = e.response && e.response.payloadData;
if (p != null) ws.push({ url: (wsUrl[e.requestId] || '').replace(origin, ''), dir, data: String(p).slice(0, cfg.respMax) });
};
cdp.on('Network.webSocketFrameSent', onFrame('send'));
cdp.on('Network.webSocketFrameReceived', onFrame('recv'));
} catch (e) { console.log('[ws] CDP capture unavailable:', e.message); }
}
// inject localStorage on every document
if (cfg.localStorage) {
await page.evaluateOnNewDocument((kv) => {
try { for (const k in kv) localStorage.setItem(k, kv[k]); } catch (e) {}
}, cfg.localStorage);
}
// fallback: block full-page navigations to the login url
await page.evaluateOnNewDocument((pat) => {
const bad = u => { try { return String(u).indexOf(pat) >= 0; } catch (e) { return false; } };
try {
const d = Object.getOwnPropertyDescriptor(Location.prototype, 'href');
Object.defineProperty(Location.prototype, 'href', { configurable: true,
get() { return d.get.call(this); },
set(v) { if (bad(v)) return; return d.set.call(this, v); } });
const a = Location.prototype.assign, r = Location.prototype.replace;
Location.prototype.assign = function (v) { if (bad(v)) return; return a.call(this, v); };
Location.prototype.replace = function (v) { if (bad(v)) return; return r.call(this, v); };
} catch (e) {}
}, cfg.loginUrlPattern);
await page.setRequestInterception(true);
page.on('request', async (req) => {
const u = req.url(), m = req.method();
if (/\.js(\?|$)/.test(u) && /\/assets\/|\/static\/|\/js\//.test(u)) chunks.add(u.split('/').pop());
// suppress fallback login navigations (after the app has bootstrapped)
if (req.isNavigationRequest() && req.frame() === page.mainFrame()
&& u.includes(cfg.loginUrlPattern) && rec.length > 3) {
return req.respond({ status: 204, body: '' });
}
if (!cfg.apiPattern.test(u)) return req.continue();
const entry = { m, u: u.replace(origin, '').split('?')[0], full: u, b: req.postData() ? req.postData().slice(0, 400) : null };
rec.push(entry);
// explicit stubs (menu / permission forgery) win
const stub = (cfg.stubs || []).find(s => s._re.test(u));
if (stub) return req.respond({ status: 200, contentType: 'application/json', body: JSON.stringify(stub.body) });
// SSE: forwarding a text/event-stream would hang the handler — record + short-circuit
const accept = (req.headers().accept || '');
if (/text\/event-stream/.test(accept)) { sse.add(entry.u); return req.respond({ status: 200, contentType: 'application/json', body: '{}' }); }
if (!cfg.forward) {
return req.respond({ status: 200, contentType: 'application/json',
body: neutralize('{"data":{},"list":[],"total":0}', cfg.neutralize) });
}
// forward real request, then rewrite the unauthorized code field
try {
const headers = Object.assign({}, req.headers());
if (cfg.cookies) headers.cookie = cfg.cookies.map(c => `${c.name}=${cookieValue(c.value)}`).join('; ');
const r = await fetch(u, { method: m, headers, body: (m !== 'GET' && m !== 'HEAD') ? req.postData() : undefined });
const t = await r.text();
if (cfg.captureResponses) { entry.resp = t.slice(0, cfg.respMax); entry.ct = r.headers.get('content-type') || ''; }
req.respond({ status: 200, contentType: 'application/json', body: neutralize(t, cfg.neutralize) });
} catch (e) {
req.respond({ status: 200, contentType: 'application/json', body: '{"response_code":0,"code":0,"data":{}}' });
}
});
// set auth cookies
if (cfg.cookies) {
for (const c of cfg.cookies)
await page.setCookie({ name: c.name, value: cookieValue(c.value), domain: host, path: c.path || '/' });
}
// boot
await page.goto(cfg.baseUrl, { waitUntil: cfg.waitUntil, timeout: 45000 }).catch(e => console.log('[goto]', e.message));
await new Promise(r => setTimeout(r, cfg.waitMs));
const shell = await page.evaluate(() => ({
url: location.href,
loginForm: !!document.querySelector('input[type=password]'),
links: [...document.querySelectorAll('a[href^="/"]')].map(a => a.getAttribute('href'))
}));
console.log(`[*] after boot: url=${shell.url} loginForm=${shell.loginForm}`);
if (shell.loginForm) console.log('[!] still on login — recheck render-gate facts (cookies/localStorage) in config');
// discovered menu links extend the route list
const routes = [...new Set([...cfg.routes, ...shell.links.filter(h => h && h.length > 1)])];
const perRoute = {};
for (const rt of routes) {
const before = rec.length;
try { await page.goto(origin + rt, { waitUntil: cfg.waitUntil, timeout: cfg.routeTimeout }); } catch (e) {}
await new Promise(r => setTimeout(r, cfg.perRouteMs));
const fresh = [...new Set(rec.slice(before).map(r => r.m + ' ' + r.u))];
perRoute[rt] = fresh;
}
const uniq = [...new Set(rec.map(r => r.m + ' ' + r.u))].sort();
const wsUniq = [...new Set(ws.map(f => f.url))].filter(Boolean).sort();
const outdir = path.dirname(process.argv[2] || '.');
fs.writeFileSync(path.join(outdir, 'runtime_api.json'),
JSON.stringify({ shell, uniq, perRoute, rec, ws, wsEndpoints: wsUniq, sse: [...sse] }, null, 2));
// merge with static
let merged = new Set(uniq.map(x => x.split(' ')[1]));
const staticFile = path.join(outdir, 'api_static.txt');
if (fs.existsSync(staticFile)) fs.readFileSync(staticFile, 'utf8').split('\n').filter(Boolean).forEach(p => merged.add(p));
fs.writeFileSync(path.join(outdir, 'api_merged.txt'), [...merged].sort().join('\n'));
console.log(`[+] runtime endpoints (with method): ${uniq.length}`);
console.log(`[+] chunks seen at runtime: ${chunks.size}`);
if (cfg.recordWs) console.log(`[+] websocket endpoints: ${wsUniq.length} (${ws.length} frames) | SSE endpoints: ${sse.size}`);
if (cfg.captureResponses) console.log(`[+] response bodies captured for ${rec.filter(r => r.resp != null).length}/${rec.length} calls`);
console.log(`[+] merged unique paths: ${merged.size} -> api_merged.txt`);
console.log(`[+] full detail -> runtime_api.json (per-route + bodies + ws frames + sse)`);
await browser.close();
})();
+123
View File
@@ -0,0 +1,123 @@
#!/usr/bin/env python3
"""
spider_mpa.py <BASE_URL> <OUTDIR> [--cookie "k=v; k2=v2"] [--max 200] [--depth 4]
参考模板 — 非通用成品。执行前须按目标调整 --exclude、cookie、depth/max 等同域策略。
Fallback for NON-SPA targets (traditional server-rendered MPAs: Django/Rails/PHP/
JSP, classic admin panels). When there is no JS endpoint bundle, the API surface
lives in HTML <form action>, <a href>, and inline-JS ajax urls. This BFS-crawls
the same origin and extracts:
- forms.txt : METHOD action [param1, param2, ...] (the real "endpoints")
- links.txt : every same-origin URL reached
- api_inline.txt : url-ish strings found in inline <script> / onclick (fetch/ajax)
Stdlib only. Same-origin, bounded, polite. Provide --cookie for an authed crawl.
"""
import sys, os, re, ssl, argparse, urllib.request, urllib.parse, time
from html.parser import HTMLParser
from collections import deque
CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
UA = "Mozilla/5.0 (spa-api-recon spider)"
SKIP_EXT = re.compile(r'\.(css|png|jpe?g|gif|svg|ico|woff2?|ttf|pdf|zip|mp4|webp|js)(\?|$)', re.I)
API_HINT = re.compile(r'/(?:api|rest|service|ajax|action|do|rpc|graphql|v\d|admin|backend)\b', re.I)
def fetch(url, cookie):
h = {"User-Agent": UA}
if cookie: h["Cookie"] = cookie
try:
with urllib.request.urlopen(urllib.request.Request(url, headers=h), context=CTX, timeout=20) as r:
ct = r.headers.get("Content-Type", "")
if "html" not in ct and "xml" not in ct: return "", ct
return r.read().decode("utf-8", "ignore"), ct
except Exception:
return "", ""
class Page(HTMLParser):
def __init__(self):
super().__init__()
self.links, self.scripts, self.inline = [], [], []
self.forms, self._cur = [], None
self._in_script = False
def handle_starttag(self, tag, attrs):
a = dict(attrs)
if tag == "a" and a.get("href"): self.links.append(a["href"])
elif tag == "script":
self._in_script = True
if a.get("src"): self.scripts.append(a["src"])
elif tag == "form":
self._cur = {"action": a.get("action", ""), "method": (a.get("method") or "GET").upper(), "params": []}
elif tag in ("input", "select", "textarea", "button") and self._cur is not None:
n = a.get("name")
if n: self._cur["params"].append(n)
# ajax-ish handlers
for v in a.values():
for m in re.findall(r'''["'](/[^"']{2,120})["']''', v or ""):
if API_HINT.search(m): self.inline.append(m)
def handle_endtag(self, tag):
if tag == "script": self._in_script = False
elif tag == "form" and self._cur is not None:
self.forms.append(self._cur); self._cur = None
def handle_data(self, data):
if self._in_script and data:
for m in re.findall(r'''["'`](/[A-Za-z0-9_\-./{}$:?=&]{2,120})["'`]''', data):
if API_HINT.search(m): self.inline.append(m)
def main():
ap = argparse.ArgumentParser()
ap.add_argument("base"); ap.add_argument("outdir")
ap.add_argument("--cookie", default=""); ap.add_argument("--max", type=int, default=200)
ap.add_argument("--depth", type=int, default=4); ap.add_argument("--delay", type=float, default=0.1)
ap.add_argument("--include", default=""); ap.add_argument("--exclude", default="logout|signout|delete|remove|destroy")
args = ap.parse_args()
base = args.base if args.base.startswith("http") else "https://" + args.base
os.makedirs(args.outdir, exist_ok=True)
origin = "{0.scheme}://{0.netloc}".format(urllib.parse.urlparse(base))
inc = re.compile(args.include) if args.include else None
exc = re.compile(args.exclude, re.I) if args.exclude else None
seen, forms, inline, links = set(), {}, set(), set()
q = deque([(base, 0)]); seen.add(base.split("#")[0])
n = 0
while q and n < args.max:
url, d = q.popleft(); n += 1
html, ct = fetch(url, args.cookie)
if args.delay: time.sleep(args.delay)
if not html: continue
p = Page()
try: p.feed(html)
except Exception: pass
links.add(url.replace(origin, "") or "/")
for f in p.forms:
act = urllib.parse.urljoin(url, f["action"] or url)
key = f["method"] + " " + act.replace(origin, "")
forms.setdefault(key, set()).update(f["params"])
for s in p.inline: inline.add(s)
if d < args.depth:
for href in p.links:
if href.startswith(("mailto:", "tel:", "javascript:", "#")): continue
nxt = urllib.parse.urljoin(url, href).split("#")[0]
if not nxt.startswith(origin): continue
if SKIP_EXT.search(nxt): continue
if exc and exc.search(nxt): continue
if inc and not inc.search(nxt): continue
if nxt not in seen:
seen.add(nxt); q.append((nxt, d + 1))
with open(os.path.join(args.outdir, "forms.txt"), "w") as fh:
for k in sorted(forms):
ps = ", ".join(sorted(forms[k]))
fh.write(f"{k} [{ps}]\n")
open(os.path.join(args.outdir, "links.txt"), "w").write("\n".join(sorted(links)))
open(os.path.join(args.outdir, "api_inline.txt"), "w").write("\n".join(sorted(inline)))
print(f"[+] crawled {n} pages (cap {args.max}, depth {args.depth})")
print(f"[+] forms (endpoints): {len(forms)} -> forms.txt")
print(f"[+] inline ajax urls: {len(inline)} -> api_inline.txt")
print(f"[+] pages reached: {len(links)} -> links.txt")
if not args.cookie:
print("[i] no --cookie: only public pages crawled. Pass an authed session cookie for the full surface.")
if __name__ == "__main__":
main()