direct_400 = sorted({normal(i["domain"]) for i in load_json("bad_request_results.json").get("bad_request", []) if normal(i["domain"])}) indirect_400 = sorted({normal(i["domain"]) for i in load_json("probe_indirect_results.json") if i.get("kind") == "400_BAD_REQUEST"}) overlap = sorted(set(direct_400) & set(indirect_400)) merged = sorted(set(direct_400) | set(indirect_400))
defmain(): print(f"== (1) 枚举直接 fork: {OWNER_REPO} ==") forks = fetch_all_forks() trimmed = [trim(f) for f in forks] names = sorted({t["full_name"] for t in trimmed if t["full_name"]})
defdump(fn, obj): withopen(os.path.join(ROOT, fn), "w", encoding="utf-8") as f: json.dump(obj, f, ensure_ascii=False, indent=2) print(f" wrote {fn}")
dump("forks_direct_full.json", trimmed) dump("forks.json", trimmed) dump("forks_all.json", trimmed) withopen(os.path.join(ROOT, "direct_forks.txt"), "w", encoding="utf-8") as f: f.write("\n".join(names) + "\n") withopen(os.path.join(ROOT, "all_forks.txt"), "w", encoding="utf-8") as f: f.write("\n".join(names) + "\n") total_sub = sum(t["forks_count"] or0for t in trimmed) print(f"直接 fork 总数: {len(names)};子 fork 数合计: {total_sub}")
defok(r): ifnot re.fullmatch(r"[\w.\-]+/[\w.\-]+", r): returnFalse if r.lower().endswith(STATIC_EXT): returnFalse ifany(r.startswith(b) or r.lower().startswith(b) for b in BANNED_PREFIX): returnFalse returnTrue
keys = sorted({r for r in raw if ok(r) and r.lower() != ROOT_REPO.lower()})
withopen(os.path.join(ROOT, "net_repo_keys.json"), "w", encoding="utf-8") as f: json.dump(keys, f, ensure_ascii=False, indent=2) withopen(os.path.join(ROOT, "indirect_forks.txt"), "w", encoding="utf-8") as f: f.write("\n".join(keys) + "\n") withopen(os.path.join(ROOT, "indirect_forks_clean.txt"), "w", encoding="utf-8") as f: f.write("\n".join(keys) + "\n")
defget_file(repo, path): """api.github.com contents API(raw CDN 在本环境卡顿),main→master 回退""" for branch in ("main", "master"): url = f"https://api.github.com/repos/{repo}/contents/{path}?ref={branch}" for attempt inrange(3): try: with urllib.request.urlopen(urllib.request.Request(url, headers=HDRS), timeout=20) as r: return ("ok", r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: if e.code == 404: break if e.code in (403, 429): time.sleep(60) continue time.sleep(2 * attempt) except Exception: time.sleep(2 * attempt) return ("404", "")
defcheck_repo(repo): res = {"repo": repo} st1, c1 = get_file(repo, "files/index.js") # 首选 files/index.js m1 = ARGO_RE.search(c1) if st1 == "ok"elseNone res["files_idx_fallback"] = m1.group(2) if m1 elseNone st2, c2 = get_file(repo, "index.js") # 次选根 index.js m2 = ARGO_RE.search(c2) if st2 == "ok"elseNone res["root_idx_fallback"] = m2.group(2) if m2 elseNone return res
defmain(): repos = [l.strip() for l inopen(os.path.join(ROOT, "direct_forks.txt"), encoding="utf-8-sig") if l.strip()] repos = sorted(repos)[:93] print(f"== (5) 初轮 ARGO 检测: {len(repos)} 仓库 ==") out = [] with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(check_repo, r): r for r in repos} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): out.append(fut.result()) if i % 20 == 0: print(f" {i}/{len(repos)}") out.sort(key=lambda r: r["repo"]) withopen(os.path.join(ROOT, "argo_results.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) ff = [r for r in out if r["files_idx_fallback"]] rf = [r for r in out if r["root_idx_fallback"]] nf = [r for r in out ifnot r["files_idx_fallback"] andnot r["root_idx_fallback"]] print(f"files/index.js 非空 fallback: {len(ff)};根 index.js 非空: {len(rf)};空/缺失: {len(nf)}")
defget_file(repo, path): """api.github.com contents API,main→master 回退,重试 3 次""" for branch in ("main", "master"): url = f"https://api.github.com/repos/{repo}/contents/{path}?ref={branch}" for attempt inrange(3): try: with urllib.request.urlopen(urllib.request.Request(url, headers=HDRS), timeout=20) as r: rem = r.headers.get("X-RateLimit-Remaining") if rem == "0": raise RateLimited("remaining=0") return ("ok", r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: if e.code == 404: break# 下一 branch if e.code in (403, 429): reset = e.headers.get("X-RateLimit-Reset") wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit] sleep {wait}s") time.sleep(wait) continue time.sleep(2 * attempt) except RateLimited: reset = None try: with urllib.request.urlopen(urllib.request.Request( "https://api.github.com/rate_limit", headers=HDRS), timeout=15) as r: reset = json.loads(r.read())["resources"]["core"]["reset"] except Exception: pass wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit remaining=0] sleep {wait}s") time.sleep(wait) except Exception: time.sleep(2 * attempt) return ("404", "")
defcheck_repo(repo): res = {"repo": repo} st1, c1 = get_file(repo, "files/index.js") m1 = ARGO_RE.search(c1) if st1 == "ok"elseNone res["files_idx_fallback"] = m1.group(2) if m1 elseNone st2, c2 = get_file(repo, "index.js") m2 = ARGO_RE.search(c2) if st2 == "ok"elseNone res["root_idx_fallback"] = m2.group(2) if m2 elseNone res["_cache"] = {} if st1 == "ok": res["_cache"]["files/index.js"] = c1 if st2 == "ok": res["_cache"]["index.js"] = c2 return res
defvalid_domain(d): ifnot d: returnFalse d = d.strip().rstrip("/").strip() if d in JUNK or d == DEFAULT_DOMAIN: returnFalse if"ip6.arpa"in d or d == "free.hr"or d.endswith(".free.hr") or d == "cho.zzx.free": returnFalse returnTrue
defmain(): data = json.load(open(os.path.join(ROOT, "forks_all.json"), encoding="utf-8-sig")) repos = sorted({d["full_name"] for d in data if d.get("full_name")}) print(f"== (6) 全部 fork ARGO 检测: {len(repos)} 仓库 ==") out = [] with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(check_repo, r): r for r in repos} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): out.append(fut.result()) if i % 50 == 0: print(f" {i}/{len(repos)}") out.sort(key=lambda r: r["repo"]) # 拆缓存:argo_all_results 只留元数据,文件内容进 file_content_cache.json cache = {} for r in out: c = r.pop("_cache", {}) if c: cache[r["repo"]] = c withopen(os.path.join(ROOT, "argo_all_results.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) withopen(os.path.join(ROOT, "file_content_cache.json"), "w", encoding="utf-8") as f: json.dump(cache, f, ensure_ascii=False)
doms = set() for r in out: for k in ("files_idx_fallback", "root_idx_fallback"): v = r.get(k) if valid_domain(v): doms.add(v.strip().rstrip("/").strip()) withopen(os.path.join(ROOT, "custom_domains.txt"), "w", encoding="utf-8") as f: f.write("\n".join(sorted(doms)) + "\n")
ff = sum(1for r in out if r["files_idx_fallback"]) rf = sum(1for r in out if r["root_idx_fallback"]) print(f"完成:files 非空 {ff} / root 非空 {rf} / custom_domains.txt {len(doms)} 域名")
defget_file(repo, path): """api.github.com contents API,main→master 回退,重试 3 次""" for branch in ("main", "master"): url = f"https://api.github.com/repos/{repo}/contents/{path}?ref={branch}" for attempt inrange(3): try: with urllib.request.urlopen(urllib.request.Request(url, headers=HDRS), timeout=20) as r: rem = r.headers.get("X-RateLimit-Remaining") if rem == "0": raise RateLimited("remaining=0") return ("ok", r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: if e.code == 404: break if e.code in (403, 429): reset = e.headers.get("X-RateLimit-Reset") wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit] sleep {wait}s") time.sleep(wait) continue time.sleep(2 * attempt) except RateLimited: reset = None try: with urllib.request.urlopen(urllib.request.Request( "https://api.github.com/rate_limit", headers=HDRS), timeout=15) as r: reset = json.loads(r.read())["resources"]["core"]["reset"] except Exception: pass wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit remaining=0] sleep {wait}s") time.sleep(wait) except Exception: time.sleep(2 * attempt) return ("404", "")
classRateLimited(Exception): pass
defcheck_repo(repo): res = {"repo": repo} st1, c1 = get_file(repo, "files/index.js") m1 = ARGO_RE.search(c1) if st1 == "ok"elseNone res["files_index"] = m1.group(2) if m1 elseNone st2, c2 = get_file(repo, "index.js") m2 = ARGO_RE.search(c2) if st2 == "ok"elseNone res["root_index"] = m2.group(2) if m2 elseNone res["_cache"] = {} if st1 == "ok": res["_cache"]["files/index.js"] = c1 if st2 == "ok": res["_cache"]["index.js"] = c2 return res
defmain(): repos = set() p1 = os.path.join(ROOT, "indirect_forks_clean.txt") p2 = os.path.join(ROOT, "net_repo_keys.json") if os.path.exists(p1): repos |= {l.strip() for l inopen(p1, encoding="utf-8-sig") if l.strip()} if os.path.exists(p2): repos |= set(json.load(open(p2, encoding="utf-8-sig"))) repos = sorted(r for r in repos if r and r.lower() != ROOT_REPO.lower()) print(f"== (7) 间接 fork ARGO 检测: {len(repos)} 仓库 ==") out = [] with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(check_repo, r): r for r in repos} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): out.append(fut.result()) if i % 50 == 0: print(f" {i}/{len(repos)}") out.sort(key=lambda r: r["repo"]) cache_path = os.path.join(ROOT, "file_content_cache.json") cache = json.load(open(cache_path, encoding="utf-8-sig")) if os.path.exists(cache_path) else {} for r in out: c = r.pop("_cache", {}) if c: cache.setdefault(r["repo"], {}).update(c) withopen(os.path.join(ROOT, "argo_indirect_results.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) withopen(cache_path, "w", encoding="utf-8") as f: json.dump(cache, f, ensure_ascii=False) fi = sum(1for r in out if r["files_index"]) ri = sum(1for r in out if r["root_index"]) print(f"完成:files_index 非空 {fi} / root_index 非空 {ri}")
defapi_get(url): req = urllib.request.Request(url, headers=HDRS) try: with urllib.request.urlopen(req, timeout=30) as r: rem = r.headers.get("X-RateLimit-Remaining") body = r.read().decode("utf-8", "replace") if rem == "0": raise RateLimited("remaining=0") return json.loads(body) except urllib.error.HTTPError as e: if e.code in (403, 429): raise RateLimited(f"http{e.code}") if e.code == 404: returnNone raise
defload_state(): sp = os.path.join(ROOT, "sub_forks_state.json") if os.path.exists(sp): return json.load(open(sp, encoding="utf-8-sig")) seeds = [] p = os.path.join(ROOT, "indirect_forks_clean.txt") if os.path.exists(p): seeds = [l.strip() for l inopen(p, encoding="utf-8-sig") if l.strip()] return {"visited": [], "queue": seeds, "results": []}
defsave_state(st): withopen(os.path.join(ROOT, "sub_forks_state.json"), "w", encoding="utf-8") as f: json.dump(st, f, ensure_ascii=False) withopen(os.path.join(ROOT, "sub_forks_full.json"), "w", encoding="utf-8") as f: json.dump(st["results"], f, ensure_ascii=False, indent=2)
deffetch_children(repo): children = [] page = 1 whileTrue: data = api_get(f"https://api.github.com/repos/{repo}/forks?per_page=100&page={page}") ifnot data: break children.extend(data) iflen(data) < 100: break page += 1 return children
defmain(): st = load_state() visited = set(st["visited"]) queue = deque(r for r in st["queue"] if r notin visited) results = st["results"] have = {r["full_name"] for r in results} print(f"== (8) BFS 子 fork: 已访问 {len(visited)},队列 {len(queue)},已有结果 {len(results)} ==")
count = 0 while queue: repo = queue.popleft() if repo in visited: continue try: children = fetch_children(repo) except RateLimited as e: queue.appendleft(repo) st.update(visited=sorted(visited), queue=list(queue), results=results) save_state(st) print(f"[rate-limit {e}] 任务重新入队,状态已保存,退出。剩余队列 {len(queue)}") return except Exception as e: print(f" [err] {repo}: {e!r}(跳过)") visited.add(repo) continue
visited.add(repo) for c in children: fn = c.get("full_name") ifnot fn or fn in have or fn == repo: continue have.add(fn) results.append({"full_name": fn, "parent": repo, "forks_count": c.get("forks_count"), "created_at": c.get("created_at"), "pushed_at": c.get("pushed_at"), "default_branch": c.get("default_branch")}) queue.append(fn) count += 1 if count % 25 == 0: st.update(visited=sorted(visited), queue=list(queue), results=results) save_state(st) print(f" visited={len(visited)} queue={len(queue)} results={len(results)}") time.sleep(0.15)
# 收尾:forks_all.json = 直接 + 间接 + 子 fork 并集 allnames = set() p = os.path.join(ROOT, "forks_direct_full.json") if os.path.exists(p): allnames |= {d["full_name"] for d in json.load(open(p, encoding="utf-8-sig")) if d.get("full_name")} allnames |= {r["full_name"] for r in results} p = os.path.join(ROOT, "indirect_forks_clean.txt") if os.path.exists(p): allnames |= {l.strip() for l inopen(p, encoding="utf-8-sig") if l.strip()} merged = [{"full_name": n, "source": "subfork_union"} for n insorted(allnames)] withopen(os.path.join(ROOT, "forks_all.json"), "w", encoding="utf-8") as f: json.dump(merged, f, ensure_ascii=False, indent=2) withopen(os.path.join(ROOT, "all_forks.txt"), "w", encoding="utf-8") as f: f.write("\n".join(sorted(allnames)) + "\n") print(f"forks_all.json / all_forks.txt: {len(allnames)} 全量 fork")
defget_file(repo, path): for branch in ("main", "master"): url = f"https://api.github.com/repos/{repo}/contents/{path}?ref={branch}" for attempt inrange(3): try: with urllib.request.urlopen(urllib.request.Request(url, headers=HDRS), timeout=20) as r: return ("ok", r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: if e.code == 404: break if e.code in (403, 429): time.sleep(60) continue time.sleep(2 * attempt) except Exception: time.sleep(2 * attempt) return ("404", "")
defcheck_repo(repo): res = {"repo": repo} st1, c1 = get_file(repo, "files/index.js") m1 = ARGO_RE.search(c1) if st1 == "ok"elseNone res["files_index"] = m1.group(2) if m1 elseNone st2, c2 = get_file(repo, "index.js") m2 = ARGO_RE.search(c2) if st2 == "ok"elseNone res["root_index"] = m2.group(2) if m2 elseNone res["_cache"] = {} if st1 == "ok": res["_cache"]["files/index.js"] = c1 if st2 == "ok": res["_cache"]["index.js"] = c2 return res
defmain(): sub = json.load(open(os.path.join(ROOT, "sub_forks_full.json"), encoding="utf-8-sig")) ind = json.load(open(os.path.join(ROOT, "argo_indirect_results.json"), encoding="utf-8-sig")) covered = {r["repo"] for r in ind} todo = sorted({r["full_name"] for r in sub} - covered) print(f"子 fork 补检: {len(todo)} 仓库") ifnot todo: return new = [] with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(check_repo, r): r for r in todo} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): new.append(fut.result()) if i % 20 == 0: print(f" {i}/{len(todo)}") cache_path = os.path.join(ROOT, "file_content_cache.json") cache = json.load(open(cache_path, encoding="utf-8-sig")) if os.path.exists(cache_path) else {} for r in new: c = r.pop("_cache", {}) if c: cache.setdefault(r["repo"], {}).update(c) merged = sorted(ind + new, key=lambda r: r["repo"]) withopen(os.path.join(ROOT, "argo_indirect_results.json"), "w", encoding="utf-8") as f: json.dump(merged, f, ensure_ascii=False, indent=2) withopen(cache_path, "w", encoding="utf-8") as f: json.dump(cache, f, ensure_ascii=False) fi = sum(1for r in new if r["files_index"]) ri = sum(1for r in new if r["root_index"]) print(f"补检完成:新增 {len(new)}(files 非空 {fi} / root 非空 {ri})," f"argo_indirect 总数 {len(merged)},缓存 {len(cache)}")
defprobe(domain): res = {"domain": domain, "scheme": None, "status": None, "kind": None, "snippet": ""} last = "" for scheme in ("https", "http"): try: r = requests.get(f"{scheme}://{domain}", headers=HEADERS, timeout=15, verify=False, allow_redirects=True) text = r.text[:3000] res.update(scheme=scheme, status=r.status_code) res["kind"] = "bad_request"if (r.status_code == 400or"Bad Request"in text) else"other" res["snippet"] = text[:200] return res except Exception as e: last = repr(e)[:200] res["kind"] = "other" res["snippet"] = last or"unreachable" return res
defmain(): domains = sorted({normal(l) for l inopen(os.path.join(ROOT, "custom_domains.txt"), encoding="utf-8-sig") if normal(l)}) print(f"== (9) Bad Request 探测: {len(domains)} 域名 ==") results = [] with concurrent.futures.ThreadPoolExecutor(max_workers=10) as ex: futs = {ex.submit(probe, d): d for d in domains} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): results.append(fut.result()) if i % 50 == 0: print(f" {i}/{len(domains)}") results.sort(key=lambda r: r["domain"]) bad = [r for r in results if r["kind"] == "bad_request"] other = [r for r in results if r["kind"] != "bad_request"] withopen(os.path.join(ROOT, "bad_request_results.json"), "w", encoding="utf-8") as f: json.dump({"bad_request": bad, "other": other}, f, ensure_ascii=False, indent=2) print(f"bad_request: {len(bad)};other: {len(other)}") for r in bad: print(f" [400] {r['domain']} (scheme={r['scheme']}, status={r['status']})")
defcollect_domains(): data = json.load(open(os.path.join(ROOT, "argo_indirect_results.json"), encoding="utf-8-sig")) d2repos = {} for r in data: for k in ("files_index", "root_index"): v = normal(r.get(k)) ifnot v or v in JUNK or v == DEFAULT_DOMAIN: continue if"ip6.arpa"in v or v == "free.hr"or v.endswith(".free.hr") or v == "cho.zzx.free": continue d2repos.setdefault(v, set()).add(r["repo"]) return {d: sorted(rs) for d, rs insorted(d2repos.items())}
defclassify(domain): res = {"domain": domain, "scheme": None, "status": None, "kind": None, "snippet": ""} last = "" for scheme in ("https", "http"): try: r = requests.get(f"{scheme}://{domain}", headers=HEADERS, timeout=15, verify=False, allow_redirects=True) status, text = r.status_code, r.text[:3000] res.update(scheme=scheme, status=status) if status == 400or"Bad Request"in text: res["kind"] = "400_BAD_REQUEST" elif status == 502: res["kind"] = "502_BACKEND_DOWN" elif status == 530and"1033"in text: res["kind"] = "530_TUNNEL_DOWN" elif status == 530: res["kind"] = "530_OTHER" elif status == 200: res["kind"] = "200_OK" else: res["kind"] = f"HTTP_{status}" res["snippet"] = text[:300] return res except requests.exceptions.SSLError: last = "ssl error" continue except requests.exceptions.ConnectionError as e: s = repr(e) if"getaddrinfo"in s or"Name or service not known"in s or"nodename nor servname"in s: res["kind"] = "DNS_FAIL" res["snippet"] = "getaddrinfo failed" else: res["kind"] = "CONN_FAIL" res["snippet"] = s[:200] res["scheme"] = scheme return res except Exception as e: res.update(scheme=scheme, status="ERR", kind="ERROR", snippet=repr(e)[:200]) return res res["kind"] = "UNREACHABLE" res["snippet"] = last return res
defmain(): d2repos = collect_domains() domains = list(d2repos) print(f"== (10) 间接域名探测: {len(domains)} 唯一域名 ==") results = [] with concurrent.futures.ThreadPoolExecutor(max_workers=10) as ex: futs = {ex.submit(classify, d): d for d in domains} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): r = fut.result() r["repos"] = d2repos[r["domain"]] results.append(r) if i % 20 == 0: print(f" {i}/{len(domains)}") results.sort(key=lambda r: r["domain"]) withopen(os.path.join(ROOT, "probe_indirect_results.json"), "w", encoding="utf-8") as f: json.dump(results, f, ensure_ascii=False, indent=2) from collections import Counter cnt = Counter(r["kind"] for r in results) print("分类统计:", dict(cnt)) b400 = [r["domain"] for r in results if r["kind"] == "400_BAD_REQUEST"] print(f"400_BAD_REQUEST: {len(b400)}") for d in b400: print(f" [400] {d}")
defvalid_domain(d): v = normal(d) ifnot v or v in JUNK or v == DEFAULT_DOMAIN: returnFalse if"ip6.arpa"in v or v == "free.hr"or v.endswith(".free.hr") or v == "cho.zzx.free": returnFalse returnTrue
defcandidate_repos(): repos = {} p1 = os.path.join(ROOT, "argo_all_results.json") p2 = os.path.join(ROOT, "argo_indirect_results.json") if os.path.exists(p1): for r in json.load(open(p1, encoding="utf-8-sig")): dom = r.get("files_idx_fallback") or r.get("root_idx_fallback") if valid_domain(dom): repos[r["repo"]] = normal(dom) if os.path.exists(p2): for r in json.load(open(p2, encoding="utf-8-sig")): dom = r.get("files_index") or r.get("root_index") if valid_domain(dom): repos.setdefault(r["repo"], normal(dom)) return repos
defmain(): repos = candidate_repos() cache_path = os.path.join(ROOT, "file_content_cache.json") cache = json.load(open(cache_path, encoding="utf-8-sig")) if os.path.exists(cache_path) else {} print(f"== (11) UUID 提取: {len(repos)} 候选仓库(缓存命中 {sum(1for r in repos if r in cache)})==")
out = {}
deffrom_cache(repo): for path in ("files/index.js", "index.js"): c = cache.get(repo, {}).get(path) if c: m = UUID_RE.search(c) if m: return m.group(1) returnNone
deffetch_uuid(repo): for path in ("files/index.js", "index.js"): st, c = get_file(repo, path) if st == "ok": m = UUID_RE.search(c) if m: return m.group(1) elif st != "404": returnf"error: {st}" return"NOT_FOUND"
todo = [] for repo in repos: u = from_cache(repo) if u: out[repo] = u else: todo.append(repo)
if todo: print(f" 缓存未命中 {len(todo)},API 补抓 ...") with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(fetch_uuid, r): r for r in todo} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): out[futs[fut]] = fut.result() if i % 20 == 0: print(f" {i}/{len(todo)}")
withopen(os.path.join(ROOT, "uuid_results.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) ok = sum(1for v in out.values() if re.fullmatch(r"[0-9a-fA-F-]{36}", v or"")) print(f"完成:提取到 UUID {ok} / {len(out)}")
defvalid_indirect_domain(d): v = normal(d or"") ifnot v or v in JUNK or v == DEFAULT_DOMAIN: returnFalse if"ip6.arpa"in v or v == "free.hr"or v.endswith(".free.hr") or v == "cho.zzx.free": returnFalse returnTrue
defmain(): # 1) 直接 400 direct_400 = sorted({normal(i["domain"]) for i in load_json("bad_request_results.json").get("bad_request", []) if normal(i["domain"])}) # 2) 间接 400 indirect_400 = sorted({normal(i["domain"]) for i in load_json("probe_indirect_results.json") if i.get("kind") == "400_BAD_REQUEST"}) # 3) 合并 overlap = sorted(set(direct_400) & set(indirect_400)) merged = sorted(set(direct_400) | set(indirect_400)) out = {"total": len(merged), "direct_count": len(direct_400), "indirect_count": len(indirect_400), "overlap": overlap, "domains": merged} withopen(os.path.join(ROOT, "bad_request_all.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) print(f"bad_request_all.json: total={len(merged)} direct={len(direct_400)} " f"indirect={len(indirect_400)} overlap={overlap}")
# 4) 归并间接自定义域名 cp = os.path.join(ROOT, "custom_domains.txt") cur = {normal(l) for l inopen(cp, encoding="utf-8-sig")} if os.path.exists(cp) elseset() cur.discard("") ind = set() for r in load_json("argo_indirect_results.json"): for k in ("files_index", "root_index"): v = r.get(k) if valid_indirect_domain(v): ind.add(normal(v)) new = sorted(cur | ind) withopen(cp, "w", encoding="utf-8") as f: f.write("\n".join(new) + "\n") print(f"custom_domains.txt: 原 {len(cur)} -> 归并后 {len(new)}(间接新增 {len(set(new) - cur)})")
defis_bad_domain(d): v = normal(d) ifnot v or v in BAD_DOMAINS: returnTrue if"ip6.arpa"in v: returnTrue if v == "free.hr"or v.endswith(".free.hr"): returnTrue returnFalse
defpick_domain(r): for k in ("files_idx_fallback", "files_index", "root_idx_fallback", "root_index"): v = normal(r.get(k)) if v andnot is_bad_domain(v): return v returnNone
defdump(fn, obj): withopen(os.path.join(ROOT, fn), "w", encoding="utf-8") as f: json.dump(obj, f, ensure_ascii=False, indent=2) print(f" wrote {fn} ({len(obj)} 条)")
defmain(): uuids = load("uuid_results.json") records = {} for src in ("argo_all_results.json", "argo_indirect_results.json"): for r in load(src): repo = r["repo"] dom = pick_domain(r) uu = uuids.get(repo) if dom and uu and UUID_OK.match(uu): records[repo] = {"domain": dom, "uuid": uu}
repo_domain_uuid = dict(sorted(records.items())) repo_domains_map = {k: v["domain"] for k, v in repo_domain_uuid.items()} rel_repo_uuid = {k: v["uuid"] for k, v in repo_domain_uuid.items()}
d2repos, d2uuid = {}, {} for repo, v in repo_domain_uuid.items(): d2repos.setdefault(v["domain"], []).append(repo) d2uuid[v["domain"]] = v["uuid"] rel_domain_repo = {d: sorted(rs) for d, rs insorted(d2repos.items())} rel_domain_uuid = dict(sorted(d2uuid.items()))
defmain(): print(f"== (1) 枚举直接 fork: {OWNER_REPO} ==") forks = fetch_all_forks() trimmed = [trim(f) for f in forks] names = sorted({t["full_name"] for t in trimmed if t["full_name"]})
defdump(fn, obj): withopen(os.path.join(ROOT, fn), "w", encoding="utf-8") as f: json.dump(obj, f, ensure_ascii=False, indent=2) print(f" wrote {fn}")
dump("forks_direct_full.json", trimmed) dump("forks.json", trimmed) dump("forks_all.json", trimmed) withopen(os.path.join(ROOT, "direct_forks.txt"), "w", encoding="utf-8") as f: f.write("\n".join(names) + "\n") withopen(os.path.join(ROOT, "all_forks.txt"), "w", encoding="utf-8") as f: f.write("\n".join(names) + "\n") total_sub = sum(t["forks_count"] or0for t in trimmed) print(f"直接 fork 总数: {len(names)};子 fork 数合计: {total_sub}")
defok(r): ifnot re.fullmatch(r"[\w.\-]+/[\w.\-]+", r): returnFalse if r.lower().endswith(STATIC_EXT): returnFalse ifany(r.startswith(b) or r.lower().startswith(b) for b in BANNED_PREFIX): returnFalse returnTrue
keys = sorted({r for r in raw if ok(r) and r.lower() != ROOT_REPO.lower()})
withopen(os.path.join(ROOT, "net_repo_keys.json"), "w", encoding="utf-8") as f: json.dump(keys, f, ensure_ascii=False, indent=2) withopen(os.path.join(ROOT, "indirect_forks.txt"), "w", encoding="utf-8") as f: f.write("\n".join(keys) + "\n") withopen(os.path.join(ROOT, "indirect_forks_clean.txt"), "w", encoding="utf-8") as f: f.write("\n".join(keys) + "\n")
defget_file(repo, path): """api.github.com contents API(raw CDN 在本环境卡顿),main→master 回退""" for branch in ("main", "master"): url = f"https://api.github.com/repos/{repo}/contents/{path}?ref={branch}" for attempt inrange(3): try: with urllib.request.urlopen(urllib.request.Request(url, headers=HDRS), timeout=20) as r: return ("ok", r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: if e.code == 404: break if e.code in (403, 429): time.sleep(60) continue time.sleep(2 * attempt) except Exception: time.sleep(2 * attempt) return ("404", "")
defcheck_repo(repo): res = {"repo": repo} st1, c1 = get_file(repo, "files/index.js") # 首选 files/index.js m1 = ARGO_RE.search(c1) if st1 == "ok"elseNone res["files_idx_fallback"] = m1.group(2) if m1 elseNone st2, c2 = get_file(repo, "index.js") # 次选根 index.js m2 = ARGO_RE.search(c2) if st2 == "ok"elseNone res["root_idx_fallback"] = m2.group(2) if m2 elseNone return res
defmain(): repos = [l.strip() for l inopen(os.path.join(ROOT, "direct_forks.txt"), encoding="utf-8-sig") if l.strip()] repos = sorted(repos)[:93] print(f"== (5) 初轮 ARGO 检测: {len(repos)} 仓库 ==") out = [] with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(check_repo, r): r for r in repos} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): out.append(fut.result()) if i % 20 == 0: print(f" {i}/{len(repos)}") out.sort(key=lambda r: r["repo"]) withopen(os.path.join(ROOT, "argo_results.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) ff = [r for r in out if r["files_idx_fallback"]] rf = [r for r in out if r["root_idx_fallback"]] nf = [r for r in out ifnot r["files_idx_fallback"] andnot r["root_idx_fallback"]] print(f"files/index.js 非空 fallback: {len(ff)};根 index.js 非空: {len(rf)};空/缺失: {len(nf)}")
defget_file(repo, path): """api.github.com contents API,main→master 回退,重试 3 次""" for branch in ("main", "master"): url = f"https://api.github.com/repos/{repo}/contents/{path}?ref={branch}" for attempt inrange(3): try: with urllib.request.urlopen(urllib.request.Request(url, headers=HDRS), timeout=20) as r: rem = r.headers.get("X-RateLimit-Remaining") if rem == "0": raise RateLimited("remaining=0") return ("ok", r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: if e.code == 404: break# 下一 branch if e.code in (403, 429): reset = e.headers.get("X-RateLimit-Reset") wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit] sleep {wait}s") time.sleep(wait) continue time.sleep(2 * attempt) except RateLimited: reset = None try: with urllib.request.urlopen(urllib.request.Request( "https://api.github.com/rate_limit", headers=HDRS), timeout=15) as r: reset = json.loads(r.read())["resources"]["core"]["reset"] except Exception: pass wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit remaining=0] sleep {wait}s") time.sleep(wait) except Exception: time.sleep(2 * attempt) return ("404", "")
defcheck_repo(repo): res = {"repo": repo} st1, c1 = get_file(repo, "files/index.js") m1 = ARGO_RE.search(c1) if st1 == "ok"elseNone res["files_idx_fallback"] = m1.group(2) if m1 elseNone st2, c2 = get_file(repo, "index.js") m2 = ARGO_RE.search(c2) if st2 == "ok"elseNone res["root_idx_fallback"] = m2.group(2) if m2 elseNone res["_cache"] = {} if st1 == "ok": res["_cache"]["files/index.js"] = c1 if st2 == "ok": res["_cache"]["index.js"] = c2 return res
defvalid_domain(d): ifnot d: returnFalse d = d.strip().rstrip("/").strip() if d in JUNK or d == DEFAULT_DOMAIN: returnFalse if"ip6.arpa"in d or d == "free.hr"or d.endswith(".free.hr") or d == "cho.zzx.free": returnFalse returnTrue
defmain(): data = json.load(open(os.path.join(ROOT, "forks_all.json"), encoding="utf-8-sig")) repos = sorted({d["full_name"] for d in data if d.get("full_name")}) print(f"== (6) 全部 fork ARGO 检测: {len(repos)} 仓库 ==") out = [] with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(check_repo, r): r for r in repos} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): out.append(fut.result()) if i % 50 == 0: print(f" {i}/{len(repos)}") out.sort(key=lambda r: r["repo"]) # 拆缓存:argo_all_results 只留元数据,文件内容进 file_content_cache.json cache = {} for r in out: c = r.pop("_cache", {}) if c: cache[r["repo"]] = c withopen(os.path.join(ROOT, "argo_all_results.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) withopen(os.path.join(ROOT, "file_content_cache.json"), "w", encoding="utf-8") as f: json.dump(cache, f, ensure_ascii=False)
doms = set() for r in out: for k in ("files_idx_fallback", "root_idx_fallback"): v = r.get(k) if valid_domain(v): doms.add(v.strip().rstrip("/").strip()) withopen(os.path.join(ROOT, "custom_domains.txt"), "w", encoding="utf-8") as f: f.write("\n".join(sorted(doms)) + "\n")
ff = sum(1for r in out if r["files_idx_fallback"]) rf = sum(1for r in out if r["root_idx_fallback"]) print(f"完成:files 非空 {ff} / root 非空 {rf} / custom_domains.txt {len(doms)} 域名")
defget_file(repo, path): """api.github.com contents API,main→master 回退,重试 3 次""" for branch in ("main", "master"): url = f"https://api.github.com/repos/{repo}/contents/{path}?ref={branch}" for attempt inrange(3): try: with urllib.request.urlopen(urllib.request.Request(url, headers=HDRS), timeout=20) as r: rem = r.headers.get("X-RateLimit-Remaining") if rem == "0": raise RateLimited("remaining=0") return ("ok", r.read().decode("utf-8", "replace")) except urllib.error.HTTPError as e: if e.code == 404: break if e.code in (403, 429): reset = e.headers.get("X-RateLimit-Reset") wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit] sleep {wait}s") time.sleep(wait) continue time.sleep(2 * attempt) except RateLimited: reset = None try: with urllib.request.urlopen(urllib.request.Request( "https://api.github.com/rate_limit", headers=HDRS), timeout=15) as r: reset = json.loads(r.read())["resources"]["core"]["reset"] except Exception: pass wait = min(600, max(20, int(reset) - int(time.time()) + 5)) if reset else120 print(f" [rate-limit remaining=0] sleep {wait}s") time.sleep(wait) except Exception: time.sleep(2 * attempt) return ("404", "")
classRateLimited(Exception): pass
defcheck_repo(repo): res = {"repo": repo} st1, c1 = get_file(repo, "files/index.js") m1 = ARGO_RE.search(c1) if st1 == "ok"elseNone res["files_index"] = m1.group(2) if m1 elseNone st2, c2 = get_file(repo, "index.js") m2 = ARGO_RE.search(c2) if st2 == "ok"elseNone res["root_index"] = m2.group(2) if m2 elseNone res["_cache"] = {} if st1 == "ok": res["_cache"]["files/index.js"] = c1 if st2 == "ok": res["_cache"]["index.js"] = c2 return res
defmain(): repos = set() p1 = os.path.join(ROOT, "indirect_forks_clean.txt") p2 = os.path.join(ROOT, "net_repo_keys.json") if os.path.exists(p1): repos |= {l.strip() for l inopen(p1, encoding="utf-8-sig") if l.strip()} if os.path.exists(p2): repos |= set(json.load(open(p2, encoding="utf-8-sig"))) repos = sorted(r for r in repos if r and r.lower() != ROOT_REPO.lower()) print(f"== (7) 间接 fork ARGO 检测: {len(repos)} 仓库 ==") out = [] with concurrent.futures.ThreadPoolExecutor(max_workers=12) as ex: futs = {ex.submit(check_repo, r): r for r in repos} for i, fut inenumerate(concurrent.futures.as_completed(futs), 1): out.append(fut.result()) if i % 50 == 0: print(f" {i}/{len(repos)}") out.sort(key=lambda r: r["repo"]) cache_path = os.path.join(ROOT, "file_content_cache.json") cache = json.load(open(cache_path, encoding="utf-8-sig")) if os.path.exists(cache_path) else {} for r in out: c = r.pop("_cache", {}) if c: cache.setdefault(r["repo"], {}).update(c) withopen(os.path.join(ROOT, "argo_indirect_results.json"), "w", encoding="utf-8") as f: json.dump(out, f, ensure_ascii=False, indent=2) withopen(cache_path, "w", encoding="utf-8") as f: json.dump(cache, f, ensure_ascii=False) fi = sum(1for r in out if r["files_index"]) ri = sum(1for r in out if r["root_index"]) print(f"完成:files_index 非空 {fi} / root_index 非空 {ri}")