-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathscrape_docs.py
More file actions
executable file
·108 lines (88 loc) · 3.91 KB
/
Copy pathscrape_docs.py
File metadata and controls
executable file
·108 lines (88 loc) · 3.91 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
#!/usr/bin/env python3
"""Scrape indeks endpoint Just One API dari dokumentasinya.
Docs pakai VitePress: daftar path lengkap ada di HTML halaman root, tapi
parameter tiap endpoint baru ada di halaman detailnya. Skrip ini:
1. Ambil semua path /zh/api/<platform>/<endpoint> dari halaman root
2. Buka tiap halaman detail -> ambil URL lengkap + nama parameter
3. Simpan endpoints.json
Hemat waktu: pakai --sample N kalau cuma mau sebagian (docs itu 341 endpoint).
"""
from __future__ import annotations
import argparse
import html
import json
import re
import sys
import time
import urllib.parse
import urllib.request
from pathlib import Path
UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36")
ROOT = "https://docs.justoneapi.com/zh/"
HERE = Path(__file__).resolve().parent
def get(url: str, timeout: int = 30) -> str:
r = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(r, timeout=timeout) as f:
return f.read().decode("utf-8", "ignore")
def teks_bersih(h: str) -> str:
h = re.sub(r"<script.*?</script>", " ", h, flags=re.S)
h = re.sub(r"<style.*?</style>", " ", h, flags=re.S)
h = re.sub(r"<[^>]+>", " ", h)
return re.sub(r"\s+", " ", html.unescape(h))
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--sample", type=int, default=0, help="batasi jumlah endpoint (0 = semua)")
ap.add_argument("--out", default=str(HERE / "endpoints.json"))
a = ap.parse_args()
print("ambil halaman root…", file=sys.stderr)
root = get(ROOT)
paths = sorted(set(re.findall(r'"(/zh/api/[a-zA-Z0-9/_.\-]+)"', root)))
print(f"{len(paths)} path ditemukan", file=sys.stderr)
if a.sample:
paths = paths[:a.sample]
hasil: dict[str, list] = {}
for i, p in enumerate(paths, 1):
plat = p.split("/zh/api/", 1)[1].split("/", 1)[0]
try:
t = teks_bersih(get("https://docs.justoneapi.com" + p))
except Exception as e:
print(f" [{i}/{len(paths)}] {p} GAGAL: {e}", file=sys.stderr)
continue
m = re.search(r"https://api\.justoneapi\.com(/api/[a-zA-Z0-9/_.\-]+)", t)
if not m: # halaman kategori, bukan endpoint
continue
url_api = m.group(1)
# Nama endpoint dari sidebar: "GET 帖子详情 (V1)" lalu path-nya
nama = ""
m2 = re.search(r"GET\s+([^A-Z/]{2,40}?)\s*\(V\s*\*?[\d.]*\)", t)
if m2:
nama = m2.group(1).strip()
# Param: pola di tabel "请求参数" -> "token query string 是 - 说明"
params = []
seg = t[t.find("请求参数"):t.find("请求参数") + 2500]
for pm in re.finditer(r"(\w+)\s+(query|body|path|header)\s+string\s+是", seg):
params.append(pm.group(1))
if not params:
for pm in re.finditer(r"(\w+)\s+(query|body|path)\s+string", seg):
params.append(pm.group(1))
metode = "POST" if re.search(r"HTTP 方法:\s*POST|POST\s*$", seg or t[t.find("接口路径"):t.find("接口路径") + 400]) else "GET"
hasil.setdefault(plat, []).append({
"path": url_api.replace("/api/", ""),
"url": "https://api.justoneapi.com" + url_api,
"nama": nama,
"method": metode,
"params": params,
"docs": "https://docs.justoneapi.com" + p,
})
if i % 20 == 0:
print(f" [{i}/{len(paths)}] …", file=sys.stderr)
time.sleep(0.15) # sopan ke server mereka
semua = sum(len(v) for v in hasil.values())
Path(a.out).write_text(json.dumps(hasil, ensure_ascii=False, indent=2))
print(f"\nSELESAI: {semua} endpoint di {len(hasil)} platform -> {a.out}", file=sys.stderr)
for plat, eps in sorted(hasil.items(), key=lambda x: -len(x[1])):
print(f" {plat:38s} {len(eps)}", file=sys.stderr)
return 0
if __name__ == "__main__":
sys.exit(main())