Files
ctbjrj/tools/preheat_prices.py
T

222 lines
8.2 KiB
Python

# -*- coding: utf-8 -*-
"""全量价格在线预热:遍历全树所有系列的所有规格组合,经本地旧 python 代理(8080,在线模式)
逐个请求 webGetPrice 存入 proxy_cache;完成后重跑 migrate_price_cache.py 平移到新服务。
用法:先双击 启动旧代理预热数据.bat(联网),确认 http://localhost:8080/standalone 能开,
然后: python tools/preheat_prices.py (全量,约 3-6 万请求,20-60 分钟)
python tools/preheat_prices.py 1122 1506 (只预热指定系列,调试用)
组合策略:每系列"全默认组合 + 每组每个选项单独变化"——覆盖换任一规格的真实查询主路径;
可选 ALL=1 环境变量改全笛卡尔积(每系列上限 300)。
(请求走 http.client 直连 127.0.0.1:8080 本地代理,不访问任何外部地址。)
"""
import base64
import http.client
import json
import os
import sys
import time
import zipfile
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
ZIPS = os.path.join(ROOT, "成套报价软件", "bin", "Debug", "元件库网页", "data", "dian", "electricData_zips")
ONLY = sys.argv[1:]
全组合 = os.environ.get("ALL") == "1"
HOST, PORT, PATH = "127.0.0.1", 8080, "/dx2/pcprice/default/dx2/interface/webGetPrice"
def 全树系列():
z = zipfile.ZipFile(os.path.join(ZIPS, "s.zip"))
d = json.loads(z.read("s.js"))
return [(x["a"], x["b"]) for x in d if x.get("c") == 1]
def 在线拉ZIP(v逻辑名):
"""经本地旧代理在线拉 electricData 包(自动缓存);返回 bytes(ZIP)。"""
v = base64.b64encode(v逻辑名.encode()).decode().rstrip("=")
try:
conn = http.client.HTTPConnection(HOST, PORT, timeout=25)
conn.request("GET", "/electricData?v=" + v + "&t=" + str(int(time.time() * 1000)))
r = conn.getresponse()
body = r.read()
conn.close()
return body if r.status == 200 and len(body) > 100 else None
except Exception:
return None
def 树系列(body):
"""从树 ZIP bytes 解析叶子系列号(s 字段)列表。"""
import io as _io
try:
z = zipfile.ZipFile(_io.BytesIO(body))
d = json.loads(z.read(z.namelist()[0]))
return [x["s"] for x in d if isinstance(x, dict) and x.get("s")]
except Exception:
return []
def 全部系列():
"""f.zip=厂家列表(o=代表系列号=树id)→拉每厂树→汇总全部叶子系列号。"""
z = zipfile.ZipFile(os.path.join(ZIPS, "f.zip"))
f = json.loads(z.read(z.namelist()[0]))
# ★ 树 id = f 字段(正泰=1 德力西=2,与 1.zip/2.zip 吻合);o 是代表系列号不是树 id
厂家 = sorted(set(x["f"] for x in f
if isinstance(x, dict) and x.get("f") not in (0, "0", None, "", False)),
key=str)
print("厂家树数:", len(厂家), flush=True)
系列 = []
拉树成功 = 0
for i, o in enumerate(厂家):
# 本地已有树的直接读
zp = os.path.join(ZIPS, str(o) + ".zip")
leaves = []
if os.path.exists(zp):
try:
zz = zipfile.ZipFile(zp)
leaves = 树系列(zz.read(zz.namelist()[0]))
except Exception:
pass
if not leaves:
body = 在线拉ZIP("z_seri/%s.jss" % o)
if body:
leaves = 树系列(body)
time.sleep(0.1)
if leaves:
拉树成功 += 1
系列.extend(leaves)
if (i + 1) % 50 == 0:
print(" 厂家进度 %d/%d, 已拉树 %d, 系列累计 %d" % (i + 1, len(厂家), 拉树成功, len(系列)), flush=True)
去重 = sorted(set(str(x) for x in 系列))
print("树拉取 %d/%d, 系列总数(去重) %d" % (拉树成功, len(厂家), len(去重)), flush=True)
return [int(x) for x in 去重]
def 在线拉系列ZIP(系列id):
"""经本地旧代理在线拉系列规格包(z_seri/N.jss);返回解析出的 b 数组。"""
body = 在线拉ZIP("z_seri/%s.jss" % 系列id)
if body is None:
return None
import io as _io
try:
z = zipfile.ZipFile(_io.BytesIO(body))
data = json.loads(z.read(z.namelist()[0]))
return data.get("b", []) if isinstance(data, dict) else []
except Exception:
return None
def 规格组合(系列id):
"""从系列 ZIP 的 b 数组解析规格组:ad=组号,ab=选项id。
返回组合列表(选项连串,如 "337-350-15-9-29-30-37")。"""
zp = os.path.join(ZIPS, str(系列id) + ".zip")
b = None
if os.path.exists(zp):
z = zipfile.ZipFile(zp)
inner = z.namelist()[0]
data = json.loads(z.read(inner))
data = None
zp = os.path.join(ZIPS, str(系列id) + ".zip")
import io as _io
if os.path.exists(zp):
try:
data = json.loads(zipfile.ZipFile(zp).read(zipfile.ZipFile(zp).namelist()[0]))
except Exception:
data = None
if data is None:
body = 在线拉ZIP("z_seri/%s.jss" % 系列id)
if body is None:
return []
try:
data = json.loads(zipfile.ZipFile(_io.BytesIO(body)).read(
zipfile.ZipFile(_io.BytesIO(body)).namelist()[0]))
except Exception:
return []
# ★ 规格组=顶层 c 键里 bc==0 的项(组名 aj),选项=ar[].as;b 数组是附件树不是规格
groups = {}
c = data.get("c") if isinstance(data, dict) else None
if not c:
return []
for 组 in c:
if not isinstance(组, dict) or 组.get("bc") != 0:
continue
p_ = 组.get("p")
ar = 组.get("ar") or []
选项 = [o["as"] for o in ar if isinstance(o, dict) and o.get("as") not in (None, 0)]
if p_ is not None and 选项:
groups.setdefault(p_, [])
for a in 选项:
if a not in groups[p_]:
groups[p_].append(a)
if not groups:
return []
ads = sorted(groups)
默认 = [str(groups[ad][0]) for ad in ads]
组合 = ["-".join(默认)]
for i, ad in enumerate(ads):
for ab in groups[ad][1:]:
c = list(默认)
c[i] = str(ab)
组合.append("-".join(c))
if 全组合:
import itertools
allc = ["-".join(p) for p in itertools.product(*[[str(a) for a in groups[ad]] for ad in ads])]
组合 = allc[:300]
return 组合
def 查价(系列id, 组合):
参数 = {"s": str(系列id), "options": [组合], "time": int(time.time() * 1000), "from": "dx2jsapi"}
info = base64.b64encode(json.dumps(参数, ensure_ascii=False).encode("utf-8")).decode()
try:
conn = http.client.HTTPConnection(HOST, PORT, timeout=15)
conn.request("GET", PATH + "?infoJson=" + info)
r = conn.getresponse()
body = r.read()
conn.close()
ok = b'"code": 0' in body or b'"code":0' in body
return len(body), ok
except Exception as e:
return -1, str(e)[:50]
def main():
if ONLY:
系列 = [int(o) for o in ONLY]
print("指定系列数:", len(系列))
else:
系列 = 全部系列() # f.zip 厂家→每厂树→全部叶子系列号
import threading
lock = threading.Lock()
stat = [0, 0, 0] # 序列idx, 总, 成功
def 处理(sid):
组合 = 规格组合(sid)
局部总 = 局部成 = 0
for c in 组合:
n, ok = 查价(sid, c)
局部总 += 1
if ok:
局部成 += 1
time.sleep(0.01)
with lock:
stat[0] += 1
stat[1] += 局部总
stat[2] += 局部成
if stat[0] % 100 == 0:
速率 = stat[1] / max(1, time.time() - t0)
print("[%d/%d] 系列%s: %d组合 累计%d 成功%d (%.1f/s)" %
(stat[0], len(系列), sid, len(组合), stat[1], stat[2], 速率), flush=True)
t0 = time.time()
from concurrent.futures import ThreadPoolExecutor
with ThreadPoolExecutor(max_workers=4) as pool:
list(pool.map(处理, 系列))
总, 成功 = stat[1], stat[2]
print()
print("完成: 请求 %d, 成功 %d, 用时 %.0f 分钟" % (总, 成功, (time.time() - t0) / 60))
print("下一步: python tools/migrate_price_cache.py 平移到新服务")
if __name__ == "__main__":
main()