# -*- coding: utf-8 -*- """全量价格在线预热:遍历全树所有系列的所有规格组合,经本地旧 python 代理(8080,在线模式) 逐个请求 webGetPrice 存入 proxy_cache;完成后重跑 migrate_price_cache.py 平移到新服务。 用法:先双击 启动旧代理预热数据.bat(联网),确认 http://localhost:8080/standalone 能开, 然后: python tools/preheat_prices.py (全量,约 3-6 万请求,20-60 分钟) python tools/preheat_prices.py 1122 1506 (只预热指定系列,调试用) 组合策略:每系列"全默认组合 + 每组每个选项单独变化"——覆盖换任一规格的真实查询主路径; 可选 ALL=1 环境变量改全笛卡尔积(每系列上限 300)。 (请求走 http.client 直连 127.0.0.1:8080 本地代理,不访问任何外部地址。) """ import base64 import http.client import json import os import sys import time import zipfile ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) ZIPS = os.path.join(ROOT, "成套报价软件", "bin", "Debug", "元件库网页", "data", "dian", "electricData_zips") ONLY = sys.argv[1:] 全组合 = os.environ.get("ALL") == "1" HOST, PORT, PATH = "127.0.0.1", 8080, "/dx2/pcprice/default/dx2/interface/webGetPrice" def 全树系列(): z = zipfile.ZipFile(os.path.join(ZIPS, "s.zip")) d = json.loads(z.read("s.js")) return [(x["a"], x["b"]) for x in d if x.get("c") == 1] def 在线拉ZIP(v逻辑名): """经本地旧代理在线拉 electricData 包(自动缓存);返回 bytes(ZIP)。""" v = base64.b64encode(v逻辑名.encode()).decode().rstrip("=") try: conn = http.client.HTTPConnection(HOST, PORT, timeout=25) conn.request("GET", "/electricData?v=" + v + "&t=" + str(int(time.time() * 1000))) r = conn.getresponse() body = r.read() conn.close() return body if r.status == 200 and len(body) > 100 else None except Exception: return None def 树系列(body): """从树 ZIP bytes 解析叶子系列号(s 字段)列表。""" import io as _io try: z = zipfile.ZipFile(_io.BytesIO(body)) d = json.loads(z.read(z.namelist()[0])) return [x["s"] for x in d if isinstance(x, dict) and x.get("s")] except Exception: return [] def 全部系列(): """f.zip=厂家列表(o=代表系列号=树id)→拉每厂树→汇总全部叶子系列号。""" z = zipfile.ZipFile(os.path.join(ZIPS, "f.zip")) f = json.loads(z.read(z.namelist()[0])) # ★ 树 id = f 字段(正泰=1 德力西=2,与 1.zip/2.zip 吻合);o 是代表系列号不是树 id 厂家 = sorted(set(x["f"] for x in f if isinstance(x, dict) and x.get("f") not in (0, "0", None, "", False)), key=str) print("厂家树数:", len(厂家), flush=True) 系列 = [] 拉树成功 = 0 for i, o in enumerate(厂家): # 本地已有树的直接读 zp = os.path.join(ZIPS, str(o) + ".zip") leaves = [] if os.path.exists(zp): try: zz = zipfile.ZipFile(zp) leaves = 树系列(zz.read(zz.namelist()[0])) except Exception: pass if not leaves: body = 在线拉ZIP("z_seri/%s.jss" % o) if body: leaves = 树系列(body) time.sleep(0.1) if leaves: 拉树成功 += 1 系列.extend(leaves) if (i + 1) % 50 == 0: print(" 厂家进度 %d/%d, 已拉树 %d, 系列累计 %d" % (i + 1, len(厂家), 拉树成功, len(系列)), flush=True) 去重 = sorted(set(str(x) for x in 系列)) print("树拉取 %d/%d, 系列总数(去重) %d" % (拉树成功, len(厂家), len(去重)), flush=True) return [int(x) for x in 去重] def 在线拉系列ZIP(系列id): """经本地旧代理在线拉系列规格包(z_seri/N.jss);返回解析出的 b 数组。""" body = 在线拉ZIP("z_seri/%s.jss" % 系列id) if body is None: return None import io as _io try: z = zipfile.ZipFile(_io.BytesIO(body)) data = json.loads(z.read(z.namelist()[0])) return data.get("b", []) if isinstance(data, dict) else [] except Exception: return None def 规格组合(系列id): """从系列 ZIP 的 b 数组解析规格组:ad=组号,ab=选项id。 返回组合列表(选项连串,如 "337-350-15-9-29-30-37")。""" zp = os.path.join(ZIPS, str(系列id) + ".zip") b = None if os.path.exists(zp): z = zipfile.ZipFile(zp) inner = z.namelist()[0] data = json.loads(z.read(inner)) data = None zp = os.path.join(ZIPS, str(系列id) + ".zip") import io as _io if os.path.exists(zp): try: data = json.loads(zipfile.ZipFile(zp).read(zipfile.ZipFile(zp).namelist()[0])) except Exception: data = None if data is None: body = 在线拉ZIP("z_seri/%s.jss" % 系列id) if body is None: return [] try: data = json.loads(zipfile.ZipFile(_io.BytesIO(body)).read( zipfile.ZipFile(_io.BytesIO(body)).namelist()[0])) except Exception: return [] # ★ 规格组=顶层 c 键里 bc==0 的项(组名 aj),选项=ar[].as;b 数组是附件树不是规格 groups = {} c = data.get("c") if isinstance(data, dict) else None if not c: return [] for 组 in c: if not isinstance(组, dict) or 组.get("bc") != 0: continue p_ = 组.get("p") ar = 组.get("ar") or [] 选项 = [o["as"] for o in ar if isinstance(o, dict) and o.get("as") not in (None, 0)] if p_ is not None and 选项: groups.setdefault(p_, []) for a in 选项: if a not in groups[p_]: groups[p_].append(a) if not groups: return [] ads = sorted(groups) 默认 = [str(groups[ad][0]) for ad in ads] 组合 = ["-".join(默认)] for i, ad in enumerate(ads): for ab in groups[ad][1:]: c = list(默认) c[i] = str(ab) 组合.append("-".join(c)) if 全组合: import itertools allc = ["-".join(p) for p in itertools.product(*[[str(a) for a in groups[ad]] for ad in ads])] 组合 = allc[:300] return 组合 def 查价(系列id, 组合): 参数 = {"s": str(系列id), "options": [组合], "time": int(time.time() * 1000), "from": "dx2jsapi"} info = base64.b64encode(json.dumps(参数, ensure_ascii=False).encode("utf-8")).decode() try: conn = http.client.HTTPConnection(HOST, PORT, timeout=15) conn.request("GET", PATH + "?infoJson=" + info) r = conn.getresponse() body = r.read() conn.close() ok = b'"code": 0' in body or b'"code":0' in body return len(body), ok except Exception as e: return -1, str(e)[:50] def main(): if ONLY: 系列 = [int(o) for o in ONLY] print("指定系列数:", len(系列)) else: 系列 = 全部系列() # f.zip 厂家→每厂树→全部叶子系列号 import threading lock = threading.Lock() stat = [0, 0, 0] # 序列idx, 总, 成功 def 处理(sid): 组合 = 规格组合(sid) 局部总 = 局部成 = 0 for c in 组合: n, ok = 查价(sid, c) 局部总 += 1 if ok: 局部成 += 1 time.sleep(0.01) with lock: stat[0] += 1 stat[1] += 局部总 stat[2] += 局部成 if stat[0] % 100 == 0: 速率 = stat[1] / max(1, time.time() - t0) print("[%d/%d] 系列%s: %d组合 累计%d 成功%d (%.1f/s)" % (stat[0], len(系列), sid, len(组合), stat[1], stat[2], 速率), flush=True) t0 = time.time() from concurrent.futures import ThreadPoolExecutor with ThreadPoolExecutor(max_workers=4) as pool: list(pool.map(处理, 系列)) 总, 成功 = stat[1], stat[2] print() print("完成: 请求 %d, 成功 %d, 用时 %.0f 分钟" % (总, 成功, (time.time() - t0) / 60)) print("下一步: python tools/migrate_price_cache.py 平移到新服务") if __name__ == "__main__": main()