Files
rakuten-api/scripts/probe_checkout.py
T

229 lines
8.5 KiB
Python

"""Rakuten 下单流程只读探针:摸清 cart / 加购响应 / 确认页结构
只读模式:不发任何写请求。只 GET 现有购物车页(用户可能在历史会话里已有商品),
分析页面结构、找出 item_id 与"レジに進む"链接的位置。
要跑通完整流程(加购 → 确认页),加 --mutate 参数;加购后尝试自动清理。
用法:
.venv/Scripts/python.exe scripts/probe_checkout.py # 只读
.venv/Scripts/python.exe scripts/probe_checkout.py --mutate # 含加购与清理
.venv/Scripts/python.exe scripts/probe_checkout.py --mutate --item-url https://item.rakuten.co.jp/...
"""
from __future__ import annotations
import argparse
import asyncio
import json
import re
import sys
from pathlib import Path
# 允许 `python scripts/probe_checkout.py` 直接运行
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
import httpx
from app.shared.config import Settings, get_settings
from app.shared.proxy import httpx_client_options
from app.trading.core import auth_site
PROBE_DIR = Path(__file__).resolve().parent.parent / ".probe" / "checkout"
PROBE_DIR.mkdir(parents=True, exist_ok=True)
# 默认测试商品:一个低单价、稳定的乐天市场直营商品(用户可覆盖)
# 留空则只跑只读流程;--mutate 必须显式传 --item-url
DEFAULT_ITEM_URL = ""
MOBILE_UA = auth_site.RAKUTEN_USER_AGENT
def load_cookies(state_path: Path) -> list[dict]:
state = json.loads(state_path.read_text(encoding="utf-8"))
return state.get("cookies", [])
def build_client(cookies: list[dict], settings: Settings) -> httpx.AsyncClient:
client = httpx.AsyncClient(
headers=auth_site.PROFILES["rakuten"].headers(),
timeout=30.0,
follow_redirects=True,
http2=True,
**httpx_client_options(settings),
)
for cookie in cookies:
name = cookie.get("name")
value = cookie.get("value")
if not name or value is None:
continue
client.cookies.set(
name, value, domain=cookie.get("domain") or "", path=cookie.get("path") or "/"
)
return client
def save(name: str, content: str | bytes, *, is_bytes: bool = False) -> Path:
path = PROBE_DIR / name
if is_bytes:
path.write_bytes(content) # type: ignore[arg-type]
else:
path.write_text(content, encoding="utf-8") # type: ignore[arg-type]
print(f" saved → {path}")
return path
async def probe_readonly(client: httpx.AsyncClient) -> None:
print("\n=== 步骤 1:GET 购物车页 ===")
resp = await client.get(auth_site.RAKUTEN_CART_URL)
print(f" status={resp.status_code} final_url={resp.url}")
save("01-cart.html", resp.text)
body = resp.text
logged_out = auth_site.RAKUTEN_LOGGED_OUT_MARKER in body
print(f" logged_in={not logged_out}")
# 找页面里出现的 item_id(购物车里所有商品)
# 常见模式:data-item-id="..."、name="item_id" value="..."、/item/.../ 等
item_ids = set(re.findall(r'"item_id"\s*[:=]\s*"?(\d+)', body))
print(f" cart item_ids={sorted(item_ids)}")
# 找 "レジに進む" 链接
checkout_links = re.findall(
r'href=["\']([^"\']*checkout[^"\']*)["\']', body, re.IGNORECASE
)
checkout_links += re.findall(
r'href=["\']([^"\']*step\.rakuten[^"\']*)["\']', body, re.IGNORECASE
)
print(f" checkout links={sorted(set(checkout_links))[:5]}")
# 如果有 checkout 链接,尝试 GET
if checkout_links:
target = checkout_links[0]
if not target.startswith("http"):
target = "https://sp.cart.step.rakuten.co.jp" + target
print(f"\n=== 步骤 2:GET checkout 链接 {target} ===")
resp2 = await client.get(target)
print(f" status={resp2.status_code} final_url={resp2.url}")
save("02-checkout.html", resp2.text)
else:
print("\n 购物车里没有可结账商品(或链接结构变化),跳过 checkout 探测")
async def probe_with_mutation(client: httpx.AsyncClient, item_url: str) -> None:
"""完整跑一遍:加购 → 校验 → 进确认页。完成后尝试清理"""
from app.scraping.parsers.item import parse_item_detail
from app.scraping.parsers.state import require_state_marker
import app.scraping.parsers.state as state_mod
print(f"\n=== 加载商品 purchase 块:{item_url} ===")
resp = await client.get(item_url)
print(f" item page status={resp.status_code}")
save("00-item-page.html", resp.text)
# 抽 __INITIAL_STATE__
m = re.search(
r'<script[^>]*>window\.\_\_INITIAL_STATE\_\_\s*=\s*(\{.*?\});?\s*</script>',
resp.text, re.DOTALL,
)
if not m:
# 备用:脚本不一定有 window. 前缀
m = re.search(
r'\_\_INITIAL\_STATE\_\_\s*[:=]\s*(\{.*?\})\s*[;<]',
resp.text, re.DOTALL,
)
if not m:
print(" ✗ 抽不出 __INITIAL_STATE__,可能不是手机版模板或反爬被触发")
return
state = json.loads(m.group(1))
save("00-item-state.json", json.dumps(state, ensure_ascii=False, indent=2))
shop_code_match = re.search(r"/([^/]+)/([^/]+)/?$", item_url.rstrip("/"))
shop_code = shop_code_match.group(1) if shop_code_match else ""
detail = parse_item_detail(state, item_url=item_url, shop_code=shop_code, include_sku_variants=True)
pu = detail.purchase
print(f" cart_url={pu.cart_url}")
print(f" form_fields={pu.form_fields}")
print(f" quantity_field={pu.quantity_field} variant_field={pu.variant_field} options_field={pu.options_field}")
if not pu.cart_url:
print(" ✗ basketDomain 为空,不能加购")
return
# 组装加购表单
payload = dict(pu.form_fields)
payload[pu.quantity_field or "units"] = "1"
# 单一库存商品 variant_id 已在 form_fields 里;多规格需调用方传
print(f"\n=== POST 加购 → {pu.cart_url} ===")
print(f" payload={payload}")
resp = await client.post(pu.cart_url, data=payload)
print(f" status={resp.status_code} final_url={resp.url}")
save("03-cart-add-response.html", resp.text)
print("\n=== GET 购物车页(校验)===")
resp = await client.get(auth_site.RAKUTEN_CART_URL)
print(f" status={resp.status_code}")
save("04-cart-after-add.html", resp.text)
item_id = pu.form_fields.get("item_id", "")
found = item_id and item_id in resp.text
print(f" item_id={item_id} in_cart={found}")
# 找 レジに進む 链接
checkout_links = re.findall(
r'href=["\']([^"\']*checkout[^"\']*)["\']', resp.text, re.IGNORECASE
)
if checkout_links:
target = checkout_links[0]
if not target.startswith("http"):
target = "https://sp.cart.step.rakuten.co.jp" + target
print(f"\n=== GET 确认页 {target} ===")
resp = await client.get(target)
print(f" status={resp.status_code} final_url={resp.url}")
save("05-checkout-confirm.html", resp.text)
# 在确认页里找金额
amounts = re.findall(r"[¥¥]?\s*([\d,]+)\s*円", resp.text)
print(f" 金额候选(前 5)={amounts[:5]}")
print("\n=== 清理:尝试从购物车删除 ===")
# 找删除链接
del_matches = re.findall(
r'href=["\']([^"\']*(?:delete|del|remove)[^"\']*)["\']',
resp.text, re.IGNORECASE,
)
if del_matches:
print(f" delete 链接={del_matches[:3]}")
# 真实删除需 POST + 各种 token,留给用户手动清
async def main() -> int:
parser = argparse.ArgumentParser(description="Rakuten 下单流程探针")
parser.add_argument("--mutate", action="store_true", help="包含加购写操作(默认只读)")
parser.add_argument("--item-url", default=DEFAULT_ITEM_URL, help="测试商品 URL")
args = parser.parse_args()
settings = get_settings()
state_path = settings.auth_state_path / "rakuten_state.json"
if not state_path.exists():
print(f"✗ 找不到登录态文件:{state_path}")
return 1
cookies = load_cookies(state_path)
print(f"加载 {len(cookies)} 条 cookie 自 {state_path}")
async with build_client(cookies, settings) as client:
await probe_readonly(client)
if args.mutate:
if not args.item_url:
print("\n✗ --mutate 需要搭配 --item-url")
return 1
await probe_with_mutation(client, args.item_url)
print(f"\n探针输出目录:{PROBE_DIR}")
return 0
if __name__ == "__main__":
raise SystemExit(asyncio.run(main()))