Files
wiki/skills/jd-fapai-scrape/scrape.py
T

564 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
京东法拍房源抓取脚本 v3
- 项目链接 (detail_url): https://m.jd.com/product/{skuId}.html
- 拍卖轮次 (auction_round): 优先从 getAuctionLabelConfigs API 获取,无标签时从折扣率推断
- 去重 key = paimaiId(每次拍卖唯一,同一房产的一拍/二拍为不同 paimaiId,各自保留)
- 增量更新:遇到已知 paimaiId 即停止滚动,不全量下载
- 按发布时间降序排序(最新在前)
用法:
python3 scrape.py --keyword 东莞市 --output-dir data # 首次抓取
python3 scrape.py --keyword 东莞市 --output-dir data -i # 增量更新
"""
import argparse
import csv
import datetime
import json
import os
import sys
import time
from collections import Counter
BASE_URL = (
"https://pro.m.jd.com/mall/active/3Rja7L2jmC72Ta8eoa31VFDvaAjf/index.html"
"?pageParamMiddle=judicature_search_middle"
"&pageParam=judicature_icon_search_after"
"&pageFrom=judicature_search_home"
"&tabParam=all"
"&fixSearchParams=false"
"&navh=49"
"&stath=33"
"&tttparams=W6cw9AU4eyJhZGRyZXNzSWQiOjEzODUwMzYwMCwiYXJlYUNvZGUiOjAsImNvcm5lciI6MSwiZF9icmFuZCI6IkhPTk9SIiwiZGwiOjIsImdMYXQiOiIyMi42MjM4MjkiLCJnTG5nIjoiMTE0LjAyNzk0OCIsImdwc19hcmVhIjoiMTlfMTYwN18zMTU1XzYyMTE5IiwibGF0IjoyMi41OTkwMzcsImxic0FyZWEiOiIxOV8xNjA3XzMxNTVfNjIxMTkiLCJsYnNMYXQiOiIyMi42MDA1MzciLCJsYnNMbmciOiIxMTQuMDE0Njk4IiwibG5nIjoxMTQuMDE2ODksIm1vZGVsIjoiVkVSLUFOMTAiLCJvcyI6IjE2IiwicG9zTGF0IjoiMjIuNjIzODI5IiwicG9zTG5nIjoiMTE0LjAyNzk0OCIsInByc3RhdGUiOiIwIiwic2NhbGUiOjMsInVlbXBzIjoiMC0yLTk5OSIsInVuX2FyZWEiOiIxOV8xNjA3XzQ3Mzg4XzYyMTM5Iiwid2lkdGgiOjEwNj8B9"
"&spo_lng=114.01689"
"&spo_lbsEffect=2"
"&spo_reqSource=1"
"&spo_multiThirdCateIds={cate_id}"
"&spo_lat=22.599037"
"&spo_keyword={keyword}"
"&spo_sortField={sort_field}"
)
STATUS_MAP = {0: "未开始", 1: "进行中", 2: "已结束", 3: "已撤回", 4: "已流拍"}
AUCTION_TYPE_MAP = {1: "诉讼拍卖", 5: "司法拍卖", 7: "商业拍卖"}
def find_chromium():
for path in [
"/usr/bin/chromium-browser", "/usr/bin/chromium",
"/usr/bin/google-chrome", "/usr/bin/google-chrome-stable",
]:
if os.path.isfile(path):
return path
return None
def ts_to_str(ts):
if ts:
try:
return datetime.datetime.fromtimestamp(ts / 1000).strftime(
"%Y-%m-%d %H:%M:%S"
)
except Exception:
return str(ts)
return ""
def infer_round_from_discount(discount_rate):
"""从折扣率推断拍卖轮次(7折≈一拍, 5.6折≈二拍/变卖, 10折≈无折扣)"""
if not discount_rate:
return ""
if discount_rate >= 9.5:
return "无折扣"
if discount_rate >= 6.5:
return "一拍(推断)"
if discount_rate >= 5.0:
return "二拍/变卖(推断)"
return "低折扣"
def load_existing(json_path):
if not os.path.exists(json_path):
return [], set()
with open(json_path, "r", encoding="utf-8") as f:
data = json.load(f)
known_ids = {item["paimaiId"] for item in data if "paimaiId" in item}
return data, known_ids
def fetch_listings(keyword, cate_id, sort_field, max_scrolls,
known_paimai_ids=None, early_stop=True):
url = BASE_URL.format(cate_id=cate_id, keyword=keyword, sort_field=sort_field)
print(f"搜索 URL: {url}")
print(f"关键词: {keyword} | 类目ID: {cate_id} | 排序: {sort_field}")
if known_paimai_ids is not None and early_stop:
print(f"增量模式: 已知 {len(known_paimai_ids)} 个 paimaiId,遇到已知项即停止")
try:
from playwright.sync_api import sync_playwright
except ImportError:
print("错误: 未安装 Playwright,请运行: pip install playwright")
sys.exit(1)
chromium_path = find_chromium()
with sync_playwright() as p:
launch_args = ["--no-sandbox", "--disable-gpu", "--disable-dev-shm-usage"]
if chromium_path:
browser = p.chromium.launch(
executable_path=chromium_path, headless=True, args=launch_args
)
else:
try:
browser = p.chromium.launch(headless=True, args=launch_args)
except Exception:
print("错误: 未找到 Chromium,请运行:")
print(" python3 -m playwright install chromium")
sys.exit(1)
context = browser.new_context(
user_agent=(
"Mozilla/5.0 (Linux; Android 13; VER-AN10) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/116.0.0.0 Mobile Safari/537.36"
),
viewport={"width": 375, "height": 812},
is_mobile=True,
)
page = context.new_page()
# 用 dict 在闭包中累积数据
all_listings = {}
current_info = {}
label_configs = {}
state = {"search_call_count": 0, "stopped_early": False}
def handle_response(response):
resp_url = response.url
if "api.m.jd.com" not in resp_url or "functionId=" not in resp_url:
return
try:
body = response.text()
if not body or len(body) <= 50:
return
data = json.loads(body)
except Exception:
return
if "getSearchData" in resp_url:
state["search_call_count"] += 1
result_data = data.get("data", {}).get("resultData", [])
for item in result_data:
d = item.get("data", {})
if d and "paimaiId" in d:
all_listings[d["paimaiId"]] = d
# 增量模式:本批全部已知就停
if known_paimai_ids is not None and early_stop and result_data:
new_in_batch = [
item for item in result_data
if item.get("data", {}).get("paimaiId") not in known_paimai_ids
]
if len(new_in_batch) == 0:
print(f" 第 {state['search_call_count']} 批数据全部已知,停止滚动")
state["stopped_early"] = True
elif "getPaimaiCurrentInfoByIdsForApi" in resp_url:
if data.get("statusCode") == 200:
for pid, info in data.get("data", {}).items():
current_info[pid] = info
elif "getAuctionLabelConfigs" in resp_url:
if data.get("code") == 0:
for pid, config in data.get("data", {}).items():
labels = config.get("labelConfigs", [])
label_configs[pid] = [
l.get("labelName", "") for l in labels
]
page.on("response", handle_response)
print("正在打开页面...")
page.goto(url, wait_until="networkidle", timeout=60000)
print("页面加载完成。")
# 从 SSR 内嵌数据中提取 page 1API 只返回 page 2+page 1 在 HTML 里)
ssr_listings = page.evaluate("""() => {
const scripts = document.querySelectorAll('script');
for (const s of scripts) {
const text = s.textContent || '';
if (!text.includes('resultData') || !text.includes('paimaiId')) continue;
const idx = text.indexOf('"resultData"');
if (idx === -1) continue;
const arrStart = text.indexOf('[', idx);
if (arrStart === -1) continue;
let depth = 0, arrEnd = -1;
for (let i = arrStart; i < text.length; i++) {
if (text[i] === '[') depth++;
else if (text[i] === ']') { depth--; if (depth === 0) { arrEnd = i + 1; break; } }
}
if (arrEnd === -1) continue;
try {
const arr = JSON.parse(text.substring(arrStart, arrEnd));
return arr.map(item => item.data || item).filter(d => d && d.paimaiId);
} catch(e) { continue; }
}
return [];
}""")
if ssr_listings:
for d in ssr_listings:
if "paimaiId" in d:
all_listings[d["paimaiId"]] = d
# 增量模式:检查 page 1 是否全部已知
if known_paimai_ids is not None and early_stop:
new_in_ssr = [d for d in ssr_listings
if d.get("paimaiId") not in known_paimai_ids]
if len(new_in_ssr) == 0:
print(f" SSR page 1: {len(ssr_listings)} 条全部已知,停止")
state["stopped_early"] = True
print(f" SSR page 1: 提取 {len(ssr_listings)} 条")
time.sleep(3)
# 无限滚动
prev_count = 0
no_progress = 0
for i in range(max_scrolls):
if state["stopped_early"]:
break
page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
time.sleep(2)
curr_count = len(all_listings)
if curr_count > prev_count:
label_count = len(label_configs)
print(f" 第 {i+1} 次滚动: 累计 {curr_count} 条 (标签 {label_count})")
prev_count = curr_count
no_progress = 0
else:
no_progress += 1
page_text = page.evaluate("document.body.innerText")
if "没有更多" in page_text or "到底了" in page_text:
print(f" 第 {i+1} 次滚动时到达底部")
break
if no_progress >= 5:
print(f" 连续 {no_progress} 次无新数据,停止")
break
if not state["stopped_early"]:
page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
time.sleep(3)
# 滚动结束后等待标签 API 响应
time.sleep(2)
browser.close()
# 合并三个 API 的数据
results = []
for pid, listing in all_listings.items():
# 构造项目链接
sku_id = listing.get("skuId", "")
listing["detail_url"] = (
f"https://m.jd.com/product/{sku_id}.html" if sku_id else ""
)
# 合并实时信息
if str(pid) in current_info:
info = current_info[str(pid)]
listing["_currentPrice"] = info.get("currentPrice")
listing["_auctionStatus"] = info.get("auctionStatus")
listing["_bidCount"] = info.get("bidCount")
listing["_accessNumber"] = info.get("accessNumber")
listing["_startTime"] = info.get("startTime")
listing["_endTime"] = info.get("endTime")
# 拍卖轮次:优先用 API 标签,无标签时从折扣率推断
labels = label_configs.get(str(pid), [])
round_labels = [
l for l in labels if l in ("一拍", "二拍", "三拍", "变卖", "重新拍卖")
]
if round_labels:
listing["auction_round"] = "、".join(round_labels)
else:
listing["auction_round"] = infer_round_from_discount(
listing.get("discountRate")
)
listing["all_labels"] = "、".join(labels) if labels else ""
results.append(listing)
return results, state["stopped_early"]
def sort_by_publish_time(listings):
# API (spo_sortField=11) 已按发布时间降序返回,直接保持原顺序
return list(listings)
def merge_incremental(new_listings, existing_path, now_str):
with open(existing_path, "r", encoding="utf-8") as f:
existing = json.load(f)
existing_map = {item["paimaiId"]: item for item in existing if "paimaiId" in item}
new_paimai_ids = set()
new_items = []
updated_items = []
for item in new_listings:
pid = item.get("paimaiId")
if pid is None:
continue
new_paimai_ids.add(pid)
if pid not in existing_map:
item["_first_seen"] = now_str
item["_last_seen"] = now_str
new_items.append(item)
else:
old = existing_map[pid]
old["_last_seen"] = now_str
for key in ["_currentPrice", "_auctionStatus", "_bidCount",
"_accessNumber", "_startTime", "_endTime",
"currentPrice", "currentPriceCN", "paimaiStatus",
"displayStatus", "auction_round", "all_labels"]:
if key in item:
old[key] = item[key]
updated_items.append(old)
merged = []
seen = set()
for item in new_listings:
pid = item.get("paimaiId")
if pid and pid not in seen:
if pid in existing_map:
merged.append(existing_map[pid])
else:
merged.append(item)
seen.add(pid)
for item in existing:
pid = item.get("paimaiId")
if pid and pid not in seen:
merged.append(item)
seen.add(pid)
return merged, new_items, updated_items
def save_csv(listings, csv_path):
with open(csv_path, "w", newline="", encoding="utf-8-sig") as f:
writer = csv.writer(f)
writer.writerow([
"序号", "拍卖ID", "标题", "项目链接", "位置", "用途", "户型",
"面积(㎡)", "小区", "均价", "起拍价(元)", "当前价(元)",
"当前价(万)", "折扣率", "拍卖轮次", "标签", "状态",
"出价次数", "围观次数", "开始时间", "结束时间",
"SKU ID", "拍卖类型", "首次发现", "最后更新",
])
for i, item in enumerate(listings):
house = item.get("houseAttributes", {})
writer.writerow([
i + 1,
item.get("paimaiId", ""),
item.get("title", ""),
item.get("detail_url", ""),
"、".join(item.get("location", [])),
house.get("housePurpose", ""),
house.get("houseTypeInfo", ""),
house.get("houseArea", ""),
house.get("plotName", ""),
house.get("averagePrice", ""),
item.get("startPrice", ""),
item.get("_currentPrice", item.get("currentPrice", "")),
item.get("currentPriceCN", ""),
item.get("discountRate", ""),
item.get("auction_round", ""),
item.get("all_labels", ""),
STATUS_MAP.get(
item.get("_auctionStatus", item.get("paimaiStatus")), ""
),
item.get("_bidCount", ""),
item.get("_accessNumber", ""),
ts_to_str(item.get("_startTime")),
ts_to_str(item.get("_endTime")),
item.get("skuId", ""),
AUCTION_TYPE_MAP.get(
item.get("auctionType"),
str(item.get("auctionType", "")),
),
item.get("_first_seen", ""),
item.get("_last_seen", ""),
])
def print_stats(listings, new_items=None, updated_items=None, early_stopped=False):
print(f"\n{'=' * 60}")
print(f"房源总数: {len(listings)} 条")
if new_items is not None:
print(f"本次新增: {len(new_items)} 条")
if updated_items is not None:
print(f"本次更新: {len(updated_items)} 条")
if early_stopped:
print(f"(增量模式: 遇到已知数据即停止,未全量下载)")
print(f"{'=' * 60}")
prices = [
item.get("_currentPrice") or item.get("currentPrice", 0)
for item in listings
if item.get("_currentPrice") or item.get("currentPrice")
]
if prices:
print(f"\n价格统计:")
print(f" 最低: ¥{min(prices):,.2f}")
print(f" 最高: ¥{max(prices):,.2f}")
print(f" 平均: ¥{sum(prices) / len(prices):,.2f}")
print(f" 中位: ¥{sorted(prices)[len(prices) // 2]:,.2f}")
statuses = Counter(
STATUS_MAP.get(
item.get("_auctionStatus", item.get("paimaiStatus")), "未知"
)
for item in listings
)
print(f"\n状态分布:")
for s, c in statuses.most_common():
print(f" {s}: {c} 条")
rounds = Counter(item.get("auction_round", "") for item in listings)
print(f"\n拍卖轮次分布:")
for r, c in rounds.most_common():
if r:
print(f" {r}: {c} 条")
if new_items:
print(f"\n新增房源 (前 10 条):")
for i, item in enumerate(new_items[:10]):
price = item.get("_currentPrice") or item.get("currentPrice", 0)
print(f" [{i+1}] {item.get('title', 'N/A')[:50]}")
print(f" ¥{price:,.0f} | {item.get('auction_round','')} | "
f"{ts_to_str(item.get('_startTime'))}")
print(f" {item.get('detail_url','')}")
def scrape(keyword="塘厦", cate_id="15", sort_field="11",
max_scrolls=100, output_dir="output/法拍",
incremental=False, csv_only=False):
os.makedirs(output_dir, exist_ok=True)
now_str = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S")
print(f"运行时间: {now_str}")
print(f"模式: {'增量更新' if incremental else '首次抓取'}")
print()
json_path = os.path.join(output_dir, f"{keyword}_法拍房源.json")
csv_path = os.path.join(output_dir, f"{keyword}_法拍房源.csv")
known_ids = None
if incremental:
existing, known_ids = load_existing(json_path)
if not known_ids:
print("未找到已有 JSON 数据,转为首次抓取模式")
incremental = False
else:
print(f"已有数据: {len(existing)} 条, 已知 {len(known_ids)} 个 paimaiId")
if csv_only:
print("提示: csv_only 模式下增量更新仍需 JSON 做合并,临时使用后删除")
new_listings, early_stopped = fetch_listings(
keyword, cate_id, sort_field, max_scrolls,
known_paimai_ids=known_ids if incremental else None,
early_stop=incremental,
)
print(f"\n本次抓取: {len(new_listings)} 条")
if incremental:
merged, new_items, updated_items = merge_incremental(
new_listings, json_path, now_str
)
print(f"合并后总数: {len(merged)} 条")
print(f"新增: {len(new_items)} 条 | 更新: {len(updated_items)} 条")
if not csv_only:
with open(json_path, "w", encoding="utf-8") as f:
json.dump(merged, f, indent=2, ensure_ascii=False)
save_csv(merged, csv_path)
if csv_only and os.path.exists(json_path):
os.remove(json_path)
log_path = os.path.join(output_dir, f"{keyword}_增量日志.log")
with open(log_path, "a", encoding="utf-8") as f:
f.write(f"\n{'=' * 60}\n")
f.write(f"时间: {now_str}\n")
f.write(f"新增: {len(new_items)} 条 | 更新: {len(updated_items)}\n")
f.write(f"提前停止: {'是' if early_stopped else '否'}\n")
for item in new_items:
price = item.get("_currentPrice") or item.get("currentPrice", 0)
f.write(f" [新] {item.get('title', 'N/A')[:60]} | "
f"{item.get('auction_round','')} | ¥{price:,.0f} | "
f"{ts_to_str(item.get('_startTime'))}\n"
f" {item.get('detail_url','')}\n")
for item in updated_items:
price = item.get("_currentPrice") or item.get("currentPrice", 0)
f.write(f" [更] {item.get('title', 'N/A')[:60]} | "
f"{item.get('auction_round','')} | ¥{price:,.0f} | "
f"{ts_to_str(item.get('_startTime'))}\n")
print(f"\n输出文件:")
if not csv_only:
print(f" JSON: {json_path}")
print(f" CSV: {csv_path}")
print(f" 日志: {log_path}")
print_stats(merged, new_items, updated_items, early_stopped=early_stopped)
return merged
else:
sorted_listings = sort_by_publish_time(new_listings)
for item in sorted_listings:
item["_first_seen"] = now_str
item["_last_seen"] = now_str
if not csv_only:
with open(json_path, "w", encoding="utf-8") as f:
json.dump(sorted_listings, f, indent=2, ensure_ascii=False)
save_csv(sorted_listings, csv_path)
print(f"\n输出文件:")
if not csv_only:
print(f" JSON: {json_path}")
print(f" CSV: {csv_path}")
print_stats(sorted_listings)
return sorted_listings
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="京东法拍房源抓取(支持增量更新)")
parser.add_argument("--keyword", default="塘厦", help="搜索关键词 (默认: 塘厦)")
parser.add_argument("--cate-id", default="15", help="类目ID (默认: 15=法拍房)")
parser.add_argument("--sort-field", default="11", help="排序 (默认: 11=最新发布)")
parser.add_argument("--max-scrolls", type=int, default=100, help="最大滚动次数")
parser.add_argument("--output-dir", default="output/法拍", help="输出目录 (默认: output/法拍)")
parser.add_argument("--incremental", "-i", action="store_true",
help="增量更新:只下载新房源,遇到已知项即停止")
parser.add_argument("--csv-only", action="store_true",
help="只输出 CSV,不保存 JSON(增量模式下 JSON 临时使用后删除)")
args = parser.parse_args()
scrape(
keyword=args.keyword,
cate_id=args.cate_id,
sort_field=args.sort_field,
max_scrolls=args.max_scrolls,
output_dir=args.output_dir,
incremental=args.incremental,
csv_only=args.csv_only,
)