#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import csv
import json
import os
import random
import re
import subprocess
import sys
import time
import traceback
import urllib.error
import urllib.request
import uuid
from pathlib import Path
COOKIE = """
""".strip()
KEYWORDS = ["咖啡"]
PAGES = 1
SORT = "general" # general综合 / time_descending最新 / popularity_descending最热
NOTE_TYPE = 0 # 0不限 1视频 2图文
OUT_DIR = "out"
SHOW_TOP = 10
DELAY = 1.5
PAGE_SIZE = 20
HERE = Path(__file__).resolve().parent
SDK_DIR = HERE / "xhs_sdk"
SIGNER = HERE / "xhs_sign.mjs"
SEARCH_HOST = "https://so.xiaohongshu.com"
SEARCH_URI = "/api/sns/web/v2/search/notes"
ME_HOST = "https://edith.xiaohongshu.com"
ME_URI = "/api/sns/web/v2/user/me"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36"
CSV_COLS = ["keyword", "page", "note_id", "type", "title", "author", "author_id",
"liked", "collected", "commented", "shared", "publish_time", "url"]
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
def _http_raw(url, data=None, timeout=60, html=False):
headers = {"user-agent": UA, "accept-language": "zh-CN,zh;q=0.9", "referer": "https://www.xiaohongshu.com/"}
if html:
headers["accept"] = "text/html,*/*;q=0.8"
headers["upgrade-insecure-requests"] = "1"
else:
headers["accept"] = "*/*"
headers["origin"] = "https://www.xiaohongshu.com"
req = urllib.request.Request(url, data=data, headers=headers)
with urllib.request.urlopen(req, timeout=timeout) as r:
return r.read()
def _http_retry(url, data=None, tries=3, timeout=60, html=False):
last_err = None
for i in range(tries):
try:
return _http_raw(url, data=data, timeout=timeout, html=html)
except Exception as e:
last_err = e
if i < tries -1:
time.sleep(1.5*(i+1))
raise last_err
def _is_sec_script(u):
if "as.xiaohongshu.com/api/sec/v1/ds" in u: return True
if re.search(r"fe-static\.xhscdn\.com/as/(v1|v2)/(ds|fp)/", u): return True
if re.search(r"fe-static\.xhscdn\.com/as/v1/[0-9a-f]+/.*/public/", u): return True
return False
def _sbtsource_extras():
try:
body = json.dumps({"callFrom":"web","callback":"","type":"ds","appId":"xhs-pc-web"}).encode()
raw = _http_retry("https://as.xiaohongshu.com/api/sec/v1/sbtsource", data=body, tries=2)
d = json.loads(raw.decode("utf-8","replace")).get("data") or {}
return [u for u in (d.get("signUrl"), d.get("url")) if u]
except Exception:
return []
def _safe_name(url, idx):
base = url.split("?")[0].rstrip("/").rsplit("/",1)[-1] or f"script{idx}.js"
base = re.sub(r"[^\w.-]","_",base)
if not base.endswith(".js"): base += ".js"
return f"{idx:02d}_{base}"
def sdk_ready():
mf = SDK_DIR / "manifest.json"
if not mf.exists(): return False
try: m = json.loads(mf.read_text(encoding="utf-8"))
except Exception: return False
files = [s if isinstance(s,str) else s.get("file") for s in m.get("scripts",[])]
if not files or not all((SDK_DIR/f).exists() for f in files if f): return False
vf = m.get("vendor",{}).get("file") if isinstance(m.get("vendor"),dict) else None
return bool(vf and (SDK_DIR/vf).exists())
def ensure_sdk(force=False):
if not force and sdk_ready(): return
SDK_DIR.mkdir(parents=True, exist_ok=True)
print("首次运行:下载小红书签名脚本,只需一次…")
html_text = None
for page_url in ("https://www.xiaohongshu.com/explore","https://www.xiaohongshu.com/"):
try:
html_text = _http_retry(page_url, html=True).decode("utf-8","replace")
break
except Exception: continue
sec_urls, vendor_url = [], None
if html_text:
for u in re.findall(r'<script[^>]+src=["\']([^"\']+)["\']', html_text):
u = u.replace("&","&")
if not u.startswith("http"): continue
if re.search(r"vendor-dynamic\.[0-9a-f]+\.js",u): vendor_url=u
elif _is_sec_script(u): sec_urls.append(u)
if not vendor_url:
m = re.search(r"https://fe-static\.xhscdn\.com/formula-static/xhs-pc-web/public/resource/js/vendor-dynamic\.[0-9a-f]+\.js", html_text)
if m: vendor_url = m.group(0)
ds_main = "https://as.xiaohongshu.com/api/sec/v1/ds?appId=xhs-pc-web"
if ds_main not in sec_urls: sec_urls.insert(0, ds_main)
for u in _sbtsource_extras():
if u not in sec_urls: sec_urls.append(u)
seen, ordered = set(), []
for u in sec_urls:
if u not in seen: seen.add(u); ordered.append(u)
manifest = {"fetchedAt":time.strftime("%Y-%m-%dT%H:%M:%S"),"scripts":[]}
ok = 0
for i,u in enumerate(ordered,1):
name = "sec_ds.js" if "sec/v1/ds" in u else _safe_name(u,i)
try:
code = _http_retry(u)
(SDK_DIR/name).write_bytes(code)
manifest["scripts"].append({"file":name,"from":u,"bytes":len(code)})
ok +=1
print(f" [OK] {name:<46} {len(code):6d} B")
except Exception as e:
print(f" [!] 跳过 {u[:70]} :: {str(e)[:60]}")
if vendor_url:
try:
code = _http_retry(vendor_url)
(SDK_DIR/"vendor-dynamic.js").write_bytes(code)
manifest["vendor"] = {"file":"vendor-dynamic.js","from":vendor_url,"bytes":len(code)}
print(" [OK] vendor-dynamic.js")
except Exception as e:
print(f" [!] vendor-dynamic下载失败: {str(e)[:60]}")
if not manifest.get("vendor") and (SDK_DIR/"vendor-dynamic.js").exists():
manifest["vendor"] = {"file":"vendor-dynamic.js","from":"(cached)","bytes":0}
print(" [i] 使用本地缓存vendor-dynamic.js")
(SDK_DIR/"manifest.json").write_text(json.dumps(manifest,indent=2),encoding="utf-8")
if not ok: raise RuntimeError("签名脚本下载失败,检查网络")
if not manifest.get("vendor"): raise RuntimeError("找不到vendor-dynamic.js,请浏览器打开小红书首页再重试")
print(f"下载完成 -> {SDK_DIR}\n")
class Signer:
def __init__(self, cookie):
if not SIGNER.exists(): raise RuntimeError(f"找不到 {SIGNER.name},放在py同目录!")
try:
self.proc = subprocess.Popen(["node",str(SIGNER),cookie,str(SDK_DIR)],
stdin=subprocess.PIPE,stdout=subprocess.PIPE,stderr=subprocess.DEVNULL,
text=True,encoding="utf-8",bufsize=1)
except FileNotFoundError: raise RuntimeError("未安装Node.js,请安装node并加入环境变量")
line = self.proc.stdout.readline()
if not line: raise RuntimeError("签名进程无响应")
msg = json.loads(line)
if not msg.get("ready"): raise RuntimeError(f"签名进程启动失败: {msg.get('error')}")
self._seq =0
def sign(self, uri, body):
body_str = json.dumps(body,ensure_ascii=False,separators=(",",":"))
self._seq +=1
self.proc.stdin.write(json.dumps({"uri":uri,"body":body_str,"id":self._seq},ensure_ascii=False)+"\n")
self.proc.stdin.flush()
line = self.proc.stdout.readline()
res = json.loads(line)
if "error" in res: raise RuntimeError(f"签名失败:{res['error']}")
return {"x-s":res["xs"],"x-t":res["xt"],"x-s-common":res["xsc"]}, body_str
def close(self):
try:
self.proc.stdin.write('{"bye":true}\n')
self.proc.stdin.flush()
self.proc.wait(timeout=3)
except Exception:
try: self.proc.kill()
except Exception:pass
class Xhs:
def __init__(self, cookie, delay):
self.cookie = " ".join(cookie.split())
if "a1=" not in self.cookie: raise RuntimeError("Cookie缺少a1字段,无法签名")
self.delay = delay
self.session_id = str(uuid.uuid4())
self.signer = Signer(self.cookie)
def _headers(self, uri, body):
sig, body_str = self.signer.sign(uri, body)
headers = {
"content-type":"application/json;charset=UTF-8",
"user-agent":UA,
"x-s":sig["x-s"], "x-t":sig["x-t"], "x-s-common":sig["x-s-common"],
"x-b3-traceid":uuid.uuid4().hex[:16],
"x-xray-traceid":uuid.uuid4().hex,
"origin":"https://www.xiaohongshu.com",
"referer":"https://www.xiaohongshu.com/",
"accept-language":"zh-CN,zh;q=0.9",
"cookie":self.cookie
}
return headers, body_str
def _call(self, method, host, uri, body=None):
headers, body_str = self._headers(uri, body)
data = body_str.encode("utf-8") if body_str else None
req = urllib.request.Request(host+uri, data=data, headers=headers, method=method)
try:
with urllib.request.urlopen(req,timeout=30) as r:
raw, status = r.read().decode("utf-8","replace"), r.status
except urllib.error.HTTPError as e:
raw, status = e.read().decode("utf-8","replace"), e.code
except Exception as e:
return 0, {"code":-1,"msg":f"网络错误:{e}"}
try: return status, json.loads(raw)
except Exception: return status, {"code":-1,"msg":f"非JSON响应:{raw[:160]}"}
def check_login(self):
st, data = self._call("GET", ME_HOST, ME_URI)
d = data.get("data") or {}
return {"http":st,"code":data.get("code"),"msg":data.get("msg"),
"guest":d.get("guest"),"user_id":d.get("user_id"),
"nickname":d.get("nickname") or d.get("user_nickname")}
def search(self, keyword, page=1, page_size=20, sort="general", note_type=0, search_id=None):
body = {
"keyword":keyword,"page":page,"page_size":page_size,
"search_id":search_id or "".join(random.choice("abcdefghijklmnopqrstuvwxyz0123456789") for _ in range(21)),
"sort":sort,"note_type":note_type,"ext_flags":[],"geo":"",
"image_formats":["jpg","webp","avif"],"session_id":self.session_id
}
return self._call("POST", SEARCH_HOST, SEARCH_URI, body)
def sleep(self):
time.sleep(self.delay + random.random()*0.5)
def close(self):
self.signer.close()
def to_num(v):
if v is None: return 0
m = re.match(r"^([\d.]+)\s*([万亿wW]?)", str(v).strip())
if not m: return 0
try: n = float(m.group(1))
except ValueError: return 0
unit = m.group(2)
if unit and unit in "万wW": n *=10000
if unit == "亿": n *=100000000
return int(round(n))
def extract(keyword, page, data):
rows = []
d = data.get("data") or {}
for it in d.get("items") or []:
c = it.get("note_card")
if not c: continue
nid = it.get("id") or ""
token = it.get("xsec_token") or c.get("xsec_token") or ""
tag = next((x.get("text") for x in (c.get("corner_tag_info") or []) if x.get("type")=="publish_time"),"")
u = c.get("user") or {}
it_info = c.get("interact_info") or {}
rows.append({
"keyword":keyword,"page":page,"note_id":nid,"type":c.get("type","normal"),
"title":" ".join(str(c.get("display_title") or c.get("title") or "").split()),
"author":u.get("nickname") or u.get("nick_name") or "",
"author_id":u.get("user_id") or "",
"liked":to_num(it_info.get("liked_count")),
"collected":to_num(it_info.get("collected_count")),
"commented":to_num(it_info.get("comment_count")),
"shared":to_num(it_info.get("shared_count")),
"publish_time":tag,
"url":f"https://www.xiaohongshu.com/explore/{nid}?xsec_token={token}&xsec_source=pc_search" if token else f"https://www.xiaohongshu.com/explore/{nid}"
})
return rows, bool(d.get("has_more")), d.get("search_id")
def explain(status, data):
if status ==461: return False,True,"签名校验失败(461)"
if status ==406: return True,False,"请求被拒(406)接口变更"
code = data.get("code")
if code ==0: return False,False,None
if code ==-101: return True,False,"Cookie失效,请重新导出cookie"
if code ==-104: return True,False,"游客会话,请登录"
if code ==300011: return True,False,"账号风控拦截,降低频率"
return False,False,f"code={code} msg={data.get('msg','')}"
def main():
ensure_sdk()
xhs = Xhs(COOKIE, DELAY)
try:
me = xhs.check_login()
if me["code"] !=0: raise RuntimeError(f"登录失败: {me.get('msg')}, code={me.get('code')}")
print(f"[OK] 登录成功:{me.get('nickname') or '?'} ({me.get('user_id')})\n")
outdir = HERE / OUT_DIR
outdir.mkdir(parents=True, exist_ok=True)
csv_path = outdir / f"notes-{time.strftime('%Y-%m-%d')}.csv"
if not csv_path.exists():
with csv_path.open("w",newline="",encoding="utf-8-sig") as f:
csv.writer(f).writerow(CSV_COLS)
seen = set()
try:
with csv_path.open("r",newline="",encoding="utf-8-sig") as f:
for row in csv.DictReader(f):
if row.get("keyword") and row.get("note_id"):
seen.add((row["keyword"],row["note_id"]))
except Exception: pass
if seen: print(f"[i] 已有 {len(seen)} 条历史记录,自动去重\n")
blocked = {"v":False}
def flush(rows):
fresh = [r for r in rows if r["note_id"] and (r["keyword"],r["note_id"]) not in seen]
for r in fresh: seen.add((r["keyword"],r["note_id"]))
if fresh:
with csv_path.open("a",newline="",encoding="utf-8-sig") as f:
csv.writer(f).writerows([[r[c] for c in CSV_COLS] for r in fresh])
return len(fresh)
def run_keyword(kw):
search_id, total, shown = None,0,0
if SHOW_TOP>0:
print(f" 标题{' '*30}作者{' '*10}点赞")
print(" "+"-"*74)
for page in range(1, PAGES+1):
attempt=0
while True:
status, data = xhs.search(kw, page=page, page_size=PAGE_SIZE, sort=SORT, note_type=NOTE_TYPE, search_id=search_id)
stop, retry, msg = explain(status, data)
if not msg: break
if stop:
print(f" [X] {msg}")
blocked["v"]=True
return total
if retry and attempt<3:
attempt +=1
print(f" p{page} [!] {msg}(重试{attempt}次)")
time.sleep(3*attempt)
continue
print(f" p{page} [!] {msg}")
return total
rows, has_more, sid = extract(kw, page, data)
if sid: search_id = sid
added = flush(rows)
total += added
if SHOW_TOP>0:
for r in rows:
if shown >= SHOW_TOP:break
title = (" ".join(r["title"].split()))[:34].ljust(34)
author = r["author"][:12].ljust(12)
print(f" {title} @{author} {r['liked']}")
shown +=1
print(f" -> p{page} 共{len(rows)}条,新增{added},累计{total}", flush=True)
if not has_more or not rows or page >= PAGES: break
xhs.sleep()
return total
for kw in KEYWORDS:
if blocked["v"]:break
print(f"\n>> 搜索关键词:{kw}")
run_keyword(kw)
if not blocked["v"]: xhs.sleep()
print("\n"+"="*76)
print(f"完成,总记录数:{len(seen)}")
print(f"结果文件:{csv_path}")
if blocked["v"]:
print("\n[!] 风控拦截,任务中止!")
return 4
return 0
finally:
xhs.close()
if __name__ == "__main__":
try:
ret = main()
except KeyboardInterrupt:
print("\n已手动终止")
ret = 130
except RuntimeError as e:
print("\n"+"!"*64)
print(e)
print("!"*64)
ret=1
except Exception as e:
print("\n"+"!"*64)
print(f"[异常] {type(e).__name__}: {e}")
traceback.print_exc()
print("!"*64)
ret=1
# 右键运行暂停窗口
try:
input("\n按回车键关闭窗口…")
except:
pass
sys.exit(ret)