# -*- coding: utf-8 -*-
"""
各平台 SSE/gRPC 响应解析器
从原始响应体中提取: answer, thinking_process, search_results, image_links, video_links, related_questions
"""
import json
import re
from datetime import datetime
VIDEO_DOMAINS = ("douyin.com", "bilibili.com", "youtube.com", "youtu.be", "ixigua.com",
"v.qq.com", "youku.com", "tv.sohu.com")
IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".gif", ".webp", ".bmp", ".svg", ".tiff")
def _extract_body_videos(text):
"""从正文中提取独立出现的视频链接 (不含搜索信源中的视频帖子)"""
if not text:
return []
seen = set()
videos = []
for m in re.finditer(r'(?:^|\s)(https?://\S+)', text, re.M):
url = m.group(1).rstrip('.,;!?)')
if any(vd in url for vd in VIDEO_DOMAINS):
if url not in seen:
seen.add(url)
videos.append(url)
# Markdown 链接中的视频
for m in re.finditer(r'\[([^\]]*)\]\((https?://[^\s\)]+)\)', text):
url = m.group(2)
if any(vd in url for vd in VIDEO_DOMAINS) and url not in seen:
seen.add(url)
videos.append(url)
return videos
def _extract_site_name(url):
"""从 URL 提取站点名 (如 autohome.com.cn → 汽车之家)"""
from urllib.parse import urlparse
try:
host = urlparse(url).hostname or ""
except:
return ""
host = host.lower()
for prefix in ("www.", "m.", "wap.", "auto.", "club.", "k.", "pad."):
if host.startswith(prefix):
host = host[len(prefix):]
break
KNOWN = {
"autohome.com.cn": "汽车之家", "dongchedi.com": "懂车帝",
"sina.cn": "新浪", "sina.com.cn": "新浪",
"sohu.com": "搜狐", "163.com": "网易", "qq.com": "腾讯",
"bilibili.com": "哔哩哔哩", "zhihu.com": "知乎",
"baidu.com": "百度", "baijiahao.baidu.com": "百家号",
"toutiao.com": "今日头条", "douyin.com": "抖音",
"yiche.com": "易车", "bitauto.com": "易车", "pcauto.com.cn": "太平洋汽车",
"xcar.com.cn": "爱卡汽车", "cheshi.com": "网上车市",
"stockstar.com": "证券之星", "36kr.com": "36氪",
"zol.com.cn": "中关村在线", "youth.cn": "中国青年网",
"thepaper.cn": "澎湃新闻", "ifeng.com": "凤凰网",
"mp.weixin.qq.com": "微信公众号", "weixin.qq.com": "微信公众号",
"jiemian.com": "界面新闻", "caixin.com": "财新",
"ithome.com": "IT之家", "itbear.com.cn": "ITBear",
}
for domain, name in KNOWN.items():
if host.endswith(domain):
return name
# 取主域名 (处理 .com.cn 等双后缀)
parts = host.split(".")
if len(parts) >= 3 and parts[-2] in ("com", "co", "org", "net", "gov", "edu"):
return parts[-3]
return parts[-2] if len(parts) >= 2 else host
def _extract_body_images(text):
"""从正文中提取图片链接 (Markdown  和 HTML
)"""
if not text:
return []
seen = set()
images = []
# Markdown:  or 
for m in re.finditer(r'!\[[^\]]*\]\((https?://[^\s\)\"]+)', text):
url = m.group(1)
if url not in seen:
seen.add(url)
images.append(url)
# HTML:
for m in re.finditer(r'
]*src=["\']?(https?://[^\s"\'>\)]+)', text, re.I):
url = m.group(1)
if url not in seen:
seen.add(url)
images.append(url)
# 独立图片URL行 (常见于AI回答中直接给出图片链接)
for m in re.finditer(r'(?:^|\s)(https?://\S+?\.(?:jpg|jpeg|png|gif|webp|bmp|svg)(?:\?\S*)?)', text, re.I):
url = m.group(1)
if url not in seen:
seen.add(url)
images.append(url)
return images
def _normalize_date(value):
"""统一各平台的发布日期为 YYYY-MM-DD 格式"""
if not value:
return ""
# Unix 时间戳 (int/float)
if isinstance(value, (int, float)):
if value <= 0:
return ""
try:
ts = value / 1000 if value > 1e10 else value
return datetime.fromtimestamp(ts).strftime("%Y-%m-%d")
except:
return ""
s = str(value).strip()
if not s or s == "0":
return ""
# ISO 8601: 2026-06-16T16:00:00Z 或 2026-06-16T16:00:00+08:00
if "T" in s:
try:
from datetime import timezone, timedelta
s_clean = s.replace("Z", "+00:00")
dt = datetime.fromisoformat(s_clean)
# 转为北京时间 (UTC+8)
dt_bj = dt.astimezone(timezone(timedelta(hours=8)))
return dt_bj.strftime("%Y-%m-%d")
except:
pass
# 纯数字字符串 (时间戳)
if s.replace(".", "").isdigit():
try:
ts = float(s)
if ts > 1e10:
ts = ts / 1000
return datetime.fromtimestamp(ts).strftime("%Y-%m-%d")
except:
return ""
# 已经是 YYYY-MM-DD
if re.match(r"^\d{4}-\d{2}-\d{2}$", s):
return s
# 中文日期: 2026年07月01日
m = re.match(r"^(\d{4})年(\d{1,2})月(\d{1,2})日", s)
if m:
return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
# 尝试通用解析
try:
from dateutil.parser import parse as dateparse
return dateparse(s).strftime("%Y-%m-%d")
except:
return s
# ============================================================
# 编码修复
# ============================================================
def _fix_doubao_text(text):
"""豆包 HAR 编码修复: UTF-8 字节被按 CP1252/Latin-1 解码"""
if not text or all(ord(c) < 128 for c in text):
return text
raw = bytearray()
for ch in text:
cp = ord(ch)
if cp < 256:
raw.append(cp)
else:
try:
raw.extend(ch.encode("cp1252"))
except UnicodeEncodeError:
raw.extend(ch.encode("utf-8"))
try:
return bytes(raw).decode("utf-8")
except UnicodeDecodeError:
return bytes(raw).decode("utf-8", errors="replace")
# ============================================================
# DeepSeek: SSE 流式, fragment type 区分 THINK/RESPONSE
# ============================================================
def parse_deepseek(body, req_body=None):
"""
接口: POST https://chat.deepseek.com/api/v0/chat/completion
认证: Bearer Token + PoW签名 + hif-dliq/hif-leim
响应: SSE data:{json}
关键字段:
- 搜索结果: p="response/fragments/-1/results", v=array
- 回答文本: v=string (属于 RESPONSE fragment)
- 思考过程: v=string (属于 THINK fragment)
- 引用: answer 中 [citation:N] 对应 search_results 的 cite_index
"""
fragment_types = {}
current_fragment_id = None
current_fragment_type = None
think_parts = []
answer_parts = []
search_results = []
def _register_fragments(frag_list):
nonlocal current_fragment_id, current_fragment_type
for frag in frag_list:
ftype = frag.get("type", "")
fid = frag.get("id")
if fid is not None and ftype:
fragment_types[fid] = ftype
current_fragment_id = fid
current_fragment_type = ftype
content = frag.get("content")
if content and isinstance(content, str):
(think_parts if ftype == "THINK" else answer_parts).append(content)
for line in body.split("\n"):
line = line.strip()
if not line or line.startswith("event:"):
continue
ds = line[6:] if line.startswith("data: ") else (line[5:] if line.startswith("data:") else None)
if not ds or ds == "[DONE]":
continue
try:
data = json.loads(ds)
except json.JSONDecodeError:
continue
p = data.get("p", "")
o = data.get("o", "")
v = data.get("v")
# 搜索结果
if "results" in p and isinstance(v, list) and o in ("SET", ""):
if v and isinstance(v[0], dict) and "url" in v[0]:
for item in v:
search_results.append({
"source_url": item.get("url", ""),
"source_name": item.get("site_name", ""),
"source_title": item.get("title", ""),
"source_publish_date": _normalize_date(item.get("published_at", "")),
"source_icon": item.get("site_icon", ""),
"cite_index": item.get("cite_index"),
})
continue
# Fragment 注册 - 3 种模式
if o == "BATCH" and isinstance(v, list):
for item in v:
if isinstance(item, dict) and item.get("o") == "APPEND" and "fragments" in item.get("p", ""):
_register_fragments(item.get("v", []))
continue
if p.endswith("/fragments") and o == "APPEND" and isinstance(v, list):
_register_fragments(v)
continue
if isinstance(v, dict) and "response" in v:
_register_fragments(v["response"].get("fragments", []))
continue
# 路径模式: response/thinking_content 和 response/content 切换思考/回答
if p == "response/thinking_content":
current_fragment_type = "THINK"
if isinstance(v, str) and v:
think_parts.append(v)
continue
if p == "response/content":
current_fragment_type = "RESPONSE"
if isinstance(v, str) and v:
answer_parts.append(v)
continue
# 跳过非内容事件
if any(k in p for k in ("elapsed_secs", "status", "accumulated_token", "quasi_status",
"conversation_mode", "has_pending_fragment", "search_triggered",
"thinking_elapsed", "tips")):
continue
# 内容追加 (fragment 模式)
if p.endswith("/content") and o == "APPEND" and isinstance(v, str):
(think_parts if current_fragment_type == "THINK" else answer_parts).append(v)
continue
# 独立文本事件
if isinstance(v, str) and not o:
(think_parts if current_fragment_type == "THINK" else answer_parts).append(v)
full_answer = "".join(answer_parts)
full_think = "".join(think_parts)
# 引用匹配
citations = set(int(c) for c in re.findall(r"citation:(\d+)", full_answer))
for sr in search_results:
sr["is_referenced"] = sr.get("cite_index") in citations
# 请求体解析
question, deep_thinking, is_online = "", False, True
if req_body:
try:
req = json.loads(req_body)
question = req.get("prompt", "")
deep_thinking = req.get("thinking_enabled", False)
is_online = req.get("search_enabled", True)
except:
pass
return {
"platform": "ds", "platform_name": "DeepSeek",
"question": question, "deep_thinking": 1 if deep_thinking else 0,
"is_online": is_online,
"result": {
"answer": full_answer, "thinking_process": full_think,
"search_links": [sr["source_url"] for sr in search_results],
"cited_links": [sr["source_url"] for sr in search_results if sr["is_referenced"]],
"image_links": _extract_body_images(full_answer),
"video_links": _extract_body_videos(full_answer), "related_questions": [],
},
"sources": search_results,
}
# ============================================================
# Kimi: gRPC-Web Connect 二进制帧封装 JSON
# ============================================================
def parse_kimi(body, req_body=None):
"""
接口: POST https://www.kimi.com/apiv2/kimi.gateway.chat.v1.ChatService/Chat
认证: Bearer JWT + x-msh-device-id/session-id/traffic-id
响应: 二进制帧 (\\x00 + 4字节长度 + JSON), 多帧拼接
关键字段:
- 回答: block.text.content, op=set/append
- 搜索: message.refs.searchChunks[].base / message.references[].items[].search.base
- 图片: base.coverUrl (信源封面图)
"""
raw = body.encode("utf-8", errors="replace") if isinstance(body, str) else body
frames = []
pos = 0
while pos < len(raw):
if raw[pos:pos + 1] == b"\x00":
pos += 1
continue
brace = raw.find(b"{", pos)
if brace == -1:
break
pos = brace
depth, end, in_str, escape = 0, pos, False, False
for i in range(pos, min(pos + 500000, len(raw))):
c = raw[i:i + 1]
if escape:
escape = False
continue
if c == b"\\" and in_str:
escape = True
continue
if c == b'"':
in_str = not in_str
continue
if in_str:
continue
if c == b"{":
depth += 1
elif c == b"}":
depth -= 1
if depth == 0:
end = i
break
chunk = raw[pos:end + 1].decode("utf-8", errors="replace")
try:
frames.append(json.loads(chunk))
except json.JSONDecodeError:
pass
pos = end + 1
# 提取回答 + 思考过程
blocks = {}
think_blocks = {}
think_summary = ""
for f in frames:
block = f.get("block")
if not block:
continue
bid = block.get("id", "default")
op = f.get("op", "")
mask = f.get("mask", "")
# 正文
if "text" in mask and "think" not in mask:
content = block.get("text", {}).get("content", "")
if op == "set":
blocks[bid] = content
elif op == "append":
blocks[bid] = blocks.get(bid, "") + content
# 思考过程 (K3: block.think.content)
elif "think" in mask:
think_content = block.get("think", {}).get("content", "")
if op == "set" and "summary" in mask:
think_summary = block.get("think", {}).get("summary", "")
elif op == "set":
think_blocks[bid] = think_content
elif op == "append":
think_blocks[bid] = think_blocks.get(bid, "") + think_content
full_answer = "\n\n".join(v for v in blocks.values() if v)
full_thinking = "\n\n".join(v for v in think_blocks.values() if v)
# 提取搜索结果
seen = {}
video_links = []
_cited_urls = set()
for f in frames:
mask = f.get("mask", "")
if mask == "message.refs.searchChunks":
for chunk in f.get("message", {}).get("refs", {}).get("searchChunks", []):
base = chunk.get("base", {})
src_url = base.get("url", "")
if src_url and src_url not in seen:
seen[src_url] = {
"source_url": src_url,
"source_name": base.get("siteName", "") or _extract_site_name(src_url),
"source_title": base.get("title", ""),
"source_publish_date": _normalize_date(base.get("publishTime", "")),
"source_icon": base.get("iconUrl", "") or base.get("coverUrl", ""),
"is_referenced": False,
}
elif mask == "message.references":
for ref in f.get("message", {}).get("references", []):
for item in ref.get("items", []):
s = item.get("search", {})
base = s.get("base", {})
url = base.get("url", "")
if url:
_cited_urls.add(url)
if url not in seen:
seen[url] = {
"source_url": url,
"source_name": base.get("siteName", "") or _extract_site_name(url),
"source_title": base.get("title", ""),
"source_publish_date": _normalize_date(base.get("publishTime", "")),
"source_icon": "", "is_referenced": True,
}
# 用 references 中的 URL 回标 searchChunks 中的 is_referenced
for sr in seen.values():
if sr["source_url"] in _cited_urls:
sr["is_referenced"] = True
# 提取 chat_id + message_ids (用于分享链接)
chat_id = ""
message_ids = []
for f in frames:
chat = f.get("chat", {})
if isinstance(chat, dict) and chat.get("id"):
chat_id = chat["id"]
msg = f.get("message", {})
if isinstance(msg, dict) and msg.get("id") and msg.get("role") in ("user", "assistant"):
mid = msg["id"]
if mid not in message_ids:
message_ids.append(mid)
# 从请求体或帧中提取问题
question = ""
if req_body:
try:
idx = req_body.find("{")
if idx >= 0:
req = json.loads(req_body[idx:])
for b in req.get("message", {}).get("blocks", []):
question = b.get("text", {}).get("content", question)
except:
pass
if not question:
for f in frames:
msg = f.get("message", {})
if msg.get("role") == "user":
for b in msg.get("blocks", []):
q = b.get("text", {}).get("content", "")
if q:
question = q
return {
"platform": "kimi", "platform_name": "Kimi",
"question": question, "deep_thinking": 0, "is_online": True,
"chat_id": chat_id,
"message_ids": message_ids,
"result": {
"answer": full_answer, "thinking_process": full_thinking,
"search_links": [sr["source_url"] for sr in seen.values()],
"cited_links": [sr["source_url"] for sr in seen.values() if sr["is_referenced"]],
"image_links": _extract_body_images(full_answer),
"video_links": _extract_body_videos(full_answer), "related_questions": [],
},
"sources": list(seen.values()),
}
# ============================================================
# 千问: SSE 流式, mime_type 区分事件类型
# ============================================================
def parse_qianwen(body, req_body=None):
"""
接口: POST https://chat2.qianwen.com/api/v2/chat
认证: HMAC-SHA1 Query签名 (无需cookie), key=af54041a93cd4f6a757f
响应: SSE data:{json}, 消息在 data.messages[] 中
关键字段:
- 回答: mime_type=multi_load/iframe → content (全量替换)
- 思考: mime_type=plan_cot/post → content
- 搜索进度: mime_type=bar/progress → meta_data.list[]
- 信源: multi_load/iframe → meta_data.multi_load[] type=source_group_web
- 追问: mime_type=paa/iframe (未解析)
"""
full_content = ""
think_content = ""
_plan_cot = ""
_deep_think = ""
search_results = []
seen_urls = set()
_think_segs = {}
video_links = []
image_links = []
_seen_video_urls = set()
for line in body.split("\n"):
line = line.strip()
if not line.startswith("data:"):
continue
ds = line[5:].strip()
if ds == "[DONE]":
break
try:
data_json = json.loads(ds)
except json.JSONDecodeError:
continue
for msg in (data_json.get("data") or {}).get("messages", []):
mime_type = msg.get("mime_type", "")
content = msg.get("content", "")
meta = msg.get("meta_data", {})
if mime_type == "multi_load/iframe":
cleaned = re.sub(r"\[\((?:deep_think|multimodal_chat_think|video_note_list)[_\d]*\)\]", "", content).strip() if isinstance(content, str) else ""
if cleaned:
full_content = cleaned
for ml in (meta.get("multi_load") or []):
ml_type = ml.get("type", "")
ml_content = ml.get("content", {})
if ml_type == "source_group_web":
for sg in (ml_content.get("list") if isinstance(ml_content, dict) else []) or []:
for s in (sg.get("content", {}).get("list") or []):
url = s.get("url", "")
if url and url not in seen_urls:
seen_urls.add(url)
search_results.append({
"source_url": url,
"source_name": s.get("name", ""),
"source_title": s.get("title", ""),
"source_publish_date": _normalize_date(s.get("publish_time", "")),
"source_icon": s.get("icon", ""),
"is_referenced": False,
})
elif ml_type == "multimodal_chat_think" and isinstance(ml_content, dict):
tc = ml_content.get("think_content", "")
if tc and len(tc) > len(think_content):
think_content = tc
elif ml_type == "deep_think" and isinstance(ml_content, dict):
tc = ml_content.get("think_content", "")
if tc and len(tc) > len(_deep_think):
_deep_think = tc
elif ml_type == "ref_source_inline" and isinstance(ml_content, dict):
for doc in (ml_content.get("docs") or []):
url = doc.get("url", "") or doc.get("raw_url", "")
if url and url not in seen_urls:
seen_urls.add(url)
search_results.append({
"source_url": url,
"source_name": doc.get("host_name", "") or doc.get("name", "") or _extract_site_name(url),
"source_title": doc.get("title", ""),
"source_publish_date": _normalize_date(doc.get("publish_time", "")),
"source_icon": doc.get("icon", ""),
"is_referenced": False,
})
elif ml_type == "video_note_list" and isinstance(ml_content, dict):
for vitem in (ml_content.get("list") or []):
if vitem.get("subtype") == "video":
vurl = vitem.get("url", "")
if vurl and vurl not in _seen_video_urls:
_seen_video_urls.add(vurl)
video_links.append(vurl)
cover = vitem.get("cover", "")
if cover and cover not in _seen_video_urls:
_seen_video_urls.add(cover)
image_links.append(cover)
elif mime_type == "plan_cot/post":
if isinstance(content, str) and len(content) > len(_plan_cot):
_plan_cot = content
elif mime_type == "bar/progress":
for item in (meta.get("list") or []):
url = item.get("url", "")
if url and url not in seen_urls:
seen_urls.add(url)
search_results.append({
"source_url": url,
"source_name": item.get("host_name", "") or item.get("name", "") or _extract_site_name(url),
"source_title": item.get("title", ""),
"source_publish_date": _normalize_date(item.get("publish_time", "")),
"source_icon": "", "is_referenced": False,
})
elif mime_type == "bar/workflow":
for ml in (meta.get("multi_load") or []):
ml_type = ml.get("type", "")
ml_content = ml.get("content", {})
if ml_type == "bar_thinking" and isinstance(ml_content, dict):
body = ml_content.get("body", "")
title = ml_content.get("title", "")
seq = ml.get("source_seq", "")
if body:
_think_segs[seq] = f"[{title}] {body}" if title else body
elif ml_type == "bar_ref_source_inline" and isinstance(ml_content, dict):
for doc in (ml_content.get("docs") or []):
url = doc.get("url", "") or doc.get("raw_url", "")
if url and url not in seen_urls:
seen_urls.add(url)
search_results.append({
"source_url": url,
"source_name": doc.get("host_name", "") or doc.get("name", "") or _extract_site_name(url),
"source_title": doc.get("title", ""),
"source_publish_date": _normalize_date(doc.get("publish_time", "")),
"source_icon": doc.get("icon", ""),
"is_referenced": False,
})
# 合并 bar/workflow 思考段落 (深度思考模式)
if _think_segs and not think_content:
think_content = "\n".join(_think_segs[k] for k in sorted(_think_segs))
# 合并 plan_cot + deep_think (两者互补: plan_cot 是概要, deep_think 是详细推理)
if _plan_cot and _deep_think:
think_content = _plan_cot + "\n" + _deep_think
elif _deep_think:
think_content = _deep_think
elif _plan_cot and not think_content:
think_content = _plan_cot
question, deep_thinking = "", False
if req_body:
try:
req = json.loads(req_body)
for m in req.get("messages", []):
question = m.get("content", question)
deep_thinking = req.get("deep_search") == "1"
except:
pass
return {
"platform": "qianwen", "platform_name": "通义千问",
"question": question, "deep_thinking": 1 if deep_thinking else 0,
"is_online": True,
"result": {
"answer": full_content, "thinking_process": think_content,
"search_links": [sr["source_url"] for sr in search_results],
"cited_links": [sr["source_url"] for sr in search_results if sr["is_referenced"]],
"image_links": image_links + [u for u in _extract_body_images(full_content) if u not in _seen_video_urls],
"video_links": video_links + [u for u in _extract_body_videos(full_content) if u not in _seen_video_urls],
"related_questions": [],
},
"sources": search_results,
}
# ============================================================
# 豆包: SSE 流式, event type 区分事件
# ============================================================
def parse_doubao(body, req_body=None):
"""
接口: POST https://www.doubao.com/chat/completion?aid=497858&...
认证: 浏览器Cookie (手动登录), 关键cookie: s_v_web_id
响应: SSE event:{type}\\ndata:{json}
关键字段:
- STREAM_CHUNK patch_object=1: 内容块 (text_block/search_query_result_block)
- STREAM_CHUNK patch_object=50: 元数据 (ext.sp_v2 = "你可能还想问")
- CHUNK_DELTA: 文本增量
- block_type=10040: 思考块(thinking_block), 其子块 parent_id 匹配
- search_query_result_block.results[].text_card: 信源 (含 logo_url 图片, 抖音视频链接)
- SSE_REPLY_END: 回复结束
- error_code 710022004: 验证码; 710022022/710022013: 强制登录
"""
body = _fix_doubao_text(body)
content_blocks = {}
search_results = []
seen_urls = set()
thinking_block_id = None
last_active_block_id = None
related_questions = []
video_links = []
image_links = []
_seen_image_uris = set()
_cited_urls = set()
conversation_id = ""
# 拆分 SSE 事件
events = []
lines_buf = []
for line in body.split("\n"):
if not line.strip():
if lines_buf:
events.append(lines_buf)
lines_buf = []
elif not line.startswith(":"):
lines_buf.append(line)
if lines_buf:
events.append(lines_buf)
for evt_lines in events:
event_type = ""
data_parts = []
for line in evt_lines:
colon = line.find(":")
if colon < 0:
continue
name = line[:colon]
value = line[colon + 1:].lstrip()
if name == "event":
event_type = value.strip()
elif name == "data":
data_parts.append(value)
data_str = "\n".join(data_parts) if data_parts else ""
if not data_str or event_type in ("SSE_HEARTBEAT", "SSE_REPLY_END", "FULL_MSG_NOTIFY"):
continue
# 从 SSE_ACK 提取 conversation_id (用于分享链接)
if event_type == "SSE_ACK":
try:
ack_data = json.loads(data_str)
cid = (ack_data.get("ack_client_meta") or {}).get("conversation_id", "")
if cid:
conversation_id = str(cid)
except:
pass
continue
try:
data = json.loads(data_str)
except json.JSONDecodeError:
continue
if "error_code" in data:
continue
if event_type == "STREAM_MSG_NOTIFY":
for blk in (data.get("content") or {}).get("content_block", []):
if blk.get("block_type") == 10040 and "thinking_block" in (blk.get("content") or {}):
thinking_block_id = blk.get("block_id")
elif event_type == "STREAM_CHUNK":
for op in data.get("patch_op", []):
patch_obj = op.get("patch_object")
if patch_obj == 50:
sp_v2 = (op.get("patch_value") or {}).get("ext", {}).get("sp_v2", "")
if sp_v2:
try:
related_questions = [s.get("content", "").strip() for s in json.loads(sp_v2) if s.get("content")]
except:
pass
continue
if patch_obj != 1:
continue
for blk in (op.get("patch_value") or {}).get("content_block", []):
b_id = blk.get("block_id")
b_content = blk.get("content") or {}
parent_id = blk.get("parent_id", "")
patch_type = op.get("patch_type")
if blk.get("block_type") == 10040 and "thinking_block" in b_content:
if not thinking_block_id:
thinking_block_id = b_id
continue
# 从 meta_info.tag_info 提取正文图片 + 采用链接
for mi in (blk.get("meta_info") or []):
tag_str = mi.get("tag_info", "")
if not tag_str:
continue
try:
tag = json.loads(tag_str)
except (json.JSONDecodeError, TypeError):
continue
mi_type = mi.get("type", 0)
if mi_type == 202 and tag.get("media"):
for media in tag["media"]:
img = media.get("image", {})
thumb = img.get("thumb_url", "") or img.get("origin_url", "")
uri = img.get("uri", "")
if thumb and uri and uri not in _seen_image_uris:
_seen_image_uris.add(uri)
image_links.append(thumb)
elif mi_type == 2 and tag.get("url"):
cite_url = tag["url"].split("?")[0]
if cite_url not in _cited_urls:
_cited_urls.add(cite_url)
content = ""
if "text_block" in b_content:
content = b_content["text_block"].get("text", "")
elif "code_block" in b_content:
content = b_content["code_block"].get("code", "")
elif "search_query_result_block" in b_content:
if blk.get("is_finish"):
for res in b_content.get("search_query_result_block", {}).get("results", []):
card = res.get("text_card") or res
link = card.get("url") or card.get("link", "")
if link and link not in seen_urls:
seen_urls.add(link)
logo = card.get("logo_url") or card.get("logo_uri") or ""
pub_time = _normalize_date(card.get("publish_time_second", ""))
search_results.append({
"source_url": link,
"source_name": card.get("sitename", "") or card.get("source", ""),
"source_title": card.get("title", ""),
"source_publish_date": pub_time,
"source_icon": logo, "is_referenced": False,
})
if patch_type == 1:
content_blocks[b_id] = content_blocks.get(b_id, "") + content
elif patch_type == 2:
content_blocks[b_id] = content
last_active_block_id = b_id
if parent_id and parent_id == thinking_block_id:
content_blocks[b_id + "_type"] = "think"
elif "search_query_result_block" in b_content:
content_blocks[b_id + "_type"] = "search"
elif event_type == "CHUNK_DELTA":
text = data.get("text", "")
if text and last_active_block_id:
content_blocks[last_active_block_id] = content_blocks.get(last_active_block_id, "") + text
answer_text, think_text = "", ""
for k in sorted(k for k in content_blocks if not k.endswith("_type")):
block_type = content_blocks.get(k + "_type", "normal")
if block_type == "think":
think_text += content_blocks[k]
elif block_type != "search":
answer_text += content_blocks[k]
# 用 _cited_urls 回标搜索结果中的引用
if _cited_urls:
for sr in search_results:
if sr["source_url"].split("?")[0] in _cited_urls:
sr["is_referenced"] = True
question, deep_thinking = "", 0
if req_body:
try:
req = json.loads(req_body)
for msg in req.get("messages", []):
for cb in msg.get("content_block", []):
t = (cb.get("content") or {}).get("text_block", {}).get("text", "")
if t:
question = t
dt = req.get("option", {}).get("need_deep_think", 0)
deep_thinking = 1 if dt and dt != 0 else 0
except:
pass
return {
"platform": "doubao", "platform_name": "豆包",
"question": question, "deep_thinking": deep_thinking, "is_online": True,
"conversation_id": conversation_id,
"result": {
"answer": answer_text, "thinking_process": think_text,
"search_links": [sr["source_url"] for sr in search_results],
"cited_links": [sr["source_url"] for sr in search_results if sr["is_referenced"]],
"image_links": image_links + _extract_body_images(answer_text),
"video_links": _extract_body_videos(answer_text),
"related_questions": related_questions,
},
"sources": search_results,
}
# ============================================================
# 文心一言: SSE 流式, component 区分事件类型
# ============================================================
def parse_wenxin(body, req_body=None):
"""
接口: POST https://chat.baidu.com/aichat/api/conversation
认证: 浏览器Cookie + token (从页面HTML提取)
响应: SSE event:{type}\\ndata:{json}
关键字段:
- 回答: component=markdown-yiyan → data.value (全量替换)
- 思考: component=thinkingSteps → data.reasoningContentArr
- 信源: component=note-list → data.list[]
- 完成: metaData.state=generate-complete && endTurn
"""
full_content = ""
think_content = ""
search_results = []
seen_urls = set()
video_links = []
image_links = []
conversation_lid = ""
for block in body.split("\n\n"):
if not block.strip():
continue
event_type = ""
data_str = ""
for line in block.split("\n"):
if line.startswith("event:"):
event_type = line[6:].strip()
elif line.startswith("data:"):
data_str = line[5:].strip()
if not data_str:
continue
# 从 basedata 事件提取 lid (用于构造分享链接)
if event_type == "basedata":
try:
bd = json.loads(data_str)
conversation_lid = bd.get("lid", "")
except:
pass
continue
if event_type != "message":
continue
try:
data = json.loads(data_str)
except json.JSONDecodeError:
continue
msg = data.get("data", {}).get("message", {})
meta = msg.get("metaData", {})
gen = msg.get("content", {}).get("generator", {})
component = gen.get("component", "")
gen_data = gen.get("data", {}) if isinstance(gen.get("data"), dict) else {}
if component == "markdown-yiyan":
value = gen_data.get("value", "")
if value:
full_content += value
elif component == "thinkingSteps":
reasoning = gen_data.get("reasoningContentArr", [])
if reasoning:
chunk = "".join(reasoning)
if chunk:
think_content += chunk
for ref in (gen_data.get("referenceList") or []):
url = ref.get("url", "")
if url and url not in seen_urls:
seen_urls.add(url)
search_results.append({
"source_url": url,
"source_name": ref.get("source", "") or ref.get("author_name", ""),
"source_title": ref.get("text", "") or ref.get("abstract", "")[:60],
"source_publish_date": "",
"source_icon": ref.get("icon", ""),
"is_referenced": False,
})
elif component == "note-list":
for item in (gen_data.get("items") or gen_data.get("list") or []):
link_info = item.get("linkInfo", {}) if isinstance(item.get("linkInfo"), dict) else {}
if item.get("isVideo") or link_info.get("type") == "video":
vid_url = link_info.get("href", "")
if vid_url and vid_url not in seen_urls:
seen_urls.add(vid_url)
video_links.append(vid_url)
thumb = item.get("thumbnail", {}) if isinstance(item.get("thumbnail"), dict) else {}
thumb_src = thumb.get("src", "")
if thumb_src and thumb_src not in seen_urls:
seen_urls.add(thumb_src)
image_links.append(thumb_src)
elif component == "imageScroll":
for item in (gen_data.get("items") or []):
img_url = item.get("thumbUrl", "") or item.get("originUrl", "")
if img_url and img_url not in seen_urls:
seen_urls.add(img_url)
image_links.append(img_url)
elif component == "videoScroll":
for item in (gen_data.get("items") or []):
vid_url = item.get("url", "") or item.get("src", "")
if vid_url and vid_url not in seen_urls:
seen_urls.add(vid_url)
video_links.append(vid_url)
poster = item.get("poster", "")
if poster and poster not in seen_urls:
seen_urls.add(poster)
image_links.append(poster)
# 清理开头的乱码字符 (服务端偶发的 U+FFFD)
full_content = full_content.replace("�", "")
think_content = think_content.replace("�", "")
question = ""
deep_thinking = False
if req_body:
try:
req = json.loads(req_body)
for q in req.get("message", {}).get("query", []):
if q.get("type") == "TEXT":
question = q.get("data", {}).get("text", {}).get("query", "")
ds = req.get("message", {}).get("searchInfo", {}).get("usedModel", {}).get("modelFunction", {}).get("deepSearch", "0")
deep_thinking = ds == "1"
except:
pass
return {
"platform": "wenxin", "platform_name": "文心一言",
"question": question, "deep_thinking": 1 if deep_thinking else 0,
"is_online": True,
"conversation_lid": conversation_lid,
"result": {
"answer": full_content, "thinking_process": think_content,
"search_links": [sr["source_url"] for sr in search_results],
"cited_links": [sr["source_url"] for sr in search_results if sr["is_referenced"]],
"image_links": image_links + _extract_body_images(full_content),
"video_links": video_links + [u for u in _extract_body_videos(full_content) if u not in seen_urls],
"related_questions": [],
},
"sources": search_results,
}
# 解析器注册表
PARSERS = {
"ds": parse_deepseek,
"kimi": parse_kimi,
"qianwen": parse_qianwen,
"doubao": parse_doubao,
"wenxin": parse_wenxin,
}