tradingSystem/app/services/company_report.py

132 lines
5.5 KiB
Python
Raw Normal View History

# -*- coding: utf-8 -*-
"""个股深度评析报告的代取 (2026-09-18)。
报告由数据基座每晚生成, 上游计划与逻辑状态里只带一条链接 (company_review.report_url),
链接打开是一份 markdown 纯文本原来页面把链接开成新标签, 浏览器按纯文本显示,
用户看到的是一屏 markdown 源码现在由持仓系统后端代取一次, 页面拿到正文后在自己的
抽屉里排版, 与整页同一套样式
只做三件事:
, 按代码找链接持仓票从逻辑状态映射取, 候选票从当日计划主榜与观察档取,
/api/research 的公司评析来源一致链接只认上游给的, 不接受页面传入的任意地址
, 带超时拉取正文
, 按链接缓存十分钟报告每晚生成一次, 盘中不会变; 点开抽屉不该每次都去打数据基座
取不到就把原因写进 error 返回, 不抛异常 (页面按失败态显示并给重试)
"""
from __future__ import annotations
import logging
import time
import requests
logger = logging.getLogger(__name__)
CACHE_SEC = 600 # 正文缓存秒数
TIMEOUT_SEC = 8 # 拉取超时
MAX_BYTES = 2_000_000 # 超过这个长度当异常处理, 不往页面塞
SRC_HELD, SRC_CAND, SRC_NONE = "held", "candidate", "none"
_cache: dict = {} # url -> {"at": 时刻, "text": 正文}
def _url_of(cr) -> str | None:
if not isinstance(cr, dict):
return None
u = cr.get("report_url")
u = str(u).strip() if u else ""
return u if u.startswith(("http://", "https://")) else None
def report_url_for(code: str, *, state_map=None, plan=None) -> tuple[str | None, str]:
"""返回 (链接, 来源)。来源: held (持仓逻辑状态) / candidate (当日计划) / none。
state_map plan 可注入 (单测用); 缺省时现取, 任一路取失败只记日志不抛"""
if state_map is None:
try:
from app.services import logic_state_service
state_map = logic_state_service.state_map()
except Exception as e: # noqa: BLE001
logger.warning("[评析报告] 取持仓逻辑状态失败 %s: %s", code, e)
state_map = {}
st = (state_map or {}).get(code)
u = _url_of((st or {}).get("company_review")) if isinstance(st, dict) else None
if u:
return u, SRC_HELD
if plan is None:
try:
from app.services import plan_feed
plan = plan_feed.get_plan()
except Exception as e: # noqa: BLE001
logger.warning("[评析报告] 取候选计划失败 %s: %s", code, e)
plan = {}
for r in (plan or {}).get("main") or []:
if r.get("ts_code") == code:
u = _url_of(r.get("company_review"))
if u:
return u, SRC_CAND
for r in (plan or {}).get("observe") or []:
if r.get("ts_code") == code:
u = _url_of(r.get("company_review"))
if u:
return u, SRC_CAND
return None, SRC_NONE
def _http_get_text(url: str, timeout: int) -> str:
r = requests.get(url, timeout=timeout)
r.raise_for_status()
if len(r.content) > MAX_BYTES:
raise ValueError(f"报告过大 ({len(r.content)} 字节)")
r.encoding = r.encoding or "utf-8"
return r.text
def fetch_text(url: str, *, now=None, timeout: int = TIMEOUT_SEC, getter=None) -> tuple[str, bool]:
"""取正文, 返回 (正文, 是否命中缓存)。失败抛异常, 由 get 兜成 error。"""
now = time.time() if now is None else now
hit = _cache.get(url)
if hit and now - hit["at"] < CACHE_SEC:
return hit["text"], True
text = (getter or _http_get_text)(url, timeout)
if not isinstance(text, str) or not text.strip():
raise ValueError("报告内容为空")
_cache[url] = {"at": now, "text": text}
return text, False
def _fail_why(e: Exception, timeout: int) -> str:
if isinstance(e, requests.exceptions.Timeout):
return f"数据基座没有回应,等了 {timeout}"
if isinstance(e, requests.exceptions.ConnectionError):
return "连不上数据基座"
if isinstance(e, requests.exceptions.HTTPError):
code = getattr(getattr(e, "response", None), "status_code", None)
return f"数据基座回了 {code}" if code else f"数据基座回了错误: {e}"
return f"{type(e).__name__}: {e}"
def get(code: str, *, now=None, timeout: int = TIMEOUT_SEC, getter=None, state_map=None, plan=None) -> dict:
"""页面用的一站式取法。返回固定键: ok / ts_code / url / source / markdown / cached / fetched_at / error。"""
url, src = report_url_for(code, state_map=state_map, plan=plan)
out = {"ok": False, "ts_code": code, "url": url, "source": src, "markdown": "", "cached": False,
"fetched_at": None, "error": None}
if not url:
out["error"] = ("这只票没有个股深度评析报告的链接" +
("(不在持仓也不在当日选股计划里)" if src == SRC_NONE else ""))
return out
try:
text, cached = fetch_text(url, now=now, timeout=timeout, getter=getter)
except Exception as e: # noqa: BLE001
out["error"] = f"评析报告读取失败: {_fail_why(e, timeout)}"
logger.warning("[评析报告] %s %s: %s", code, url, out["error"])
return out
at = _cache.get(url, {}).get("at")
out.update({"ok": True, "markdown": text, "cached": cached,
"fetched_at": time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(at)) if at else None})
return out
def invalidate():
_cache.clear()