mirror of
https://github.com/ZhuLinsen/daily_stock_analysis.git
synced 2026-10-06 12:03:07 +08:00
feat: add research artifact contract (#2291)
* feat: add research artifact contract * fix: preserve research artifact identity and falsey values * fix: correct research artifact evidence quality
This commit is contained in:
@@ -15,6 +15,15 @@ from api.v1.schemas.common import (
|
||||
SuccessResponse,
|
||||
)
|
||||
from api.v1.schemas.market_phase import MarketPhaseSummary
|
||||
from api.v1.schemas.research_artifact import (
|
||||
ResearchArtifact,
|
||||
ResearchDataQuality,
|
||||
ResearchEvidenceItem,
|
||||
ResearchInvalidationCondition,
|
||||
ResearchNextAction,
|
||||
ResearchSubject,
|
||||
ResearchThesis,
|
||||
)
|
||||
from api.v1.schemas.analysis import (
|
||||
AnalyzeRequest,
|
||||
AnalysisResultResponse,
|
||||
@@ -145,6 +154,14 @@ __all__ = [
|
||||
"SuccessResponse",
|
||||
# market phase
|
||||
"MarketPhaseSummary",
|
||||
# research artifact
|
||||
"ResearchArtifact",
|
||||
"ResearchDataQuality",
|
||||
"ResearchEvidenceItem",
|
||||
"ResearchInvalidationCondition",
|
||||
"ResearchNextAction",
|
||||
"ResearchSubject",
|
||||
"ResearchThesis",
|
||||
# analysis
|
||||
"AnalyzeRequest",
|
||||
"AnalysisResultResponse",
|
||||
|
||||
@@ -14,6 +14,7 @@ from typing import Optional, List, Any, Dict, Literal
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
|
||||
from api.v1.schemas.market_phase import MarketPhaseSummary
|
||||
from api.v1.schemas.research_artifact import ResearchArtifact
|
||||
from src.schemas.decision_action import DecisionAction
|
||||
|
||||
|
||||
@@ -51,7 +52,7 @@ class HistoryItem(BaseModel):
|
||||
description="本次分析市场阶段低敏摘要",
|
||||
)
|
||||
created_at: Optional[str] = Field(None, description="创建时间")
|
||||
|
||||
|
||||
model_config = ConfigDict(json_schema_extra={
|
||||
"example": {
|
||||
"id": 1234,
|
||||
@@ -68,12 +69,12 @@ class HistoryItem(BaseModel):
|
||||
|
||||
class HistoryListResponse(BaseModel):
|
||||
"""历史记录列表响应"""
|
||||
|
||||
|
||||
total: int = Field(..., description="总记录数")
|
||||
page: int = Field(..., description="当前页码")
|
||||
limit: int = Field(..., description="每页数量")
|
||||
items: List[HistoryItem] = Field(default_factory=list, description="记录列表")
|
||||
|
||||
|
||||
model_config = ConfigDict(json_schema_extra={
|
||||
"example": {
|
||||
"total": 100,
|
||||
@@ -152,7 +153,7 @@ class ReportMeta(BaseModel):
|
||||
|
||||
class ReportSummary(BaseModel):
|
||||
"""报告概览区"""
|
||||
|
||||
|
||||
analysis_summary: Optional[str] = Field(None, description="关键结论")
|
||||
operation_advice: Optional[str] = Field(None, description="操作建议")
|
||||
action: Optional[DecisionAction] = Field(None, description="结构化建议动作 taxonomy")
|
||||
@@ -167,7 +168,7 @@ class ReportSummary(BaseModel):
|
||||
|
||||
class ReportStrategy(BaseModel):
|
||||
"""策略点位区"""
|
||||
|
||||
|
||||
ideal_buy: Optional[str] = Field(None, description="理想买入价")
|
||||
secondary_buy: Optional[str] = Field(None, description="第二买入价")
|
||||
stop_loss: Optional[str] = Field(None, description="止损价")
|
||||
@@ -252,7 +253,7 @@ class AnalysisContextPackOverview(BaseModel):
|
||||
|
||||
class ReportDetails(BaseModel):
|
||||
"""报告详情区"""
|
||||
|
||||
|
||||
news_content: Optional[str] = Field(None, description="新闻摘要")
|
||||
empty_news_disclosure: Optional[str] = Field(
|
||||
None,
|
||||
@@ -301,6 +302,10 @@ class AnalysisReport(BaseModel):
|
||||
summary: ReportSummary = Field(..., description="概览区")
|
||||
strategy: Optional[ReportStrategy] = Field(None, description="策略点位区")
|
||||
details: Optional[ReportDetails] = Field(None, description="详情区")
|
||||
structured_report: Optional[ResearchArtifact] = Field(
|
||||
None,
|
||||
description="结构化研究产物,供看板、个股研究页、监控和 Copilot 复用",
|
||||
)
|
||||
|
||||
model_config = ConfigDict(json_schema_extra={
|
||||
"example": {
|
||||
|
||||
@@ -0,0 +1,115 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Structured research artifact schemas."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Literal, Optional
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from src.schemas.decision_action import DecisionAction
|
||||
|
||||
|
||||
ResearchQualityLevel = Literal["good", "usable", "limited", "poor", "unknown"]
|
||||
ResearchDirection = Literal["bullish", "bearish", "neutral", "unknown"]
|
||||
ResearchEvidenceFreshness = Literal["fresh", "stale", "unknown"]
|
||||
ResearchInvalidationCategory = Literal[
|
||||
"price",
|
||||
"volume",
|
||||
"evidence",
|
||||
"market",
|
||||
"time",
|
||||
"data_quality",
|
||||
"manual",
|
||||
]
|
||||
ResearchInvalidationSeverity = Literal["watch", "warning", "critical"]
|
||||
|
||||
|
||||
class ResearchSubject(BaseModel):
|
||||
"""The entity being researched."""
|
||||
|
||||
stock_code: str = Field(..., min_length=1)
|
||||
stock_name: Optional[str] = None
|
||||
market: Optional[str] = None
|
||||
entity_ref: Optional[str] = Field(None, description="Optional EntityLink ref when available")
|
||||
|
||||
|
||||
class ResearchThesis(BaseModel):
|
||||
"""Structured investment thesis extracted from a report."""
|
||||
|
||||
direction: ResearchDirection = "unknown"
|
||||
summary: str = Field("", description="Concise thesis statement")
|
||||
confidence: Optional[float] = Field(None, ge=0.0, le=1.0)
|
||||
score: Optional[int] = Field(None, description="Original sentiment score when available")
|
||||
horizon: Optional[str] = None
|
||||
action: Optional[DecisionAction] = None
|
||||
action_label: Optional[str] = None
|
||||
reasons: List[str] = Field(default_factory=list)
|
||||
risks: List[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ResearchEvidenceItem(BaseModel):
|
||||
"""One evidence item with freshness and quality status."""
|
||||
|
||||
id: str = Field(..., min_length=1)
|
||||
source_type: str = Field(..., min_length=1)
|
||||
title: str = Field(..., min_length=1)
|
||||
summary: Optional[str] = None
|
||||
source: Optional[str] = None
|
||||
freshness: ResearchEvidenceFreshness = "unknown"
|
||||
quality_level: ResearchQualityLevel = "unknown"
|
||||
as_of: Optional[str] = None
|
||||
url: Optional[str] = None
|
||||
metadata: Dict[str, Any] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class ResearchInvalidationCondition(BaseModel):
|
||||
"""Condition that should invalidate or force reassessment of the thesis."""
|
||||
|
||||
id: str = Field(..., min_length=1)
|
||||
category: ResearchInvalidationCategory
|
||||
description: str = Field(..., min_length=1)
|
||||
trigger: Optional[str] = None
|
||||
severity: ResearchInvalidationSeverity = "warning"
|
||||
metric: Optional[str] = None
|
||||
threshold: Optional[str] = None
|
||||
due_at: Optional[str] = None
|
||||
metadata: Dict[str, Any] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class ResearchNextAction(BaseModel):
|
||||
"""Suggested next action for a human or workflow."""
|
||||
|
||||
action: str = Field(..., min_length=1)
|
||||
label: str = Field(..., min_length=1)
|
||||
reason: Optional[str] = None
|
||||
due_at: Optional[str] = None
|
||||
metadata: Dict[str, Any] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class ResearchDataQuality(BaseModel):
|
||||
"""Low-sensitive quality summary for the artifact inputs."""
|
||||
|
||||
level: ResearchQualityLevel = "unknown"
|
||||
overall_score: Optional[int] = Field(None, ge=0, le=100)
|
||||
source_count: int = Field(0, ge=0)
|
||||
stale_count: int = Field(0, ge=0)
|
||||
missing_blocks: List[str] = Field(default_factory=list)
|
||||
limitations: List[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ResearchArtifact(BaseModel):
|
||||
"""Structured report artifact for dashboard, stock detail, monitor and copilot reuse."""
|
||||
|
||||
schema_version: Literal["research-artifact-v1"] = "research-artifact-v1"
|
||||
artifact_id: str = Field(..., min_length=1)
|
||||
source_report_id: Optional[int] = None
|
||||
source_query_id: Optional[str] = None
|
||||
created_at: Optional[str] = None
|
||||
subject: ResearchSubject
|
||||
thesis: ResearchThesis
|
||||
evidence: List[ResearchEvidenceItem] = Field(default_factory=list)
|
||||
invalidation_conditions: List[ResearchInvalidationCondition] = Field(..., min_length=1)
|
||||
next_actions: List[ResearchNextAction] = Field(default_factory=list)
|
||||
data_quality: ResearchDataQuality = Field(default_factory=ResearchDataQuality)
|
||||
metadata: Dict[str, Any] = Field(default_factory=dict)
|
||||
@@ -3,6 +3,8 @@
|
||||
* Aligned with the API schema.
|
||||
*/
|
||||
|
||||
import type { ResearchArtifact } from './researchArtifact';
|
||||
|
||||
// ============ Request Types ============
|
||||
|
||||
export type StockReportType = 'simple' | 'detailed' | 'full' | 'brief';
|
||||
@@ -374,6 +376,7 @@ export interface AnalysisReport {
|
||||
summary: ReportSummary;
|
||||
strategy?: ReportStrategy;
|
||||
details?: ReportDetails;
|
||||
structuredReport?: ResearchArtifact | null;
|
||||
}
|
||||
|
||||
// ============ Analysis Result Types ============
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
import type { DecisionAction } from './analysis';
|
||||
|
||||
export type ResearchQualityLevel = 'good' | 'usable' | 'limited' | 'poor' | 'unknown';
|
||||
export type ResearchDirection = 'bullish' | 'bearish' | 'neutral' | 'unknown';
|
||||
export type ResearchEvidenceFreshness = 'fresh' | 'stale' | 'unknown';
|
||||
export type ResearchInvalidationCategory =
|
||||
| 'price'
|
||||
| 'volume'
|
||||
| 'evidence'
|
||||
| 'market'
|
||||
| 'time'
|
||||
| 'data_quality'
|
||||
| 'manual';
|
||||
export type ResearchInvalidationSeverity = 'watch' | 'warning' | 'critical';
|
||||
|
||||
export interface ResearchSubject {
|
||||
stockCode: string;
|
||||
stockName?: string | null;
|
||||
market?: string | null;
|
||||
entityRef?: string | null;
|
||||
}
|
||||
|
||||
export interface ResearchThesis {
|
||||
direction: ResearchDirection;
|
||||
summary: string;
|
||||
confidence?: number | null;
|
||||
score?: number | null;
|
||||
horizon?: string | null;
|
||||
action?: DecisionAction | null;
|
||||
actionLabel?: string | null;
|
||||
reasons: string[];
|
||||
risks: string[];
|
||||
}
|
||||
|
||||
export interface ResearchEvidenceItem {
|
||||
id: string;
|
||||
sourceType: string;
|
||||
title: string;
|
||||
summary?: string | null;
|
||||
source?: string | null;
|
||||
freshness: ResearchEvidenceFreshness;
|
||||
qualityLevel: ResearchQualityLevel;
|
||||
asOf?: string | null;
|
||||
url?: string | null;
|
||||
metadata: Record<string, unknown>;
|
||||
}
|
||||
|
||||
export interface ResearchInvalidationCondition {
|
||||
id: string;
|
||||
category: ResearchInvalidationCategory;
|
||||
description: string;
|
||||
trigger?: string | null;
|
||||
severity: ResearchInvalidationSeverity;
|
||||
metric?: string | null;
|
||||
threshold?: string | null;
|
||||
dueAt?: string | null;
|
||||
metadata: Record<string, unknown>;
|
||||
}
|
||||
|
||||
export interface ResearchNextAction {
|
||||
action: string;
|
||||
label: string;
|
||||
reason?: string | null;
|
||||
dueAt?: string | null;
|
||||
metadata: Record<string, unknown>;
|
||||
}
|
||||
|
||||
export interface ResearchDataQuality {
|
||||
level: ResearchQualityLevel;
|
||||
overallScore?: number | null;
|
||||
sourceCount: number;
|
||||
staleCount: number;
|
||||
missingBlocks: string[];
|
||||
limitations: string[];
|
||||
}
|
||||
|
||||
export interface ResearchArtifact {
|
||||
schemaVersion: 'research-artifact-v1';
|
||||
artifactId: string;
|
||||
sourceReportId?: number | null;
|
||||
sourceQueryId?: string | null;
|
||||
createdAt?: string | null;
|
||||
subject: ResearchSubject;
|
||||
thesis: ResearchThesis;
|
||||
evidence: ResearchEvidenceItem[];
|
||||
invalidationConditions: ResearchInvalidationCondition[];
|
||||
nextActions: ResearchNextAction[];
|
||||
dataQuality: ResearchDataQuality;
|
||||
metadata: Record<string, unknown>;
|
||||
}
|
||||
@@ -16,6 +16,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/).
|
||||
- [文档] 在中英繁 README 顶部关联 DSA arXiv 论文,并新增 `CITATION.cff` 统一项目引用信息。
|
||||
- [新功能] 新增数据源能力与数据集质量只读契约,提供 `/api/v1/data/overview` 和 `/api/v1/data/capabilities`,为大盘看板、数据中心、个股详情和自选 2.0 统一暴露 provider capability、dataset quality 和 source priority。
|
||||
- [改进] PR CI 增加文档路径检测:仅修改普通文档、非治理 Markdown 或 LICENSE 时跳过后端测试分片、Docker、Web 与桌面打包,保留轻量治理和门禁汇总;契约文档、静态 API 规格与测试 fixture 仍执行后端回归。
|
||||
- [新功能] 新增 ResearchArtifact 结构化研究产物契约,在 `AnalysisReport.structured_report` 中承载 Thesis、Evidence、Invalidation Conditions、Next Actions 和 Data Quality,并提供从现有报告生成结构化产物的后端 helper 与 Web 类型。
|
||||
- [修复] Linux/Docker 分享图补齐 Noto CJK 字体与中韩文字体栈,避免 PNG 只显示数字和英文、中文或韩文内容消失。
|
||||
- [新功能] Web Chat 意图识别层新增分词模块:`web_intent_tokenizer` 六步管道(多股票全名实体扫描 → 标点/空白切分 → 代码形提取 → 市场关键词 → 无歧义关键词 → 残存 gap 多策略 DFS 匹配)把用户消息切分为携带语义标签的 Token 序列;配套 `web_intent_types` 数据字典(Token 结构、Market 枚举、21 个语义 tag、clean/extend 双词池与正则机器)。核心原则"宁可不做,不可做错":Step 1~5 只做精确匹配,Step 6 要求整段 TAG 全覆盖(交叉验证)才产出,未覆盖片段保持空 tag 交下游 LLM 兜底;代码形 token 辨认为 `stock_code`(附 code/name/market 三元组)/ `wrong_{market}_code` / `unknown_{market}_code` 三态,token 层代码拼写统一 canonical 归一(a=6 位裸数字、hk=HK+5 位、us=大写 ticker)。意图枚举与意图识别结果随后续 `web_intent_resolver` PR 引入。新增 183 个分词单元测试。
|
||||
<!-- 新条目格式:- [类型] 描述(类型取值:新功能/改进/修复/文档/测试/chore)-->
|
||||
|
||||
@@ -46,6 +46,7 @@
|
||||
| [Bot 平台配置](bot/) | 飞书、钉钉、Discord 等 Bot 配置截图和补充说明 |
|
||||
| [实时告警中心](alerts.md) | EventMonitor 基线、Web 规则管理、通知结果、冷却状态和 Phase 边界 |
|
||||
| [DecisionSignal 决策信号专题](decision-signals.md) | AI 建议池字段语义、API、Web 展示、告警/通知/组合风险联动、后验评估、脱敏、迁移与回滚 |
|
||||
| [ResearchArtifact 结构化研究产物](research-artifact.md) | structured_report 字段、Thesis / Evidence / Invalidation / Next Action / Data Quality 契约和旧报告兼容边界 |
|
||||
| [资讯 / 情报源](intelligence-sources.md) | RSS/Atom 合规资讯源配置、测试、拉取、去重、存储、查询与安全边界 |
|
||||
| [分析上下文包契约、运行态消费与可见性](analysis-context-pack.md) | AnalysisContextPack 首版范围、字段质量状态、P1/P2 内部契约、P3 Prompt 摘要消费、P4 历史/API/Web 低敏可见性、P5 数据质量评分、P6 迁移回滚与源码锚点;完整指南补充 #1386 阶段感知分析、迁移与回滚入口 |
|
||||
| [图片识别 Prompt](image-extract-prompt.md) | 图片识别股票信息的 Prompt 与使用边界 |
|
||||
|
||||
@@ -46,6 +46,7 @@ This is the entry point for project documentation. The README covers the project
|
||||
| [Bot Platform Docs](bot/) <sub><sub></sub></sub> (Chinese-only) | Feishu, DingTalk, Discord, and related Bot configuration screenshots and notes |
|
||||
| [Real-Time Alert Center](alerts.md) <sub><sub></sub></sub> (Chinese-only) | EventMonitor baseline, Web rule management, notification attempts, cooldown state, and phase boundaries |
|
||||
| [DecisionSignal Topic](decision-signals.md) <sub><sub></sub></sub> (Chinese-only) | AI signal fields, API, Web display, alert/notification/portfolio-risk linkage, outcome evaluation, redaction, migration, and rollback |
|
||||
| [ResearchArtifact Contract](research-artifact.md) <sub><sub></sub></sub> (Chinese-only) | Structured thesis, evidence, invalidation conditions, deterministic fallback identity, and the boundary before persistence/API integration |
|
||||
| [Analysis Context Pack Contract, Runtime Consumption, And Visibility](analysis-context-pack.md) <sub><sub></sub></sub> (Chinese-only) | AnalysisContextPack first-scope boundaries, field quality states, P1/P2 internal contracts, P3 prompt-summary consumption, P4 history/API/Web low-sensitivity visibility, P5 data-quality scoring, and P6 migration/rollback notes, plus source anchors; the full guide adds #1386 market-phase analysis, migration, and rollback entry points |
|
||||
| [Image Extraction Prompt](image-extract-prompt.md) <sub><sub></sub></sub> (Chinese-only) | Prompt and boundaries for extracting stock information from images |
|
||||
| [OpenClaw Skill Integration](openclaw-skill-integration.md) <sub><sub></sub></sub> (Chinese-only) | OpenClaw / Skill external integration notes |
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
# ResearchArtifact 结构化研究产物
|
||||
|
||||
`ResearchArtifact` 是报告的结构化版本,用于后续 Dashboard、个股研究页、监控中心、自选股和 Copilot 复用同一份研究结论。
|
||||
|
||||
## 目标
|
||||
|
||||
首版解决三个问题:
|
||||
|
||||
- 把报告结论沉淀为 `thesis`,而不是只依赖 Markdown。
|
||||
- 每份结构化报告必须带 `invalidation_conditions`,明确什么时候需要推翻或重新评估。
|
||||
- 证据项统一带 `freshness` 与 `quality_level`,让页面能展示数据新鲜度和可信度。
|
||||
|
||||
## 字段
|
||||
|
||||
顶层结构:
|
||||
|
||||
- `schema_version`:固定为 `research-artifact-v1`。
|
||||
- `artifact_id`:稳定产物 id,优先使用 `report:<source_report_id>`;尚未持久化时使用 `report:<stock_code>:<query_id>`,避免批量分析中多个股票共享 query id 造成碰撞;无 query id 时回退为 `report:<stock_code>`。
|
||||
- `source_report_id`:历史报告主键。
|
||||
- `source_query_id`:分析任务 query id。
|
||||
- `created_at`:报告创建时间。
|
||||
- `subject`:标的。
|
||||
- `thesis`:结构化观点。
|
||||
- `evidence`:证据列表。
|
||||
- `invalidation_conditions`:失效条件,至少一条。
|
||||
- `next_actions`:下一步动作。
|
||||
- `data_quality`:输入质量摘要。
|
||||
- `metadata`:低敏扩展字段。
|
||||
|
||||
## Thesis
|
||||
|
||||
`thesis` 包含:
|
||||
|
||||
- `direction`:`bullish` / `bearish` / `neutral` / `unknown`。
|
||||
- `summary`:核心观点。
|
||||
- `confidence`:由情绪分换算的置信度。
|
||||
- `score`:原始情绪分。
|
||||
- `action` / `action_label`:结构化建议动作。
|
||||
- `reasons`:支持理由。
|
||||
- `risks`:主要风险。
|
||||
|
||||
## Evidence
|
||||
|
||||
`evidence` 包含:
|
||||
|
||||
- `source_type`:例如 `analysis_context`、`news`、`fundamental`、`market_structure`。
|
||||
- `freshness`:`fresh` / `stale` / `unknown`。
|
||||
- `quality_level`:`good` / `usable` / `limited` / `poor` / `unknown`。
|
||||
- `source`、`as_of`、`url`、`metadata` 等低敏扩展字段。
|
||||
|
||||
首版适配器会从 `analysis_context_pack_overview.blocks`、新闻摘要、财报、分红和市场结构中提取证据。
|
||||
|
||||
## Invalidation
|
||||
|
||||
每份 `ResearchArtifact` 必须包含至少一条 `invalidation_conditions`。
|
||||
|
||||
首版默认来源:
|
||||
|
||||
- `strategy.stop_loss` 生成价格失效条件。
|
||||
- `strategy.take_profit` 生成止盈复核条件。
|
||||
- 数据质量限制生成 `data_quality` 失效条件。
|
||||
- 新闻缺失披露生成 `evidence` 失效条件。
|
||||
- 如果没有显式条件,生成 `manual:thesis_reassessment` 兜底复核条件。
|
||||
|
||||
## 兼容边界
|
||||
|
||||
`AnalysisReport` 新增可选字段:
|
||||
|
||||
```text
|
||||
structured_report?: ResearchArtifact | null
|
||||
```
|
||||
|
||||
旧报告可以继续不返回该字段;Web 类型和后端 schema 都按可选字段处理。
|
||||
|
||||
本 PR 只定义 schema、类型和确定性 fallback helper,尚未把 helper 接入报告持久化或历史详情返回链路;该接线作为 #2278 的后续阶段完成。
|
||||
|
||||
## 实现入口
|
||||
|
||||
后端:
|
||||
|
||||
- Schema:`api/v1/schemas/research_artifact.py`
|
||||
- Helper:`src/services/research_artifact_service.py`
|
||||
- 测试:`tests/test_research_artifact_service.py`
|
||||
|
||||
Web:
|
||||
|
||||
- 类型:`apps/dsa-web/src/types/researchArtifact.ts`
|
||||
- `AnalysisReport.structuredReport`:可选消费入口
|
||||
|
||||
## 后续接入
|
||||
|
||||
1. 报告生成链路在持久化时写入 `structured_report`。
|
||||
2. 个股研究页直接读取 `thesis`、`evidence` 和 `invalidation_conditions`。
|
||||
3. 监控中心把失效条件转成可编辑规则。
|
||||
4. Dashboard 的 What Changed 可以比较两份 `ResearchArtifact` 的 thesis 与 evidence 差异。
|
||||
@@ -0,0 +1,383 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Build structured research artifacts from existing analysis reports."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Any, Dict, Iterable, List, Optional
|
||||
|
||||
from api.v1.schemas.research_artifact import ResearchArtifact
|
||||
|
||||
|
||||
_BULLISH_ACTIONS = {"buy", "add"}
|
||||
_BEARISH_ACTIONS = {"reduce", "sell", "avoid"}
|
||||
_NEUTRAL_ACTIONS = {"hold", "watch"}
|
||||
|
||||
|
||||
def build_research_artifact(
|
||||
report: Any,
|
||||
*,
|
||||
evidence_items: Optional[Iterable[Dict[str, Any]]] = None,
|
||||
invalidation_conditions: Optional[Iterable[Dict[str, Any]]] = None,
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Build a ResearchArtifact-compatible dict without mutating the source report."""
|
||||
meta = _section(report, "meta")
|
||||
summary = _section(report, "summary")
|
||||
strategy = _section(report, "strategy")
|
||||
details = _section(report, "details")
|
||||
context_overview = _value(details, "analysis_context_pack_overview")
|
||||
source_report_id = _as_int(_value(meta, "id"))
|
||||
query_id = _as_text(_value(meta, "query_id"))
|
||||
stock_code = _as_text(_value(meta, "stock_code")) or "UNKNOWN"
|
||||
artifact_id = (
|
||||
f"report:{source_report_id}"
|
||||
if source_report_id is not None
|
||||
else f"report:{stock_code}:{query_id}" if query_id else f"report:{stock_code}"
|
||||
)
|
||||
|
||||
evidence = _build_evidence(details, context_overview)
|
||||
if evidence_items is not None:
|
||||
evidence.extend(dict(item) for item in evidence_items)
|
||||
|
||||
invalidations = _build_invalidation_conditions(summary, strategy, details, context_overview)
|
||||
if invalidation_conditions is not None:
|
||||
invalidations.extend(dict(item) for item in invalidation_conditions)
|
||||
if not invalidations:
|
||||
invalidations.append(_manual_reassessment_condition())
|
||||
|
||||
payload = {
|
||||
"artifact_id": artifact_id,
|
||||
"source_report_id": source_report_id,
|
||||
"source_query_id": query_id or None,
|
||||
"created_at": _as_text(_value(meta, "created_at")) or None,
|
||||
"subject": {
|
||||
"stock_code": stock_code,
|
||||
"stock_name": _as_text(_value(meta, "stock_name")) or None,
|
||||
"market": _subject_market(meta, context_overview),
|
||||
},
|
||||
"thesis": _build_thesis(summary),
|
||||
"evidence": evidence,
|
||||
"invalidation_conditions": invalidations,
|
||||
"next_actions": _build_next_actions(summary, invalidations),
|
||||
"data_quality": _build_data_quality(context_overview, evidence),
|
||||
"metadata": {
|
||||
"source": "analysis_report",
|
||||
**dict(metadata or {}),
|
||||
},
|
||||
}
|
||||
return ResearchArtifact.model_validate(payload).model_dump(exclude_none=True)
|
||||
|
||||
|
||||
def _build_thesis(summary: Any) -> Dict[str, Any]:
|
||||
action = _as_text(_value(summary, "action")) or None
|
||||
action_label = _as_text(_value(summary, "action_label")) or _as_text(_value(summary, "operation_advice")) or None
|
||||
score = _as_int(_value(summary, "sentiment_score"))
|
||||
analysis_summary = _as_text(_value(summary, "analysis_summary"))
|
||||
trend_prediction = _as_text(_value(summary, "trend_prediction"))
|
||||
operation_advice = _as_text(_value(summary, "operation_advice"))
|
||||
reasons = [value for value in [analysis_summary, trend_prediction] if value]
|
||||
risks = []
|
||||
if action == "alert":
|
||||
risks.append(operation_advice or "Report action is alert")
|
||||
return {
|
||||
"direction": _direction_from_action_or_score(action, score),
|
||||
"summary": analysis_summary or trend_prediction or operation_advice,
|
||||
"score": score,
|
||||
"confidence": _score_to_confidence(score),
|
||||
"action": action,
|
||||
"action_label": action_label,
|
||||
"reasons": reasons,
|
||||
"risks": risks,
|
||||
}
|
||||
|
||||
|
||||
def _build_evidence(details: Any, context_overview: Any) -> List[Dict[str, Any]]:
|
||||
evidence: List[Dict[str, Any]] = []
|
||||
blocks = _value(context_overview, "blocks") or []
|
||||
for index, block in enumerate(blocks):
|
||||
key = _as_text(_value(block, "key")) or f"context_{index + 1}"
|
||||
status = _as_text(_value(block, "status"))
|
||||
evidence.append({
|
||||
"id": f"context:{key}",
|
||||
"source_type": "analysis_context",
|
||||
"title": _as_text(_value(block, "label")) or key,
|
||||
"source": _as_text(_value(block, "source")) or None,
|
||||
"freshness": _freshness_from_status(status),
|
||||
"quality_level": _quality_from_status(status),
|
||||
"metadata": {
|
||||
"status": status,
|
||||
"warnings": _value(block, "warnings") or [],
|
||||
"missing_reasons": _value(block, "missing_reasons") or [],
|
||||
},
|
||||
})
|
||||
|
||||
if _value(details, "news_content") or _value(details, "empty_news_disclosure"):
|
||||
has_news_content = bool(_value(details, "news_content"))
|
||||
evidence.append({
|
||||
"id": "news:summary",
|
||||
"source_type": "news",
|
||||
"title": "News summary",
|
||||
"summary": _as_text(_value(details, "empty_news_disclosure"))
|
||||
or _compact(_as_text(_value(details, "news_content"))),
|
||||
"freshness": "unknown",
|
||||
"quality_level": "usable" if has_news_content else "limited",
|
||||
"metadata": {"status": "available" if has_news_content else "missing"},
|
||||
})
|
||||
|
||||
if _value(details, "financial_report"):
|
||||
evidence.append({
|
||||
"id": "fundamental:financial_report",
|
||||
"source_type": "fundamental",
|
||||
"title": "Financial report",
|
||||
"freshness": "unknown",
|
||||
"quality_level": "usable",
|
||||
})
|
||||
|
||||
if _value(details, "dividend_metrics"):
|
||||
evidence.append({
|
||||
"id": "fundamental:dividend_metrics",
|
||||
"source_type": "fundamental",
|
||||
"title": "Dividend metrics",
|
||||
"freshness": "unknown",
|
||||
"quality_level": "usable",
|
||||
})
|
||||
|
||||
market_structure = _value(details, "market_structure")
|
||||
if market_structure:
|
||||
status = _as_text(_value(market_structure, "status"))
|
||||
evidence.append({
|
||||
"id": "market:structure",
|
||||
"source_type": "market_structure",
|
||||
"title": "Market structure",
|
||||
"freshness": _freshness_from_status(status),
|
||||
"quality_level": _quality_from_status(status),
|
||||
"metadata": {"status": status},
|
||||
})
|
||||
|
||||
return evidence
|
||||
|
||||
|
||||
def _build_invalidation_conditions(
|
||||
summary: Any,
|
||||
strategy: Any,
|
||||
details: Any,
|
||||
context_overview: Any,
|
||||
) -> List[Dict[str, Any]]:
|
||||
conditions: List[Dict[str, Any]] = []
|
||||
stop_loss = _as_text(_value(strategy, "stop_loss"))
|
||||
if stop_loss:
|
||||
conditions.append({
|
||||
"id": "price:stop_loss",
|
||||
"category": "price",
|
||||
"description": f"Price breaks the report stop-loss area: {stop_loss}",
|
||||
"trigger": stop_loss,
|
||||
"severity": "critical",
|
||||
"metric": "price",
|
||||
"threshold": stop_loss,
|
||||
})
|
||||
|
||||
take_profit = _as_text(_value(strategy, "take_profit"))
|
||||
if take_profit:
|
||||
conditions.append({
|
||||
"id": "price:take_profit_review",
|
||||
"category": "price",
|
||||
"description": f"Price reaches the take-profit area and requires reassessment: {take_profit}",
|
||||
"trigger": take_profit,
|
||||
"severity": "watch",
|
||||
"metric": "price",
|
||||
"threshold": take_profit,
|
||||
})
|
||||
|
||||
limitations = _quality_limitations(context_overview)
|
||||
if limitations:
|
||||
conditions.append({
|
||||
"id": "data_quality:limitations",
|
||||
"category": "data_quality",
|
||||
"description": "Input data quality limitations require reassessment when fresh data is available.",
|
||||
"trigger": "; ".join(limitations[:3]),
|
||||
"severity": "warning",
|
||||
"metadata": {"limitations": limitations},
|
||||
})
|
||||
|
||||
disclosure = _as_text(_value(details, "empty_news_disclosure"))
|
||||
if disclosure:
|
||||
conditions.append({
|
||||
"id": "evidence:news_missing",
|
||||
"category": "evidence",
|
||||
"description": "News evidence was unavailable or empty during report generation.",
|
||||
"trigger": disclosure,
|
||||
"severity": "watch",
|
||||
})
|
||||
|
||||
thesis_text = _as_text(_value(summary, "analysis_summary")) or _as_text(_value(summary, "trend_prediction"))
|
||||
if thesis_text and not conditions:
|
||||
conditions.append(_manual_reassessment_condition())
|
||||
|
||||
return conditions
|
||||
|
||||
|
||||
def _manual_reassessment_condition() -> Dict[str, Any]:
|
||||
return {
|
||||
"id": "manual:thesis_reassessment",
|
||||
"category": "manual",
|
||||
"description": "Reassess when price action, news, fundamentals, or market phase conflicts with the thesis.",
|
||||
"severity": "warning",
|
||||
}
|
||||
|
||||
|
||||
def _build_next_actions(summary: Any, invalidations: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
action = _as_text(_value(summary, "action")) or "watch"
|
||||
label = _as_text(_value(summary, "action_label")) or _as_text(_value(summary, "operation_advice")) or action
|
||||
next_actions = [{
|
||||
"action": action,
|
||||
"label": label,
|
||||
"reason": _as_text(_value(summary, "analysis_summary")) or None,
|
||||
}]
|
||||
if invalidations:
|
||||
next_actions.append({
|
||||
"action": "monitor_invalidation",
|
||||
"label": "Monitor invalidation",
|
||||
"reason": invalidations[0]["description"],
|
||||
"metadata": {"condition_id": invalidations[0]["id"]},
|
||||
})
|
||||
return next_actions
|
||||
|
||||
|
||||
def _build_data_quality(context_overview: Any, evidence: List[Dict[str, Any]]) -> Dict[str, Any]:
|
||||
data_quality = _value(context_overview, "data_quality")
|
||||
counts = _value(context_overview, "counts")
|
||||
missing_blocks = [
|
||||
_as_text(_value(block, "key"))
|
||||
for block in (_value(context_overview, "blocks") or [])
|
||||
if _as_text(_value(block, "status")) in {"missing", "fetch_failed", "not_supported"}
|
||||
]
|
||||
return {
|
||||
"level": _as_text(_value(data_quality, "level")) or _infer_quality_level(evidence),
|
||||
"overall_score": _as_int(_value(data_quality, "overall_score")),
|
||||
"source_count": sum(1 for item in evidence if _evidence_has_usable_source(item)),
|
||||
"stale_count": sum(1 for item in evidence if item.get("freshness") == "stale"),
|
||||
"missing_blocks": [block for block in missing_blocks if block],
|
||||
"limitations": _quality_limitations(context_overview),
|
||||
"metadata": {"counts": counts} if counts else {},
|
||||
}
|
||||
|
||||
|
||||
def _quality_limitations(context_overview: Any) -> List[str]:
|
||||
data_quality = _value(context_overview, "data_quality")
|
||||
limitations = _value(data_quality, "limitations") or []
|
||||
return [_as_text(item) for item in limitations if _as_text(item)]
|
||||
|
||||
|
||||
def _direction_from_action_or_score(action: Optional[str], score: Optional[int]) -> str:
|
||||
if action in _BULLISH_ACTIONS:
|
||||
return "bullish"
|
||||
if action in _BEARISH_ACTIONS:
|
||||
return "bearish"
|
||||
if action in _NEUTRAL_ACTIONS:
|
||||
return "neutral"
|
||||
if score is not None:
|
||||
if score >= 65:
|
||||
return "bullish"
|
||||
if score <= 35:
|
||||
return "bearish"
|
||||
return "neutral"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _score_to_confidence(score: Optional[int]) -> Optional[float]:
|
||||
if score is None:
|
||||
return None
|
||||
bounded = max(0, min(100, score))
|
||||
return round(abs(bounded - 50) / 50, 2)
|
||||
|
||||
|
||||
def _freshness_from_status(status: str) -> str:
|
||||
if status in {"available", "fallback", "partial", "estimated", "ok"}:
|
||||
return "fresh"
|
||||
if status == "stale":
|
||||
return "stale"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _quality_from_status(status: str) -> str:
|
||||
if status in {"available", "ok"}:
|
||||
return "good"
|
||||
if status in {"fallback", "estimated"}:
|
||||
return "usable"
|
||||
if status in {"partial", "stale"}:
|
||||
return "limited"
|
||||
if status in {"missing", "fetch_failed"}:
|
||||
return "poor"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _evidence_has_usable_source(item: Dict[str, Any]) -> bool:
|
||||
metadata = item.get("metadata")
|
||||
status = _as_text(metadata.get("status")) if isinstance(metadata, dict) else ""
|
||||
return status not in {"missing", "fetch_failed", "not_supported", "unavailable", "unknown"}
|
||||
|
||||
|
||||
def _infer_quality_level(evidence: List[Dict[str, Any]]) -> str:
|
||||
if not evidence:
|
||||
return "unknown"
|
||||
levels = {str(item.get("quality_level") or "unknown") for item in evidence}
|
||||
if "poor" in levels:
|
||||
return "poor"
|
||||
if "limited" in levels:
|
||||
return "limited"
|
||||
if "usable" in levels:
|
||||
return "usable"
|
||||
if levels == {"good"}:
|
||||
return "good"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _subject_market(meta: Any, context_overview: Any) -> Optional[str]:
|
||||
market = _value(_value(context_overview, "subject"), "market")
|
||||
return _as_text(market) or _as_text(_value(meta, "market")) or None
|
||||
|
||||
|
||||
def _section(report: Any, key: str) -> Any:
|
||||
return _value(report, key) or {}
|
||||
|
||||
|
||||
def _value(source: Any, key: str) -> Any:
|
||||
if source is None:
|
||||
return None
|
||||
if isinstance(source, dict):
|
||||
if key in source:
|
||||
return source[key]
|
||||
camel_key = _snake_to_camel(key)
|
||||
return source.get(camel_key)
|
||||
if hasattr(source, key):
|
||||
return getattr(source, key)
|
||||
return getattr(source, _snake_to_camel(key), None)
|
||||
|
||||
|
||||
def _snake_to_camel(value: str) -> str:
|
||||
head, *tail = value.split("_")
|
||||
return head + "".join(item[:1].upper() + item[1:] for item in tail)
|
||||
|
||||
|
||||
def _as_text(value: Any) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
text = str(value).strip()
|
||||
return re.sub(r"\s+", " ", text)
|
||||
|
||||
|
||||
def _as_int(value: Any) -> Optional[int]:
|
||||
try:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
return int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _compact(value: str, limit: int = 180) -> str:
|
||||
text = _as_text(value)
|
||||
if len(text) <= limit:
|
||||
return text
|
||||
return f"{text[:limit]}..."
|
||||
@@ -0,0 +1,181 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Tests for structured ResearchArtifact contract helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
|
||||
from pydantic import ValidationError
|
||||
import pytest
|
||||
|
||||
from api.v1.schemas.research_artifact import ResearchArtifact
|
||||
from src.services.research_artifact_service import build_research_artifact
|
||||
|
||||
|
||||
def test_build_research_artifact_from_report_with_evidence_and_invalidation() -> None:
|
||||
report = {
|
||||
"meta": {
|
||||
"id": 12,
|
||||
"query_id": "q-12",
|
||||
"stock_code": "600519",
|
||||
"stock_name": "贵州茅台",
|
||||
"created_at": "2026-03-19T08:00:00",
|
||||
},
|
||||
"summary": {
|
||||
"analysis_summary": "趋势维持偏强",
|
||||
"operation_advice": "持有",
|
||||
"action": "hold",
|
||||
"action_label": "持有",
|
||||
"trend_prediction": "震荡上行",
|
||||
"sentiment_score": 72,
|
||||
},
|
||||
"strategy": {
|
||||
"stop_loss": "1680",
|
||||
"take_profit": "1880",
|
||||
},
|
||||
"details": {
|
||||
"analysis_context_pack_overview": {
|
||||
"subject": {"market": "cn"},
|
||||
"blocks": [
|
||||
{
|
||||
"key": "daily_price",
|
||||
"label": "日线行情",
|
||||
"status": "available",
|
||||
"source": "tencent",
|
||||
"warnings": [],
|
||||
"missing_reasons": [],
|
||||
},
|
||||
{
|
||||
"key": "news",
|
||||
"label": "新闻",
|
||||
"status": "partial",
|
||||
"source": "anspire",
|
||||
"warnings": ["partial"],
|
||||
"missing_reasons": [],
|
||||
},
|
||||
],
|
||||
"data_quality": {
|
||||
"overall_score": 83,
|
||||
"level": "good",
|
||||
"limitations": ["新闻覆盖有限"],
|
||||
},
|
||||
},
|
||||
"news_content": "公司新闻摘要",
|
||||
},
|
||||
}
|
||||
|
||||
artifact = ResearchArtifact.model_validate(build_research_artifact(report))
|
||||
|
||||
assert artifact.schema_version == "research-artifact-v1"
|
||||
assert artifact.artifact_id == "report:12"
|
||||
assert artifact.subject.stock_code == "600519"
|
||||
assert artifact.subject.market == "cn"
|
||||
assert artifact.thesis.direction == "neutral"
|
||||
assert artifact.thesis.action == "hold"
|
||||
assert artifact.data_quality.level == "good"
|
||||
assert artifact.data_quality.source_count == 3
|
||||
assert {item.id for item in artifact.evidence} == {
|
||||
"context:daily_price",
|
||||
"context:news",
|
||||
"news:summary",
|
||||
}
|
||||
assert artifact.evidence[0].freshness == "fresh"
|
||||
assert artifact.evidence[0].quality_level == "good"
|
||||
condition_ids = {item.id for item in artifact.invalidation_conditions}
|
||||
assert "price:stop_loss" in condition_ids
|
||||
assert "data_quality:limitations" in condition_ids
|
||||
assert artifact.next_actions[-1].action == "monitor_invalidation"
|
||||
|
||||
|
||||
def test_build_research_artifact_always_includes_invalidation_conditions() -> None:
|
||||
artifact = ResearchArtifact.model_validate(build_research_artifact({
|
||||
"meta": {"query_id": "q-empty", "stock_code": "AAPL"},
|
||||
"summary": {"analysis_summary": "等待更多证据", "sentiment_score": 50},
|
||||
}))
|
||||
|
||||
assert artifact.artifact_id == "report:AAPL:q-empty"
|
||||
assert artifact.invalidation_conditions[0].id == "manual:thesis_reassessment"
|
||||
assert artifact.data_quality.level == "unknown"
|
||||
|
||||
|
||||
def test_research_artifact_requires_invalidation_conditions() -> None:
|
||||
with pytest.raises(ValidationError):
|
||||
ResearchArtifact.model_validate({
|
||||
"artifact_id": "report:bad",
|
||||
"subject": {"stock_code": "AAPL"},
|
||||
"thesis": {"summary": "missing invalidation"},
|
||||
"invalidation_conditions": [],
|
||||
})
|
||||
|
||||
|
||||
def test_fallback_artifact_id_is_unique_for_stocks_in_the_same_batch() -> None:
|
||||
first = build_research_artifact({
|
||||
"meta": {"query_id": "batch-1", "stock_code": "600519"},
|
||||
"summary": {"analysis_summary": "first"},
|
||||
})
|
||||
second = build_research_artifact({
|
||||
"meta": {"query_id": "batch-1", "stock_code": "000001"},
|
||||
"summary": {"analysis_summary": "second"},
|
||||
})
|
||||
|
||||
assert first["artifact_id"] == "report:600519:batch-1"
|
||||
assert second["artifact_id"] == "report:000001:batch-1"
|
||||
assert first["artifact_id"] != second["artifact_id"]
|
||||
|
||||
|
||||
def test_attribute_report_preserves_falsey_values() -> None:
|
||||
report = SimpleNamespace(
|
||||
meta=SimpleNamespace(query_id="batch-zero", stock_code="AAPL"),
|
||||
summary=SimpleNamespace(
|
||||
sentiment_score=0,
|
||||
analysis_summary="zero is a real score",
|
||||
action="",
|
||||
),
|
||||
strategy=SimpleNamespace(),
|
||||
details=SimpleNamespace(),
|
||||
)
|
||||
|
||||
artifact = ResearchArtifact.model_validate(build_research_artifact(report))
|
||||
|
||||
assert artifact.artifact_id == "report:AAPL:batch-zero"
|
||||
assert artifact.thesis.score == 0
|
||||
assert artifact.thesis.confidence == 1.0
|
||||
assert artifact.thesis.direction == "bearish"
|
||||
assert artifact.thesis.action is None
|
||||
|
||||
|
||||
def test_unavailable_context_blocks_do_not_inflate_source_count() -> None:
|
||||
artifact = ResearchArtifact.model_validate(build_research_artifact({
|
||||
"meta": {"query_id": "missing-only", "stock_code": "AAPL"},
|
||||
"summary": {"analysis_summary": "waiting for evidence"},
|
||||
"details": {
|
||||
"analysis_context_pack_overview": {
|
||||
"blocks": [
|
||||
{"key": "daily_price", "status": "missing"},
|
||||
{"key": "news", "status": "fetch_failed"},
|
||||
],
|
||||
},
|
||||
"empty_news_disclosure": "News evidence is unavailable.",
|
||||
},
|
||||
}))
|
||||
|
||||
assert artifact.data_quality.source_count == 0
|
||||
assert {item.id for item in artifact.evidence} == {
|
||||
"context:daily_price",
|
||||
"context:news",
|
||||
"news:summary",
|
||||
}
|
||||
|
||||
|
||||
def test_market_structure_ok_is_healthy_evidence() -> None:
|
||||
artifact = ResearchArtifact.model_validate(build_research_artifact({
|
||||
"meta": {"query_id": "market-ok", "stock_code": "600519"},
|
||||
"summary": {"analysis_summary": "market structure available"},
|
||||
"details": {"market_structure": {"status": "ok"}},
|
||||
}))
|
||||
|
||||
market_evidence = next(item for item in artifact.evidence if item.id == "market:structure")
|
||||
assert market_evidence.freshness == "fresh"
|
||||
assert market_evidence.quality_level == "good"
|
||||
assert artifact.data_quality.source_count == 1
|
||||
assert artifact.data_quality.level == "good"
|
||||
Reference in New Issue
Block a user