-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy path__init__.py
More file actions
206 lines (158 loc) · 5.85 KB
/
Copy path__init__.py
File metadata and controls
206 lines (158 loc) · 5.85 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
"""
医镜 (MediLens) - 基于 GLM-5.1 的中文医疗 NLP 工具包
MediLens 是一个专注于中文医疗领域的自然语言处理工具包,
利用智谱 GLM-5.1 大语言模型提供医疗实体识别、PHI 检测、
文本分类等核心功能。
典型用法::
from medilens import MediLens
lens = MediLens(api_key="your-api-key")
result = lens.analyze("患者三天前出现发热、咳嗽,体温最高38.5℃")
print(result.entities)
"""
from __future__ import annotations
from importlib.metadata import PackageNotFoundError
from importlib.metadata import version as _version
from typing import Any, Dict, List, Optional
from medilens.config import Settings, settings as default_settings
try:
__version__ = _version("medilens")
except PackageNotFoundError: # pragma: no cover
__version__ = "0.1.0"
__version__ = "0.1.0"
"""MediLens 包版本号"""
__author__ = "MediLens Team"
__license__ = "MIT"
# 公开的模块成员
__all__ = [
"MediLens",
"Settings",
"__version__",
]
class MediLens:
"""MediLens 主入口类,提供一站式医疗 NLP 分析能力。
封装了分析引擎、实体识别、PHI 检测、文本分类等核心功能,
通过统一的接口对外暴露。
Args:
api_key: GLM API 密钥
model: 模型名称,默认为 GLM-5.1
temperature: 生成温度 (0.0 ~ 1.0)
max_tokens: 最大输出 Token 数
settings: 自定义配置对象(与前面参数二选一)
Examples:
>>> lens = MediLens(api_key="your-key")
>>> result = lens.analyze("患者发热38.5℃三天")
"""
def __init__(
self,
api_key: Optional[str] = None,
model: Optional[str] = None,
temperature: Optional[float] = None,
max_tokens: Optional[int] = None,
settings: Optional[Settings] = None,
) -> None:
if settings is not None:
self._settings = settings
else:
self._settings = Settings(
api_key=api_key or default_settings.api_key,
model=model or default_settings.model,
temperature=temperature if temperature is not None else default_settings.temperature,
max_tokens=max_tokens if max_tokens is not None else default_settings.max_tokens,
)
self._settings.validate()
# 延迟导入以避免循环依赖
from medilens.core.analyzer import GLMClient
from medilens.core.classifier import MedicalTextClassifier
self._client = GLMClient(self._settings)
self._classifier = MedicalTextClassifier(self._client)
@property
def client(self):
"""底层 GLM API 客户端实例。"""
return self._client
@property
def settings(self) -> Settings:
"""当前配置。"""
return self._settings
# ---- 一站式分析 ----
def analyze(self, text: str, **kwargs) -> "EntityExtractionResult":
"""分析医疗文本,提取其中的医疗实体。
Args:
text: 输入的中文医疗文本
**kwargs: 传递给分析引擎的额外参数
Returns:
实体提取结果
"""
from medilens.core.entity import EntityExtractionResult
result = self._client.extract_entities(text, **kwargs)
return EntityExtractionResult.from_dict(result) if isinstance(result, dict) else result
def detect_phi(self, text: str, **kwargs) -> "PHIDetectionResult":
"""检测医疗文本中的受保护健康信息 (PHI)。
Args:
text: 输入的中文医疗文本
**kwargs: 额外参数
Returns:
PHI 检测结果
"""
result = self._client.detect_phi(text, **kwargs)
return result
def classify(self, text: str, **kwargs) -> "ClassificationResult":
"""对医疗文本进行分类。
Args:
text: 输入的中文医疗文本
**kwargs: 额外参数
Returns:
分类结果
"""
result = self._classifier.classify(text, **kwargs)
return result
def classify_text(self, text: str, **kwargs) -> dict:
"""分类医疗文本(兼容 analyzer 接口)。
Args:
text: 输入的中文医疗文本
**kwargs: 额外参数
Returns:
分类结果的字典表示
"""
result = self.classify(text, **kwargs)
return result.to_dict()
# ---- 批量处理 ----
def batch_analyze(
self,
inputs: List[str],
format: str = "json",
output_path: Optional[str] = None,
**kwargs,
) -> List[dict]:
"""批量分析医疗文本。
Args:
inputs: 文本列表或文件路径
format: 输出格式 (json/csv)
output_path: 输出文件路径
**kwargs: 额外参数
Returns:
分析结果列表
"""
from medilens.core.batch import BatchProcessor
processor = BatchProcessor(client=self._client)
return processor.process_texts(inputs, output_format=format, output_path=output_path, **kwargs)
# ---- 去标识化 ----
def deidentify(self, text: str, method: str = "MASK", **kwargs) -> str:
"""对医疗文本进行去标识化处理。
Args:
text: 输入文本
method: 去标识方法 (MASK / REPLACE / HASH / REDACT)
**kwargs: 额外参数
Returns:
去标识化后的文本
"""
result = self._client.detect_phi(text, **kwargs)
return result.apply(text, method=method)
# 便捷工厂
def create_medilens(**kwargs) -> MediLens:
"""创建 MediLens 实例的便捷函数。
Args:
**kwargs: 传给 MediLens 构造函数的参数
Returns:
MediLens 实例
"""
return MediLens(**kwargs)