feat: 启动时就自动检查下载nltk

This commit is contained in:
martsforever
2025-10-29 14:58:59 +08:00
parent b84ad4bcb8
commit 7d2c2a371f
3 changed files with 28 additions and 9 deletions
-4
View File
@@ -72,10 +72,6 @@ class MilvusService:
embed_model=self.embeddings,
)
# 预先加载NLTK语料库以避免多线程环境中的竞争条件
# 否则在多线程环境下,parser.get_nodes_from_documents 可能会出现报错信息:'WordListCorpusReader' object has no attribute '_LazyCorpusLoader__args'
load_nltk()
from llama_index.core.node_parser import SimpleNodeParser
# parser = HierarchicalNodeParser.from_defaults(chunk_sizes=[2048, 512, 128])
parser = SimpleNodeParser()
+26 -5
View File
@@ -1,14 +1,35 @@
# 预先加载NLTK语料库以避免多线程环境中的竞争条件
import asyncio
import nltk
load_flag = False
from nltk.data import find
# 判断nltk是否已经存在
def is_nltk_package_downloaded(package_name):
try:
find(f'tokenizers/{package_name}')
return True
except LookupError:
return False
# 下载nltk所需要文件
def load_nltk():
global load_flag
if not load_flag:
if not is_nltk_package_downloaded("punkt"):
print("ℹ️ 下载nltk")
nltk.download('punkt')
nltk.download('averaged_perceptron_tagger')
nltk.download('maxent_ne_chunker')
nltk.download('words')
load_flag = True
print("✅ nltk下载成功")
else:
print("✅ nltk已经存在,无需下载")
# 检查nltk是否已经下载,用于llama-index做文本分割
async def check_nltk():
print("nltk_path:", nltk.data.path)
# 预先加载NLTK语料库以避免多线程环境中的竞争条件
# 否则在多线程环境下,parser.get_nodes_from_documents 可能会出现报错信息:'WordListCorpusReader' object has no attribute '_LazyCorpusLoader__args'
await asyncio.wait_for(asyncio.to_thread(load_nltk), timeout=10)