Files

36 lines
1.1 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# 预先加载NLTK语料库以避免多线程环境中的竞争条件
import asyncio
import nltk
from nltk.data import find
# 判断nltk是否已经存在
def is_nltk_package_downloaded(package_name):
try:
find(f'tokenizers/{package_name}')
return True
except LookupError:
return False
# 下载nltk所需要文件
def load_nltk():
if not is_nltk_package_downloaded("punkt"):
print("ℹ️ 下载nltk")
nltk.download('punkt')
nltk.download('averaged_perceptron_tagger')
nltk.download('maxent_ne_chunker')
nltk.download('words')
print("✅ nltk下载成功")
else:
print("✅ nltk已经存在,无需下载")
# 检查nltk是否已经下载,用于llama-index做文本分割
async def check_nltk():
print("nltk_path:", nltk.data.path)
# 预先加载NLTK语料库以避免多线程环境中的竞争条件
# 否则在多线程环境下,parser.get_nodes_from_documents 可能会出现报错信息:'WordListCorpusReader' object has no attribute '_LazyCorpusLoader__args'
await asyncio.wait_for(asyncio.to_thread(load_nltk), timeout=10)