-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathtext_utils.py
More file actions
51 lines (41 loc) 路 2.05 KB
/
Copy pathtext_utils.py
File metadata and controls
51 lines (41 loc) 路 2.05 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
import asyncio
import functools
from typing import List
import nltk
from langchain.text_splitter import RecursiveCharacterTextSplitter
# --- 肖褍薪泻褑懈褟 褉邪蟹斜懈械薪懈褟 薪邪 锌褉械写谢芯卸械薪懈褟 ---
def split_paragraph_into_sentences(paragraph: str) -> List[str]:
"""
袪邪蟹斜懈胁邪械褌 邪斜蟹邪褑 薪邪 锌褉械写谢芯卸械薪懈褟 褋 懈褋锌芯谢褜蟹芯胁邪薪懈械屑 nltk.sent_tokenize.
"""
try:
nltk.data.find("tokenizers/punkt")
except LookupError:
nltk.download("punkt")
return nltk.sent_tokenize(paragraph)
# 袨锌褉械写械谢懈褌械 胁褋锌芯屑芯谐邪褌械谢褜薪褍褞 褎褍薪泻褑懈褞 run_sync 写谢褟 芯斜芯褉邪褔懈胁邪薪懈褟 褋懈薪褏褉芯薪薪褘褏 褎褍薪泻褑懈泄
def run_sync(func):
@functools.wraps(func)
async def wrapper(*args, **kwargs):
loop = asyncio.get_event_loop()
return await loop.run_in_executor(None, functools.partial(func, *args, **kwargs))
return wrapper
# 小芯蟹写邪泄褌械 褋懈薪褏褉芯薪薪褍褞 胁械褉褋懈褞 split_text_semantically
def split_text_semantically_sync(text: str, chunk_size: int = 1500, chunk_overlap: int = 500) -> List[str]:
"""
小懈薪褏褉芯薪薪邪褟 胁械褉褋懈褟 写谢褟 褉邪蟹斜懈械薪懈褟 褌械泻褋褌邪 薪邪 褋械屑邪薪褌懈褔械褋泻懈械 褔邪薪泻懈 褋 懈褋锌芯谢褜蟹芯胁邪薪懈械屑 Langchain.
"""
text_splitter = RecursiveCharacterTextSplitter(
chunk_size=chunk_size,
chunk_overlap=chunk_overlap,
separators=["\n\n", "\n", " ", ""], # 袩芯褉褟写芯泻 胁邪卸械薪
length_function=len, # 袠褋锌芯谢褜蟹褍械屑 褋褌邪薪写邪褉褌薪褍褞 褎褍薪泻褑懈褞 len
)
chunks = text_splitter.split_text(text)
return chunks
# 袨斜械褉薪懈褌械 胁褘蟹芯胁 split_text_semantically 胁 asyncio.to_thread
async def split_text_semantically(text: str, chunk_size: int = 3000, chunk_overlap: int = 800) -> List[str]:
"""
袗褋懈薪褏褉芯薪薪芯 褉邪蟹斜懈胁邪械褌 褌械泻褋褌 薪邪 褋械屑邪薪褌懈褔械褋泻懈械 褔邪薪泻懈 褋 懈褋锌芯谢褜蟹芯胁邪薪懈械屑 Langchain.
"""
return await asyncio.to_thread(split_text_semantically_sync, text, chunk_size, chunk_overlap)