สร้าง Realtime Research Agent แบบสตรีมมิงด้วย Ollama 0.13+ Streaming Tool Calling + Nemotron 3 Nano และ FunctionGemma
เรียนรู้การสร้างเอเจนต์วิจัยแบบเรียลไทม์ที่ค้นเว็บและสรุปผลทันทีด้วย Ollama 0.13+ และโมเดล Nemotron 3 Nano รองรับการทำงานออฟไลน์ 100% เหมาะสำหรับทีมไทย
สารบัญ

ทำไมเราต้องสร้าง Research Agent แบบออฟไลน์?
คุณเคยเจอปัญหาการส่งข้อมูลลูกค้าไปยัง API ของบริษัทภายนอกหรือไม่? การใช้บริการ LLM แบบ Cloud อาจสร้างความเสี่ยงด้านความเป็นส่วนตัวของข้อมูล โดยเฉพาะในทีมที่ต้องจัดการข้อมูลองค์กร
วันนี้เราจะสร้าง Realtime Research Agent ที่ทำงานออฟไลน์ 100% ด้วย Ollama 0.13+ ที่รองรับ Streaming Tool Calling ร่วมกับโมเดล Nemotron 3 Nano และ FunctionGemma ที่มีขนาดเล็กแต่ทรงพลัง
ทำความเข้าใจเครื่องมือที่ใช้
Ollama 0.13+ Streaming Tool Calling
Ollama เวอร์ชัน 0.13 ขึ้นไปมีฟีเจอร์ Streaming Tool Calling ที่ช่วยให้เราสามารถเรียกใช้ฟังก์ชันภายนอกได้แบบเรียลไทม์ โดยไม่ต้องรอให้โมเดลตอบกลับทั้งหมดก่อนค่อยประมวลผล
Nemotron 3 Nano
โมเดลจาก NVIDIA ที่มีขนาดเล็กแต่มีประสิทธิภาพสูงในการสรุปข้อมูลและตอบคำถาม เหมาะสำหรับการทำงานบนเครื่องที่มีทรัพยากรจำกัด
FunctionGemma
โมเดลที่ออกแบบมา specifically สำหรับการเรียกใช้ฟังก์ชัน มีความแม่นยำในการแยกแยะและเรียกใช้ tool ต่างๆ ตามความต้องการของผู้ใช้
เริ่มต้นสร้าง Research Agent
ขั้นตอนที่ 1: ติดตั้ง Ollama และโมเดล
# ติดตั้ง Ollama (macOS/Linux)
curl -fsSL https://ollama.com/install.sh | sh
# ดาวน์โหลดโมเดลที่จำเป็น
ollama pull nemotron-3-nano
ollama pull function-gemma
# ตรวจสอบเวอร์ชัน Ollama
ollama --version
ขั้นตอนที่ 2: ตั้งค่า Python Environment
# สร้าง virtual environment
python -m venv research_agent_env
source research_agent_env/bin/activate # Linux/Mac
# หรือ research_agent_env\Scripts\activate # Windows
# ติดตั้ง dependencies
pip install ollama requests beautifulsoup4 asyncio aiohttp
ขั้นตอนที่ 3: สร้าง Tool Definitions
import json
import requests
from bs4 import BeautifulSoup
import ollama
# กำหนด tools ที่จะใช้
TOOLS = [
{
"type": "function",
"function": {
"name": "search_web",
"description": "ค้นหาข้อมูลบนเว็บไซต์",
"parameters": {
"type": "object",
"properties": {
"query": {
"type": "string",
"description": "คำค้นหา"
},
"max_results": {
"type": "integer",
"description": "จำนวนผลลัพธ์สูงสุด"
}
},
"required": ["query"]
}
}
},
{
"type": "function",
"function": {
"name": "summarize_content",
"description": "สรุปเนื้อหาจาก URL",
"parameters": {
"type": "object",
"properties": {
"url": {
"type": "string",
"description": "URL ของเว็บไซต์ที่ต้องการสรุป"
},
"length": {
"type": "string",
"enum": ["short", "medium", "long"],
"description": "ความยาวของสรุป"
}
},
"required": ["url"]
}
}
}
]
สร้างฟังก์ชันสำหรับ Tool Calling
def search_web(query: str, max_results: int = 5) -> list:
"""
ค้นหาข้อมูลบนเว็บ (ตัวอย่างใช้ DuckDuckGo HTML)
"""
url = f"https://html.duckduckgo.com/html/?q={query}"
headers = {'User-Agent': 'Mozilla/5.0'}
try:
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, 'html.parser')
results = []
for item in soup.select('.result__body')[:max_results]:
title = item.select_one('.result__title').text.strip()
snippet = item.select_one('.result__snippet').text.strip()
link = item.select_one('.result__url').text.strip()
results.append({
'title': title,
'snippet': snippet,
'url': f"https://{link}"
})
return results
except Exception as e:
return [{"error": str(e)}]
def summarize_content(url: str, length: str = "medium") -> str:
"""
ดึงเนื้อหาจาก URL และสรุปด้วย Nemotron 3 Nano
"""
try:
response = requests.get(url, timeout=10)
soup = BeautifulSoup(response.text, 'html.parser')
content = ' '.join([p.text for p in soup.find_all('p')])
# ใช้ Nemotron 3 Nano สำหรับการสรุป
prompt = f"""สรุปเนื้อหาต่อไปนี้ในระดับความยาว {length}:
{content[:5000]}
กรุณาสรุปเป็นภาษาไทย:"""
response = ollama.chat(
model='nemotron-3-nano',
messages=[{'role': 'user', 'content': prompt}]
)
return response['message']['content']
except Exception as e:
return f"เกิดข้อผิดพลาด: {str(e)}"
สร้าง Streaming Research Agent
import asyncio
async def streaming_research_agent(user_query: str):
"""
Realtime Research Agent ที่สตรีมผลลัพธ์ทันที
"""
messages = [
{
"role": "system",
"content": "คุณคือผู้ช่วยวิจัยที่ใช้ tool เพื่อค้นหาข้อมูลและสรุปผล กรุณาตอบเป็นภาษาไทย"
},
{
"role": "user",
"content": user_query
}
]
# เริ่ม streaming chat กับ FunctionGemma
stream = ollama.chat(
model='function-gemma',
messages=messages,
tools=TOOLS,
stream=True
)
for chunk in stream:
if 'message' in chunk:
message = chunk['message']
# ตรวจสอบว่ามี tool call หรือไม่
if 'tool_calls' in message:
for tool_call in message['tool_calls']:
function_name = tool_call['function']['name']
function_args = tool_call['function']['arguments']
print(f"\n🔧 กำลังเรียกใช้: {function_name}")
print(f"📝 พารามิเตอร์: {function_args}")
# เรียกใช้ฟังก์ชันตามชื่อ
if function_name == "search_web":
result = search_web(**function_args)
elif function_name == "summarize_content":
result = summarize_content(**function_args)
else:
result = "ไม่พบฟังก์ชันที่ระบุ"
# ส่งผลลัพธ์กลับไปยังโมเดล
messages.append(message)
messages.append({
"role": "tool",
"content": json.dumps(result, ensure_ascii=False)
})
# สตรีมการตอบกลับหลังจากได้ผลลัพธ์
final_stream = ollama.chat(
model='nemotron-3-nano',
messages=messages,
stream=True
)
for final_chunk in final_stream:
if 'message' in final_chunk and 'content' in final_chunk['message']:
print(final_chunk['message']['content'], end='', flush=True)
# แสดงเนื้อหาที่สตรีมมา
elif 'content' in message and message['content']:
print(message['content'], end='', flush=True)
# ทดสอบการใช้งาน
async def main():
query = "ค้นหาข้อมูลล่าสุดเกี่ยวกับ AI ในประเทศไทยและสรุปแนวโน้มสำคัญ"
await streaming_research_agent(query)
if __name__ == "__main__":
asyncio.run(main())
ข้อดีของการใช้ Ollama Streaming Tool Calling
- ประหยัดเวลา: ไม่ต้องรอให้โมเดลคิดเสร็จทั้งหมด สามารถเริ่มประมวลผลได้ทันที
- ประหยัดทรัพยากร: ใช้โมเดลขนาดเล็กอย่าง Nemotron 3 Nano ลดภาระฮาร์ดแวร์
- ความเป็นส่วนตัว: ทำงานออฟไลน์ 100% ไม่ส่งข้อมูลออกไปยังเซิร์ฟเวอร์ภายนอก
- ความยืดหยุ่น: สามารถเพิ่ม tool ใหม่ๆ ได้ตามต้องการ
ตัวอย่างการใช้งานในทีมไทย
ลองนึกถึงทีมการตลาดที่ต้องวิเคราะห์คู่แข่งเป็นประจำ ด้วย Research Agent นี้ ทีมสามารถ:
- ค้นหาข่าวสารและบทความเกี่ยวกับคู่แข่งได้อัตโนมัติ
- สรุปเนื้อหาสำคัญเป็นภาษาไทยทันที
- รวบรวมข้อมูลเป็นรายงานโดยไม่ต้องส่งข้อมูลออกไปยังบริการภายนอก
ข้อจำกัดและความท้าทาย
- คุณภาพการค้นหา: การค้นหาด้วยวิธี scraping อาจไม่ได้ผลลัพธ์ดีเท่า API อย่างเป็นทางการ
- ความเร็วในการประมวลผล: ขึ้นอยู่กับฮาร์ดแวร์ของคุณ โมเดลอาจช้าลงบนเครื่องที่ไม่มี GPU
- ความแม่นยำของ Tool Calling: โมเดลขนาดเล็กอาจเรียกใช้ tool ผิดพลาดได้บ้าง
สรุป
การสร้าง Realtime Research Agent ด้วย Ollama 0.13+ Streaming Tool Calling และโมเดล Nemotron 3 Nano ร่วมกับ FunctionGemma เป็นวิธีที่ดีในการสร้างเครื่องมือวิจัยที่รักษาความเป็นส่วนตัวของข้อมูล แม้จะมีข้อจำกัดบางประการ แต่ก็เป็นจุดเริ่มต้นที่ดีสำหรับทีมที่ต้องการควบคุมข้อมูลของตนเอง
ลองนำโค้ดไปปรับใช้กับทีมของคุณ และอย่าลืมแชร์ประสบการณ์หรือปัญหาที่พบในชุมชนนักพัฒนาไทย เพื่อร่วมกันพัฒนาเครื่องมือให้ดีขึ้น
เนื้อหาที่จัดทำโดยมี AI ช่วยจะมีป้ายกำกับ "เรียบเรียงโดยมี AI ช่วย" เพื่อให้คุณทราบอย่างชัดเจน เราถือว่าความโปร่งใสเรื่องการใช้ AI เป็นสิ่งสำคัญต่อความไว้วางใจของผู้อ่าน
ความคิดเห็น (0)
ยังไม่มีความคิดเห็น — มาเป็นคนแรกกันเถอะ!