# ingest_knowledge.py
# python routes/ingest_knowledge.py
import chromadb
import os

# 1. 自動獲取路徑：確保能找到同資料夾下的 knowledge.txt
current_dir = os.path.dirname(os.path.abspath(__file__))
file_path = os.path.join(current_dir, 'knowledge.txt')

# 2. 初始化 ChromaDB (建議存在專案根目錄，方便 chat.py 讀取)
# 我們把 db 存在根目錄的 chroma_db 資料夾
db_path = os.path.join(os.path.dirname(current_dir), "chroma_db")
client = chromadb.PersistentClient(path=db_path)
collection = client.get_or_create_collection(name="care_knowledge")

# 3. 讀取文件
if os.path.exists(file_path):
    print(f"📖 正在讀取文件: {file_path}")
    with open(file_path, 'r', encoding='utf-8') as f:
        lines = f.readlines()
    
    # 過濾掉空行
    documents = [line.strip() for line in lines if line.strip()]
    ids = [f"id_{i}" for i in range(len(documents))]

    if documents:
        # 4. 寫入資料庫
        collection.add(
            documents=documents,
            ids=ids
        )
        print(f"✅ 成功向量化 {len(documents)} 條知識片段！")
        print(f"📁 資料庫路徑: {db_path}")
    else:
        print("⚠️ 檔案是空的，沒有內容可以匯入。")
else:
    print(f"❌ 錯誤：在 {file_path} 找不到 knowledge.txt")