File size: 4,398 Bytes
ece0cc5
 
 
4964210
 
ece0cc5
 
 
 
 
4964210
 
 
 
 
 
 
ece0cc5
 
 
4964210
 
 
 
 
 
 
ece0cc5
 
4964210
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ece0cc5
4964210
 
ece0cc5
4964210
 
 
 
 
 
ece0cc5
4964210
 
 
 
 
 
 
 
ece0cc5
4964210
 
ece0cc5
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
import uvicorn
from fastapi import FastAPI, Body
from sentence_transformers import SentenceTransformer
from sklearn.metrics.pairwise import cosine_similarity
import numpy as np

# Tải model 1 lần duy nhất khi Server khởi động
model = SentenceTransformer('all-MiniLM-L6-v2')
print("✅ Model đã sẵn sàng!")

product_db = {
    "ids": [],     
    "vectors": None,
    "cached_result": {}
}

#App
app = FastAPI()

@app.get("/")
async def home():
    return {"status": "AI Server is RUNNING", "model": "all-MiniLM-L6-v2"}

@app.get("/status")
async def status():
    count = len(product_db["ids"])
    return {"count": count}

@app.post("/vectorize")
async def vectorize_products(payload: dict = Body(...)):
    products = payload.get("products", [])
    k_item = payload.get("k_item",5)
    if not products:
        return {"error": "Danh sách sản phẩm trống"}
    product_names = [x['name'] for x in products]
    product_ids = [x['id'] for x in products]
    # Tính ra vector
    vectors = model.encode(product_names)
    product_db["vectors"] = vectors # lưu lại

    product_db["ids"] = product_ids
    
    sim_matrix = cosine_similarity(vectors)
    res = {}
    # Duyệt qua từng dòng của ma trận
    for i in range(len(product_ids)):
        current_id = product_ids[i]
        scores = sim_matrix[i]
        
        # Sắp xếp index giảm dần
        sorted_indices = np.argsort(scores)[::-1]
        
        # Lấy top k (bỏ qua index 0 là chính nó)
        top_indices = sorted_indices[1 : k_item + 1]
        
        # Map từ Index -> ID thật
        similar_ids = [product_ids[idx] for idx in top_indices]
        
        # Lưu vào dict kết quả (Key phải là string)
        res[str(current_id)] = [int(x) for x in similar_ids]
    product_db['cached_result'] = res 
    # Trả về đúng cấu trúc AiResponse bên Java
    return {"result": res}

@app.get("/refresh_cache")
async def refresh_cache():
    return {"result": product_db.get("cached_result", {})}

@app.post("/add_product")
async def add_product(payload: dict = Body(...)):
    # 1. Lấy dữ liệu đầu vào
    new_id = payload.get("id")
    new_name = payload.get("name")
    
    if new_id is None or not new_name:
        return {"error": "Dữ liệu không hợp lệ"}

    print(f"⚡ Đang thêm sản phẩm mới: ID {new_id} - {new_name}")

    # 2. Vector hóa CHỈ 1 SẢN PHẨM MỚI (Cực nhanh)
    # Kết quả là 1 vector (Shape: 1 x 768)
    new_vector = model.encode([new_name])

    # 3. Lấy kho dữ liệu cũ từ RAM
    existing_vectors = product_db["vectors"]
    existing_ids = product_db["ids"]
    cached_result = product_db["cached_result"]

    # Trường hợp chưa Init dữ liệu thì coi như init mới
    if existing_vectors is None or len(existing_ids) == 0:
        return {"result": {}}

    # 4. TÍNH TOÁN TƯƠNG ĐỒNG (Điểm mấu chốt của tốc độ)
    # So sánh vector mới với toàn bộ vector cũ
    # cosine_similarity(new, old) -> Trả về mảng điểm số
    sim_scores = cosine_similarity(new_vector, existing_vectors)[0]

    # 5. Tìm Top K cho sản phẩm mới
    k_item = 5
    # Sắp xếp giảm dần và lấy top k
    top_indices = np.argsort(sim_scores)[::-1][:k_item]
    
    # Map từ index sang ID thật
    similar_ids = [int(existing_ids[idx]) for idx in top_indices]

    # 6. CẬP NHẬT DATABASE TRONG RAM (Để lần sau thêm tiếp còn dùng)
    # Thêm ID mới vào danh sách
    product_db["ids"].append(new_id)
    
    # Nối vector mới vào ma trận cũ (Append vào dòng cuối)
    product_db["vectors"] = np.vstack([existing_vectors, new_vector])
    
    # Cập nhật Cache kết quả
    result_map = {str(new_id): similar_ids}
    
    # (Optional) Update ngược: Nếu sản phẩm cũ thấy sản phẩm mới này giống nó
    # Đoạn này làm phức tạp hơn, tạm thời ta chỉ quan tâm chiều xuôi: New -> Old
    cached_result.update(result_map) 

    print(f"✅ Đã thêm xong ID {new_id}. Tìm thấy {len(similar_ids)} sản phẩm tương tự.")
    
    # Trả về kết quả để Java update cache
    return {"result": result_map}

if __name__ == "__main__":
    # Hugging Face yêu cầu bắt buộc chạy ở port 7860
    uvicorn.run(app, host="0.0.0.0", port=7860)