ATS-SCORE-CHECKER / embeddings /vector_store.py
Akash-Dragon's picture
Initial commit: Resume ATS Score Checker API
4689a82
Raw
History Blame Contribute Delete
3.58 kB
import chromadb
from typing import List, Dict
import numpy as np
from config import settings
class VectorStore:
"""ChromaDB vector store for resumes and job descriptions"""
def __init__(self):
"""Initialize ChromaDB client with persistent storage."""
# Use new ChromaDB API (1.4.1+)
self.client = chromadb.PersistentClient(
path=settings.CHROMA_PERSIST_DIRECTORY
)
# Create collections
self.resume_collection = self.client.get_or_create_collection(
name="resumes",
metadata={"description": "Resume embeddings"}
)
self.job_collection = self.client.get_or_create_collection(
name="jobs",
metadata={"description": "Job description embeddings"}
)
def add_resume_chunks(self, resume_id: int, chunks: List[Dict], embeddings: np.ndarray):
"""Add resume chunks to vector store"""
ids = [f"resume_{resume_id}_chunk_{i}" for i in range(len(chunks))]
documents = [chunk['text'] for chunk in chunks]
metadatas = [
{
'resume_id': resume_id,
'section': chunk['section'],
'chunk_type': chunk['chunk_type'],
'position': chunk['position']
}
for chunk in chunks
]
self.resume_collection.add(
ids=ids,
embeddings=embeddings.tolist(),
documents=documents,
metadatas=metadatas
)
def add_job_chunks(self, job_id: int, chunks: List[Dict], embeddings: np.ndarray):
"""Add job description chunks to vector store"""
ids = [f"job_{job_id}_chunk_{i}" for i in range(len(chunks))]
documents = [chunk['text'] for chunk in chunks]
metadatas = [
{
'job_id': job_id,
'section': chunk.get('section', 'description'),
'position': chunk.get('position', i)
}
for i, chunk in enumerate(chunks)
]
self.job_collection.add(
ids=ids,
embeddings=embeddings.tolist(),
documents=documents,
metadatas=metadatas
)
def search_similar_chunks(self, query_embedding: np.ndarray, collection_name: str = "resumes",
n_results: int = 10, filter_metadata: Dict = None):
"""Search for similar chunks"""
collection = self.resume_collection if collection_name == "resumes" else self.job_collection
results = collection.query(
query_embeddings=[query_embedding.tolist()],
n_results=n_results,
where=filter_metadata
)
return results
def get_resume_chunks(self, resume_id: int):
"""Get all chunks for a resume"""
results = self.resume_collection.get(
where={"resume_id": resume_id}
)
return results
def get_job_chunks(self, job_id: int):
"""Get all chunks for a job"""
results = self.job_collection.get(
where={"job_id": job_id}
)
return results
def delete_resume_chunks(self, resume_id: int):
"""Delete all chunks for a resume"""
self.resume_collection.delete(
where={"resume_id": resume_id}
)
def delete_job_chunks(self, job_id: int):
"""Delete all chunks for a job"""
self.job_collection.delete(
where={"job_id": job_id}
)