Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
53 changes: 27 additions & 26 deletions chatbot/services/search/vocabularies.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,56 +11,57 @@

import logging

from django.core.cache import cache

from chatbot.models.company_models import Company
from chatbot.models.enums import EntityStatus, FileTypeChoices
from chatbot.services.search.config import (
VOCAB_AUTO,
VOCAB_CANDIDATES,
VOCAB_FULL,
get_search_llm_setting,
)
from chatbot.utils.company_cache import get_all_companies

logger = logging.getLogger('django')

CACHE_TTL_SECONDS = 300
ORG_CACHE_KEY = 'ai_search:org_vocab:{scope}'


def organization_vocabulary(scope=None):
"""
``{slug: [display name]}`` for every active organization, cached per scope.
``{slug: [display name]}`` for every active organization.

``scope`` is unused today (search is global) but keeps the cache key ready
for tenant-scoped search, so an unscoped cache can't leak across tenants later.
The organization list comes from Vishwa's Redis-backed company cache
(`get_all_companies`) instead of SEARCH_FILTER_ORGANIZATIONS. When the Redis
cache is cold, that helper hydrates it from the database.
"""
key = ORG_CACHE_KEY.format(scope=scope or 'global')
cached = cache.get(key)
if cached is not None:
return cached

rows = (
Company.objects
.filter(status=EntityStatus.ACTIVE)
.exclude(slug__isnull=True).exclude(slug='')
.values_list('slug', 'name')
)
# Skip inactive orgs: suggesting one would offer a filter that matches nothing.
vocabulary = {slug: [name] for slug, name in rows if slug}

cache.set(key, vocabulary, CACHE_TTL_SECONDS)
return vocabulary
return {
company.slug: [company.name]
for company in get_all_companies()
if (
company.slug
and company.status == EntityStatus.ACTIVE
)
}


def file_type_vocabulary():
"""``{mime: [label, ext, .ext]}`` from the static FileTypeChoices enum."""
vector_type_aliases = {
FileTypeChoices.CSV.value: ["project_task"],

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@Prajwal17Tunerlabs why we added aliases here ?

FileTypeChoices.XLS.value: ["xlsx_rag_optimized"],
FileTypeChoices.XLSX.value: ["xlsx_rag_optimized"],
FileTypeChoices.TXT.value: ["text", "markdown"],
}
extensions = FileTypeChoices.get_extension_mapping()
vocabulary = {}
for choice in FileTypeChoices:
dotted = extensions.get(choice, '')
bare = dotted.lstrip('.')
vocabulary[choice.value] = [str(choice.label), bare, dotted]
label = str(choice.label)
aliases = [label, bare, dotted]
aliases.extend(vector_type_aliases.get(choice.value, []))
if label:
aliases.append(f"{label}s")
if bare:
aliases.append(f"{bare}s")
vocabulary[choice.value] = list(dict.fromkeys(aliases))
return vocabulary


Expand Down
4 changes: 2 additions & 2 deletions chatbot/utils/chat_query_handler.py
Original file line number Diff line number Diff line change
Expand Up @@ -126,7 +126,7 @@ def query_database_with_metadata(
file_type: List[str] = None,
exclude_organizations: List[str] = None,
exclude_file_type: List[str] = None,
any_of: List[Dict[str, Any]] = None
any_of: List[Dict[str, Any]] = None,
):
"""
Query vector database with metadata filters for media search v2.
Expand All @@ -152,7 +152,7 @@ def query_database_with_metadata(
"filter_score": filter_score,
"detail_filter_score": detail_filter_score
}

# Add query only if provided
if query:
data["query"] = query
Expand Down
Loading