You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
362 lines
14 KiB
362 lines
14 KiB
import json
|
|
import re
|
|
import sys
|
|
import time
|
|
import requests
|
|
import threading
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
|
|
from django.core.management.base import BaseCommand
|
|
from apps.dobodbi_calendar.models import CalendarOccasions
|
|
|
|
# Target languages from زبانها.md
|
|
TARGET_LANGUAGES = {
|
|
"ar": "Arabic (عربی)",
|
|
"tr": "Turkish (ترکی)",
|
|
"fa": "Persian (فارسی)",
|
|
"ur": "Urdu (اردو)",
|
|
"bn": "Bengali (بنگالی)",
|
|
"id": "Indonesian (اندونزیایی)",
|
|
"fr": "French (فرانسوی)",
|
|
"ru": "Russian (روسی)",
|
|
"uz": "Uzbek (ازبکی)",
|
|
"ky": "Kyrgyz (قرقیزی)",
|
|
"tg": "Tajik (تاجیکی)",
|
|
"az": "Azerbaijani (آذربایجانی)",
|
|
"en": "English (انگلیسی)",
|
|
"de": "German (آلمانی)",
|
|
"zh": "Mandarin Chinese (چینی ماندارین)",
|
|
"ha": "Hausa (هوسا)",
|
|
"sw": "Swahili (سواحیلی)",
|
|
"es": "Spanish (اسپانیایی)",
|
|
}
|
|
|
|
DEFAULT_API_URL = "http://ai.newhorizonco.uk/v1/chat/completions"
|
|
DEFAULT_API_KEY = "hZwAGW4H8v87Ol8tXvqEEY"
|
|
DEFAULT_MODEL = "gemini-3.6-flash-high"
|
|
|
|
|
|
class ThreadSafeRateLimiter:
|
|
"""
|
|
Thread-safe sliding window rate limiter.
|
|
Guarantees max `max_calls` executions per `period` seconds across all threads.
|
|
"""
|
|
def __init__(self, max_calls: int = 5, period: float = 1.0):
|
|
self.max_calls = max_calls
|
|
self.period = period
|
|
self.timestamps = []
|
|
self.lock = threading.Lock()
|
|
|
|
def acquire(self):
|
|
with self.lock:
|
|
while True:
|
|
now = time.time()
|
|
# Remove timestamps outside the sliding window
|
|
self.timestamps = [t for t in self.timestamps if now - t < self.period]
|
|
if len(self.timestamps) < self.max_calls:
|
|
self.timestamps.append(now)
|
|
break
|
|
# Wait until the oldest timestamp drops out of the 1-second window
|
|
sleep_time = self.period - (now - self.timestamps[0]) + 0.005
|
|
if sleep_time > 0:
|
|
time.sleep(sleep_time)
|
|
|
|
|
|
class Command(BaseCommand):
|
|
help = "Translates CalendarOccasions titles into target languages concurrently (5 req/sec)."
|
|
|
|
def add_arguments(self, parser):
|
|
parser.add_argument(
|
|
"--rate-limit",
|
|
type=int,
|
|
default=5,
|
|
help="Maximum API requests per second (default: 5)",
|
|
)
|
|
parser.add_argument(
|
|
"--workers",
|
|
type=int,
|
|
default=5,
|
|
help="Number of concurrent worker threads (default: 5)",
|
|
)
|
|
parser.add_argument(
|
|
"--limit",
|
|
type=int,
|
|
default=0,
|
|
help="Limit number of occasions to translate (0 = all)",
|
|
)
|
|
parser.add_argument(
|
|
"--force",
|
|
action="store_true",
|
|
help="Force re-translation of all target languages even if present",
|
|
)
|
|
parser.add_argument(
|
|
"--api-url",
|
|
type=str,
|
|
default=DEFAULT_API_URL,
|
|
help="API endpoint URL",
|
|
)
|
|
parser.add_argument(
|
|
"--api-key",
|
|
type=str,
|
|
default=DEFAULT_API_KEY,
|
|
help="API bearer token",
|
|
)
|
|
parser.add_argument(
|
|
"--model",
|
|
type=str,
|
|
default=DEFAULT_MODEL,
|
|
help="AI model name",
|
|
)
|
|
|
|
def safe_write(self, msg: str, style_func=None):
|
|
"""Helper to write to stdout safely, catching Windows charmap encoding errors."""
|
|
if style_func:
|
|
msg = style_func(msg)
|
|
try:
|
|
self.stdout.write(msg)
|
|
except Exception:
|
|
enc = getattr(sys.stdout, "encoding", "utf-8") or "utf-8"
|
|
safe_msg = msg.encode(enc, errors="replace").decode(enc)
|
|
try:
|
|
self.stdout.write(safe_msg)
|
|
except Exception:
|
|
pass
|
|
|
|
def handle(self, *args, **options):
|
|
rate_limit = options["rate_limit"]
|
|
max_workers = options["workers"]
|
|
limit = options["limit"]
|
|
force = options["force"]
|
|
api_url = options["api_url"]
|
|
api_key = options["api_key"]
|
|
model = options["model"]
|
|
|
|
self.safe_write(
|
|
f"Starting Calendar Occasions translation script...\n"
|
|
f" Rate Limit: {rate_limit} req/sec\n"
|
|
f" Workers: {max_workers}\n"
|
|
f" Target Languages: {len(TARGET_LANGUAGES)} languages\n"
|
|
f" Force Re-translate: {force}",
|
|
self.style.WARNING,
|
|
)
|
|
|
|
rate_limiter = ThreadSafeRateLimiter(max_calls=rate_limit, period=1.0)
|
|
|
|
queryset = CalendarOccasions.objects.all().order_by("id")
|
|
if limit > 0:
|
|
queryset = queryset[:limit]
|
|
|
|
occasions = list(queryset)
|
|
total_count = len(occasions)
|
|
|
|
if total_count == 0:
|
|
self.safe_write("No calendar occasions found in database.", self.style.WARNING)
|
|
return
|
|
|
|
self.safe_write(f"Found {total_count} occasion objects to process.", self.style.SUCCESS)
|
|
|
|
success_count = 0
|
|
skipped_count = 0
|
|
failed_count = 0
|
|
lock = threading.Lock()
|
|
|
|
def process_occasion(occasion):
|
|
nonlocal success_count, skipped_count, failed_count
|
|
|
|
# Extract current title field structure
|
|
current_titles = occasion.title
|
|
if isinstance(current_titles, str):
|
|
current_titles = [{"text": current_titles, "title": current_titles, "language_code": "fa"}]
|
|
elif not isinstance(current_titles, list):
|
|
current_titles = []
|
|
|
|
# Map existing language_codes
|
|
existing_by_lang = {}
|
|
for item in current_titles:
|
|
if isinstance(item, dict) and "language_code" in item:
|
|
existing_by_lang[item["language_code"]] = item
|
|
|
|
# Determine source Persian text (or any available text)
|
|
source_text = None
|
|
if "fa" in existing_by_lang:
|
|
source_text = existing_by_lang["fa"].get("title") or existing_by_lang["fa"].get("text")
|
|
if not source_text and current_titles:
|
|
first_item = current_titles[0]
|
|
if isinstance(first_item, dict):
|
|
source_text = first_item.get("title") or first_item.get("text")
|
|
elif isinstance(first_item, str):
|
|
source_text = first_item
|
|
|
|
if not source_text:
|
|
with lock:
|
|
skipped_count += 1
|
|
self.safe_write(f"[ID {occasion.id}] Skipped - No source text available", self.style.NOTICE)
|
|
return
|
|
|
|
# Find missing target languages
|
|
if force:
|
|
missing_langs = list(TARGET_LANGUAGES.keys())
|
|
else:
|
|
missing_langs = [
|
|
lang for lang in TARGET_LANGUAGES.keys()
|
|
if lang not in existing_by_lang or not (existing_by_lang[lang].get("title") or existing_by_lang[lang].get("text"))
|
|
]
|
|
|
|
if not missing_langs:
|
|
with lock:
|
|
skipped_count += 1
|
|
self.safe_write(
|
|
f"[ID {occasion.id}] Skipped - All {len(TARGET_LANGUAGES)} languages already present ({source_text})",
|
|
self.style.SUCCESS,
|
|
)
|
|
return
|
|
|
|
# Call AI API to translate missing languages
|
|
translations = self.call_translation_api(
|
|
source_text=source_text,
|
|
missing_langs=missing_langs,
|
|
rate_limiter=rate_limiter,
|
|
api_url=api_url,
|
|
api_key=api_key,
|
|
model=model,
|
|
)
|
|
|
|
if not translations:
|
|
with lock:
|
|
failed_count += 1
|
|
self.safe_write(
|
|
f"[ID {occasion.id}] Failed - API returned no translations for '{source_text}'",
|
|
self.style.ERROR,
|
|
)
|
|
return
|
|
|
|
# Merge translations into occasion.title
|
|
for lang_code, translated_val in translations.items():
|
|
if translated_val and isinstance(translated_val, str):
|
|
clean_val = translated_val.strip()
|
|
existing_by_lang[lang_code] = {
|
|
"text": clean_val,
|
|
"title": clean_val,
|
|
"language_code": lang_code,
|
|
}
|
|
|
|
# Reconstruct title array keeping TARGET_LANGUAGES order first
|
|
new_title_list = []
|
|
for code in TARGET_LANGUAGES.keys():
|
|
if code in existing_by_lang:
|
|
new_title_list.append(existing_by_lang.pop(code))
|
|
|
|
# Append any remaining unexpected language objects
|
|
for item in existing_by_lang.values():
|
|
new_title_list.append(item)
|
|
|
|
occasion.title = new_title_list
|
|
occasion.save(update_fields=["title", "updated_at"])
|
|
|
|
with lock:
|
|
success_count += 1
|
|
self.safe_write(
|
|
f"[ID {occasion.id}] Translated successfully into {len(translations)} languages ('{source_text}')",
|
|
self.style.SUCCESS,
|
|
)
|
|
|
|
# Run concurrent workers
|
|
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
futures = [executor.submit(process_occasion, occ) for occ in occasions]
|
|
for future in as_completed(futures):
|
|
try:
|
|
future.result()
|
|
except Exception as exc:
|
|
self.safe_write(f"Unhandled error in thread: {exc}", self.style.ERROR)
|
|
|
|
self.safe_write(
|
|
f"\nTranslation completed!\n"
|
|
f" Total Processed: {total_count}\n"
|
|
f" Successfully Updated: {success_count}\n"
|
|
f" Skipped (Up to date): {skipped_count}\n"
|
|
f" Failed: {failed_count}",
|
|
self.style.SUCCESS,
|
|
)
|
|
|
|
def call_translation_api(
|
|
self,
|
|
source_text: str,
|
|
missing_langs: list,
|
|
rate_limiter: ThreadSafeRateLimiter,
|
|
api_url: str,
|
|
api_key: str,
|
|
model: str,
|
|
max_retries: int = 3,
|
|
) -> dict:
|
|
"""
|
|
Calls the AI completions API to translate source_text into missing_langs.
|
|
Enforces rate limiting prior to each request attempt.
|
|
"""
|
|
targets_desc = json.dumps({code: TARGET_LANGUAGES[code] for code in missing_langs}, ensure_ascii=False)
|
|
|
|
prompt = (
|
|
f"You are a professional translator for calendar events and occasions.\n"
|
|
f"Translate the following calendar occasion title from Persian into the specified target languages.\n\n"
|
|
f"Original Title: \"{source_text}\"\n\n"
|
|
f"Target Languages (JSON map of code -> Language Name):\n{targets_desc}\n\n"
|
|
f"INSTRUCTIONS:\n"
|
|
f"1. Translate accurately into each requested language code.\n"
|
|
f"2. Output strictly a JSON object mapping language_code to translation string.\n"
|
|
f"3. Do NOT include markdown code fences (```json or ```), explanations, or outside text.\n\n"
|
|
f"Example Output:\n"
|
|
f"{{\"en\": \"New Year\", \"ar\": \"السنة الجديدة\"}}"
|
|
)
|
|
|
|
headers = {
|
|
"Authorization": f"Bearer {api_key}",
|
|
"Content-Type": "application/json",
|
|
}
|
|
payload = {
|
|
"model": model,
|
|
"stream": False,
|
|
"messages": [
|
|
{
|
|
"role": "user",
|
|
"content": prompt,
|
|
}
|
|
],
|
|
}
|
|
|
|
for attempt in range(1, max_retries + 1):
|
|
rate_limiter.acquire()
|
|
try:
|
|
response = requests.post(api_url, headers=headers, json=payload, timeout=20)
|
|
if response.status_code == 200:
|
|
res_data = response.json()
|
|
raw_content = res_data["choices"][0]["message"]["content"].strip()
|
|
|
|
# Clean any markdown code blocks if present
|
|
if raw_content.startswith("```"):
|
|
lines = raw_content.splitlines()
|
|
if lines and lines[0].startswith("```"):
|
|
lines = lines[1:]
|
|
if lines and lines[-1].startswith("```"):
|
|
lines = lines[:-1]
|
|
raw_content = "\n".join(lines).strip()
|
|
|
|
try:
|
|
parsed = json.loads(raw_content)
|
|
if isinstance(parsed, dict):
|
|
# Filter only requested target language keys
|
|
return {k: v for k, v in parsed.items() if k in missing_langs and isinstance(v, str)}
|
|
except json.JSONDecodeError:
|
|
# Regex extract JSON substring if extra text was included
|
|
match = re.search(r"\{.*\}", raw_content, re.DOTALL)
|
|
if match:
|
|
try:
|
|
parsed = json.loads(match.group(0))
|
|
if isinstance(parsed, dict):
|
|
return {k: v for k, v in parsed.items() if k in missing_langs and isinstance(v, str)}
|
|
except json.JSONDecodeError:
|
|
pass
|
|
except Exception as e:
|
|
if attempt == max_retries:
|
|
self.safe_write(f"API Call exception on attempt {attempt}: {e}", self.style.ERROR)
|
|
time.sleep(1)
|
|
|
|
return {}
|