mejora: rendimiento captura RSS y velocidad de traducción
- Ingestor: reactivar ETag/Last-Modified (304), http.Client compartido, rate-limit por host y Retry-After - Traducción: batch 128, cache en Redis con TTL (efímera), commit por lote - Scraper: worker pool paralelo (SCRAPER_WORKERS) - Config: compose lee RSS_/TRANSLATOR_ variables del .env (fuente única)
This commit is contained in:
parent
71d345ec35
commit
e3936fd4cf
6 changed files with 239 additions and 48 deletions
1
.gitignore
vendored
1
.gitignore
vendored
|
|
@ -67,6 +67,7 @@ models/nllb-ct2-1.3b/
|
||||||
# Wikipedia images - downloaded by wiki_worker
|
# Wikipedia images - downloaded by wiki_worker
|
||||||
data/wiki_images/
|
data/wiki_images/
|
||||||
data/backups/
|
data/backups/
|
||||||
|
backups/
|
||||||
|
|
||||||
# Database storage (PostgreSQL)
|
# Database storage (PostgreSQL)
|
||||||
data/pgdata/
|
data/pgdata/
|
||||||
|
|
|
||||||
|
|
@ -23,7 +23,8 @@ RUN pip install --no-cache-dir \
|
||||||
protobuf==3.20.3 \
|
protobuf==3.20.3 \
|
||||||
"numpy<2" \
|
"numpy<2" \
|
||||||
psycopg2-binary \
|
psycopg2-binary \
|
||||||
langdetect
|
langdetect \
|
||||||
|
redis
|
||||||
|
|
||||||
# === ARREGLAR EL EXECUTABLE STACK ===
|
# === ARREGLAR EL EXECUTABLE STACK ===
|
||||||
RUN find /usr/local/lib/python3.11/site-packages/ctranslate2* \
|
RUN find /usr/local/lib/python3.11/site-packages/ctranslate2* \
|
||||||
|
|
|
||||||
|
|
@ -10,6 +10,7 @@ import (
|
||||||
"os/signal"
|
"os/signal"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
"sync"
|
||||||
"syscall"
|
"syscall"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
|
@ -25,6 +26,7 @@ var (
|
||||||
sleepInterval = 60
|
sleepInterval = 60
|
||||||
batchSize = 10
|
batchSize = 10
|
||||||
enrichLimit = 20
|
enrichLimit = 20
|
||||||
|
scraperWorkers = 5
|
||||||
)
|
)
|
||||||
|
|
||||||
type URLSource struct {
|
type URLSource struct {
|
||||||
|
|
@ -63,6 +65,7 @@ func loadConfig() {
|
||||||
sleepInterval = getEnvInt("SCRAPER_SLEEP", 60)
|
sleepInterval = getEnvInt("SCRAPER_SLEEP", 60)
|
||||||
batchSize = getEnvInt("SCRAPER_BATCH", 10)
|
batchSize = getEnvInt("SCRAPER_BATCH", 10)
|
||||||
enrichLimit = getEnvInt("SCRAPER_ENRICH_LIMIT", 20)
|
enrichLimit = getEnvInt("SCRAPER_ENRICH_LIMIT", 20)
|
||||||
|
scraperWorkers = getEnvInt("SCRAPER_WORKERS", 5)
|
||||||
}
|
}
|
||||||
|
|
||||||
func getEnvInt(key string, defaultValue int) int {
|
func getEnvInt(key string, defaultValue int) int {
|
||||||
|
|
@ -406,6 +409,41 @@ func processSource(ctx context.Context, source URLSource) {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// enrichConcurrent enriches noticias in parallel with a bounded worker pool.
|
||||||
|
func enrichConcurrent(ctx context.Context, noticias []Noticia) {
|
||||||
|
limiter := make(chan struct{}, scraperWorkers)
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
for _, noticia := range noticias {
|
||||||
|
wg.Add(1)
|
||||||
|
limiter <- struct{}{}
|
||||||
|
go func(n Noticia) {
|
||||||
|
defer wg.Done()
|
||||||
|
defer func() { <-limiter }()
|
||||||
|
if processEnrichment(ctx, n) {
|
||||||
|
time.Sleep(1 * time.Second)
|
||||||
|
}
|
||||||
|
}(noticia)
|
||||||
|
}
|
||||||
|
wg.Wait()
|
||||||
|
}
|
||||||
|
|
||||||
|
// scrapeSources processes fuentes_url in parallel with a bounded worker pool.
|
||||||
|
func scrapeSources(ctx context.Context, sources []URLSource) {
|
||||||
|
limiter := make(chan struct{}, scraperWorkers)
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
for _, source := range sources {
|
||||||
|
wg.Add(1)
|
||||||
|
limiter <- struct{}{}
|
||||||
|
go func(src URLSource) {
|
||||||
|
defer wg.Done()
|
||||||
|
defer func() { <-limiter }()
|
||||||
|
time.Sleep(1 * time.Second) // polite pacing across workers
|
||||||
|
processSource(ctx, src)
|
||||||
|
}(source)
|
||||||
|
}
|
||||||
|
wg.Wait()
|
||||||
|
}
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
loadConfig()
|
loadConfig()
|
||||||
logger.Println("Starting Scraper Worker")
|
logger.Println("Starting Scraper Worker")
|
||||||
|
|
@ -443,12 +481,8 @@ func main() {
|
||||||
if err != nil {
|
if err != nil {
|
||||||
logger.Printf("Error fetching noticias to enrich: %v", err)
|
logger.Printf("Error fetching noticias to enrich: %v", err)
|
||||||
} else if len(noticias) > 0 {
|
} else if len(noticias) > 0 {
|
||||||
logger.Printf("Enriching %d noticias", len(noticias))
|
logger.Printf("Enriching %d noticias (%d workers)", len(noticias), scraperWorkers)
|
||||||
for _, noticia := range noticias {
|
enrichConcurrent(ctx, noticias)
|
||||||
if processEnrichment(ctx, noticia) {
|
|
||||||
time.Sleep(3 * time.Second)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
sources, err := getActiveURLs(ctx)
|
sources, err := getActiveURLs(ctx)
|
||||||
|
|
@ -462,12 +496,8 @@ func main() {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
logger.Printf("Processing %d sources", len(sources))
|
logger.Printf("Processing %d sources (%d workers)", len(sources), scraperWorkers)
|
||||||
|
scrapeSources(ctx, sources)
|
||||||
for _, source := range sources {
|
|
||||||
processSource(ctx, source)
|
|
||||||
time.Sleep(2 * time.Second) // Rate limiting
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -92,8 +92,9 @@ services:
|
||||||
DB_NAME: ${DB_NAME:-rss}
|
DB_NAME: ${DB_NAME:-rss}
|
||||||
DB_USER: ${DB_USER:-rss}
|
DB_USER: ${DB_USER:-rss}
|
||||||
DB_PASS: ${DB_PASS}
|
DB_PASS: ${DB_PASS}
|
||||||
RSS_MAX_WORKERS: 100
|
RSS_MAX_WORKERS: ${RSS_MAX_WORKERS:-100}
|
||||||
RSS_POKE_INTERVAL_MIN: 60
|
RSS_POKE_INTERVAL_MIN: ${RSS_POKE_INTERVAL_MIN:-60}
|
||||||
|
RSS_HOST_DELAY_MS: ${RSS_HOST_DELAY_MS:-500}
|
||||||
TZ: Europe/Madrid
|
TZ: Europe/Madrid
|
||||||
networks:
|
networks:
|
||||||
- backend
|
- backend
|
||||||
|
|
@ -155,6 +156,7 @@ services:
|
||||||
SCRAPER_SLEEP: 60
|
SCRAPER_SLEEP: 60
|
||||||
SCRAPER_BATCH: 10
|
SCRAPER_BATCH: 10
|
||||||
SCRAPER_ENRICH_LIMIT: 20
|
SCRAPER_ENRICH_LIMIT: 20
|
||||||
|
SCRAPER_WORKERS: ${SCRAPER_WORKERS:-5}
|
||||||
TZ: Europe/Madrid
|
TZ: Europe/Madrid
|
||||||
networks:
|
networks:
|
||||||
- backend
|
- backend
|
||||||
|
|
@ -304,7 +306,7 @@ services:
|
||||||
DB_USER: ${DB_USER:-rss}
|
DB_USER: ${DB_USER:-rss}
|
||||||
DB_PASS: ${DB_PASS}
|
DB_PASS: ${DB_PASS}
|
||||||
TARGET_LANGS: es
|
TARGET_LANGS: es
|
||||||
TRANSLATOR_BATCH: 32
|
TRANSLATOR_BATCH: ${TRANSLATOR_BATCH:-128}
|
||||||
CT2_MODEL_PATH: /app/models/nllb-ct2
|
CT2_MODEL_PATH: /app/models/nllb-ct2
|
||||||
CT2_DEVICE: cpu
|
CT2_DEVICE: cpu
|
||||||
CT2_COMPUTE_TYPE: int8
|
CT2_COMPUTE_TYPE: int8
|
||||||
|
|
@ -313,6 +315,7 @@ services:
|
||||||
HF_HOME: /app/hf_cache
|
HF_HOME: /app/hf_cache
|
||||||
TZ: Europe/Madrid
|
TZ: Europe/Madrid
|
||||||
TRANSLATOR_ID: ${TRANSLATOR_ID:-}
|
TRANSLATOR_ID: ${TRANSLATOR_ID:-}
|
||||||
|
REDIS_URL: redis://:${REDIS_PASSWORD:-rss_redis_pass_2024}@redis:6379
|
||||||
PYTORCH_ENABLE_MPS_FALLBACK: 1
|
PYTORCH_ENABLE_MPS_FALLBACK: 1
|
||||||
volumes:
|
volumes:
|
||||||
- ./workers:/app/workers
|
- ./workers:/app/workers
|
||||||
|
|
@ -341,7 +344,7 @@ services:
|
||||||
DB_USER: ${DB_USER:-rss}
|
DB_USER: ${DB_USER:-rss}
|
||||||
DB_PASS: ${DB_PASS}
|
DB_PASS: ${DB_PASS}
|
||||||
TARGET_LANGS: es
|
TARGET_LANGS: es
|
||||||
TRANSLATOR_BATCH: 32
|
TRANSLATOR_BATCH: ${TRANSLATOR_BATCH:-128}
|
||||||
CT2_MODEL_PATH: /app/models/nllb-ct2
|
CT2_MODEL_PATH: /app/models/nllb-ct2
|
||||||
CT2_DEVICE: cpu
|
CT2_DEVICE: cpu
|
||||||
CT2_COMPUTE_TYPE: int8
|
CT2_COMPUTE_TYPE: int8
|
||||||
|
|
@ -350,6 +353,7 @@ services:
|
||||||
HF_HOME: /app/hf_cache
|
HF_HOME: /app/hf_cache
|
||||||
TZ: Europe/Madrid
|
TZ: Europe/Madrid
|
||||||
TRANSLATOR_ID: ${TRANSLATOR_ID:-}
|
TRANSLATOR_ID: ${TRANSLATOR_ID:-}
|
||||||
|
REDIS_URL: redis://:${REDIS_PASSWORD:-rss_redis_pass_2024}@redis:6379
|
||||||
PYTORCH_ENABLE_MPS_FALLBACK: 1
|
PYTORCH_ENABLE_MPS_FALLBACK: 1
|
||||||
volumes:
|
volumes:
|
||||||
- ./workers:/app/workers
|
- ./workers:/app/workers
|
||||||
|
|
|
||||||
|
|
@ -7,6 +7,7 @@ import (
|
||||||
"fmt"
|
"fmt"
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"net/url"
|
||||||
"os"
|
"os"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
@ -29,6 +30,7 @@ type Config struct {
|
||||||
MaxFailures int
|
MaxFailures int
|
||||||
PokeInterval time.Duration
|
PokeInterval time.Duration
|
||||||
FeedTimeout int
|
FeedTimeout int
|
||||||
|
HostDelay time.Duration
|
||||||
}
|
}
|
||||||
|
|
||||||
// Feed represents a row in the feeds table
|
// Feed represents a row in the feeds table
|
||||||
|
|
@ -59,8 +61,29 @@ type Noticia struct {
|
||||||
var (
|
var (
|
||||||
db *sql.DB
|
db *sql.DB
|
||||||
config Config
|
config Config
|
||||||
|
httpClient *http.Client
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// Per-host politeness limiter so parallel workers don't hammer one domain.
|
||||||
|
var (
|
||||||
|
rateMu sync.Mutex
|
||||||
|
lastRequest = make(map[string]time.Time)
|
||||||
|
)
|
||||||
|
|
||||||
|
func politeDelay(host string) {
|
||||||
|
if config.HostDelay <= 0 || host == "" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
rateMu.Lock()
|
||||||
|
now := time.Now()
|
||||||
|
next := lastRequest[host].Add(config.HostDelay)
|
||||||
|
lastRequest[host] = next
|
||||||
|
rateMu.Unlock()
|
||||||
|
if wait := next.Sub(now); wait > 0 {
|
||||||
|
time.Sleep(wait)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func loadConfig() {
|
func loadConfig() {
|
||||||
config = Config{
|
config = Config{
|
||||||
DBHost: getEnv("DB_HOST", "localhost"),
|
DBHost: getEnv("DB_HOST", "localhost"),
|
||||||
|
|
@ -72,6 +95,19 @@ func loadConfig() {
|
||||||
MaxFailures: getEnvInt("RSS_MAX_FAILURES", 10),
|
MaxFailures: getEnvInt("RSS_MAX_FAILURES", 10),
|
||||||
PokeInterval: time.Duration(getEnvInt("RSS_POKE_INTERVAL_MIN", 8)) * time.Minute,
|
PokeInterval: time.Duration(getEnvInt("RSS_POKE_INTERVAL_MIN", 8)) * time.Minute,
|
||||||
FeedTimeout: getEnvInt("RSS_FEED_TIMEOUT", 60),
|
FeedTimeout: getEnvInt("RSS_FEED_TIMEOUT", 60),
|
||||||
|
HostDelay: time.Duration(getEnvInt("RSS_HOST_DELAY_MS", 500)) * time.Millisecond,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func initHTTPClient() {
|
||||||
|
httpClient = &http.Client{
|
||||||
|
Timeout: time.Duration(config.FeedTimeout) * time.Second,
|
||||||
|
Transport: &http.Transport{
|
||||||
|
MaxIdleConns: config.MaxWorkers * 2,
|
||||||
|
MaxIdleConnsPerHost: 2,
|
||||||
|
IdleConnTimeout: 90 * time.Second,
|
||||||
|
TLSHandshakeTimeout: 10 * time.Second,
|
||||||
|
},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -200,10 +236,8 @@ func extractImage(item *gofeed.Item) string {
|
||||||
}
|
}
|
||||||
|
|
||||||
func processFeed(fp *gofeed.Parser, feed Feed, results chan<- int) {
|
func processFeed(fp *gofeed.Parser, feed Feed, results chan<- int) {
|
||||||
// Configure custom HTTP client with timeout and User-Agent
|
// Shared HTTP client with connection pooling across workers
|
||||||
client := &http.Client{
|
client := httpClient
|
||||||
Timeout: time.Duration(config.FeedTimeout) * time.Second,
|
|
||||||
}
|
|
||||||
|
|
||||||
// Create request to set User-Agent
|
// Create request to set User-Agent
|
||||||
req, err := http.NewRequest("GET", feed.URL, nil)
|
req, err := http.NewRequest("GET", feed.URL, nil)
|
||||||
|
|
@ -215,10 +249,21 @@ func processFeed(fp *gofeed.Parser, feed Feed, results chan<- int) {
|
||||||
}
|
}
|
||||||
req.Header.Set("User-Agent", "RSS2-Ingestor-Go/1.0")
|
req.Header.Set("User-Agent", "RSS2-Ingestor-Go/1.0")
|
||||||
|
|
||||||
// NOTE: We INTENTIONALLY SKIP ETag/Last-Modified headers based on user issues
|
// Reuse ETag/Last-Modified so servers can reply 304 Not Modified,
|
||||||
// If needed in future, uncomment:
|
// avoiding re-downloading unchanged feeds in every cycle.
|
||||||
// if feed.LastEtag.Valid { req.Header.Set("If-None-Match", feed.LastEtag.String) }
|
if feed.LastEtag.Valid && feed.LastEtag.String != "" {
|
||||||
// if feed.LastModified.Valid { req.Header.Set("If-Modified-Since", feed.LastModified.String) }
|
req.Header.Set("If-None-Match", feed.LastEtag.String)
|
||||||
|
}
|
||||||
|
if feed.LastModified.Valid && feed.LastModified.String != "" {
|
||||||
|
req.Header.Set("If-Modified-Since", feed.LastModified.String)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Per-host politeness: pace parallel fetches to the same domain
|
||||||
|
host := feed.URL
|
||||||
|
if u, perr := url.Parse(feed.URL); perr == nil && u.Host != "" {
|
||||||
|
host = u.Host
|
||||||
|
}
|
||||||
|
politeDelay(host)
|
||||||
|
|
||||||
resp, err := client.Do(req)
|
resp, err := client.Do(req)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|
@ -229,10 +274,29 @@ func processFeed(fp *gofeed.Parser, feed Feed, results chan<- int) {
|
||||||
}
|
}
|
||||||
defer resp.Body.Close()
|
defer resp.Body.Close()
|
||||||
|
|
||||||
|
// Honor Retry-After on rate limits without punishing the feed with failures
|
||||||
|
if resp.StatusCode == 429 || resp.StatusCode == 503 {
|
||||||
|
ra, raErr := strconv.Atoi(resp.Header.Get("Retry-After"))
|
||||||
|
if raErr == nil && ra > 0 && ra <= 3600 {
|
||||||
|
log.Printf("[Feed %d] Rate limited (%d), sleeping %ds (Retry-After)", feed.ID, resp.StatusCode, ra)
|
||||||
|
time.Sleep(time.Duration(ra) * time.Second)
|
||||||
|
}
|
||||||
|
results <- 0
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// On 304, keep the stored validators; update only if the server sends new ones.
|
||||||
if resp.StatusCode == 304 {
|
if resp.StatusCode == 304 {
|
||||||
log.Printf("[Feed %d] Not Modified (304)", feed.ID)
|
log.Printf("[Feed %d] Not Modified (304)", feed.ID)
|
||||||
// Update timestamp only? Or keep as is.
|
newEtag := feed.LastEtag.String
|
||||||
updateFeedStatus(feed.ID, feed.LastEtag.String, feed.LastModified.String, true, "")
|
newModified := feed.LastModified.String
|
||||||
|
if h := resp.Header.Get("ETag"); h != "" {
|
||||||
|
newEtag = h
|
||||||
|
}
|
||||||
|
if h := resp.Header.Get("Last-Modified"); h != "" {
|
||||||
|
newModified = h
|
||||||
|
}
|
||||||
|
updateFeedStatus(feed.ID, newEtag, newModified, true, "")
|
||||||
results <- 0
|
results <- 0
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
@ -433,6 +497,7 @@ func ingestCycle() {
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
loadConfig()
|
loadConfig()
|
||||||
|
initHTTPClient()
|
||||||
initDB()
|
initDB()
|
||||||
|
|
||||||
// Run immediately on start
|
// Run immediately on start
|
||||||
|
|
|
||||||
|
|
@ -1,8 +1,9 @@
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
import time
|
import time
|
||||||
import logging
|
import logging
|
||||||
import re
|
|
||||||
import fcntl
|
import fcntl
|
||||||
|
import hashlib
|
||||||
from typing import List, Optional
|
from typing import List, Optional
|
||||||
|
|
||||||
import psycopg2
|
import psycopg2
|
||||||
|
|
@ -20,6 +21,62 @@ LOG = logging.getLogger("translator_ct2")
|
||||||
TRANSLATOR_ID = os.environ.get("TRANSLATOR_ID", "")
|
TRANSLATOR_ID = os.environ.get("TRANSLATOR_ID", "")
|
||||||
TRANSLATOR_TOTAL = int(os.environ.get("TRANSLATOR_TOTAL", "1"))
|
TRANSLATOR_TOTAL = int(os.environ.get("TRANSLATOR_TOTAL", "1"))
|
||||||
|
|
||||||
|
CACHE_TTL = int(os.environ.get("TRANSLATION_CACHE_TTL", str(30 * 24 * 3600)))
|
||||||
|
_redis_client = None
|
||||||
|
|
||||||
|
|
||||||
|
def get_redis():
|
||||||
|
"""Return a Redis client or None. Cache is best-effort: never blocks translation."""
|
||||||
|
global _redis_client
|
||||||
|
if _redis_client is not None:
|
||||||
|
return _redis_client if _redis_client is not False else None
|
||||||
|
url = os.environ.get("REDIS_URL", "redis://localhost:6379")
|
||||||
|
try:
|
||||||
|
import redis
|
||||||
|
client = redis.Redis.from_url(
|
||||||
|
url, decode_responses=True,
|
||||||
|
socket_connect_timeout=2, socket_timeout=2,
|
||||||
|
)
|
||||||
|
client.ping()
|
||||||
|
_redis_client = client
|
||||||
|
LOG.info("Redis translation cache enabled")
|
||||||
|
return client
|
||||||
|
except Exception as e:
|
||||||
|
LOG.warning(f"Redis cache unavailable, translating without cache: {e}")
|
||||||
|
_redis_client = False
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def cache_key(lang_from: str, lang_to: str, text: str) -> str:
|
||||||
|
digest = hashlib.md5((text or "").encode("utf-8", "ignore")).hexdigest()
|
||||||
|
return f"tr:{lang_from}:{lang_to}:{digest}"
|
||||||
|
|
||||||
|
|
||||||
|
def lookup_cache(keys: List[str]) -> dict:
|
||||||
|
r = get_redis()
|
||||||
|
if not r or not keys:
|
||||||
|
return {}
|
||||||
|
try:
|
||||||
|
vals = r.mget(keys)
|
||||||
|
return {k: v for k, v in zip(keys, vals) if v}
|
||||||
|
except Exception as e:
|
||||||
|
LOG.warning(f"Redis mget error: {e}")
|
||||||
|
return {}
|
||||||
|
|
||||||
|
|
||||||
|
def store_cache(pairs, ttl: int = CACHE_TTL):
|
||||||
|
r = get_redis()
|
||||||
|
if not r or not pairs:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
pipe = r.pipeline(transaction=False)
|
||||||
|
for k, v in pairs:
|
||||||
|
if v:
|
||||||
|
pipe.setex(k, ttl, v)
|
||||||
|
pipe.execute()
|
||||||
|
except Exception as e:
|
||||||
|
LOG.warning(f"Redis cache store error: {e}")
|
||||||
|
|
||||||
|
|
||||||
def clean_text(text: str) -> str:
|
def clean_text(text: str) -> str:
|
||||||
if not text:
|
if not text:
|
||||||
|
|
@ -403,33 +460,64 @@ def process_batch(conn, rows):
|
||||||
LOG.info(f"Translating {lang_from} -> {lang_to} ({len(items)} items)")
|
LOG.info(f"Translating {lang_from} -> {lang_to} ({len(items)} items)")
|
||||||
|
|
||||||
try:
|
try:
|
||||||
|
# --- TITLES (served from Redis cache when possible) ---
|
||||||
titles = [i["titulo"] for i in items]
|
titles = [i["titulo"] for i in items]
|
||||||
translated_titles = translate_texts(lang_from, lang_to, titles)
|
title_keys = [cache_key(lang_from, lang_to, t) for t in titles]
|
||||||
|
title_cache = lookup_cache(title_keys)
|
||||||
|
if title_cache:
|
||||||
|
LOG.info(f"Title cache hits: {len(title_cache)}/{len(titles)}")
|
||||||
|
|
||||||
# Collect all body chunks across all items for a single batched call
|
translated_titles = [title_cache.get(k) for k in title_keys]
|
||||||
|
title_miss = [i for i, k in enumerate(title_keys) if k not in title_cache]
|
||||||
|
if title_miss:
|
||||||
|
miss_texts = [titles[i] for i in title_miss]
|
||||||
|
miss_translated = translate_texts(lang_from, lang_to, miss_texts)
|
||||||
|
store_cache([(title_keys[i], tr) for i, tr in zip(title_miss, miss_translated)])
|
||||||
|
for i, tr in zip(title_miss, miss_translated):
|
||||||
|
translated_titles[i] = tr
|
||||||
|
|
||||||
|
# --- BODY chunks (single batched call for cache misses) ---
|
||||||
flat_chunks = []
|
flat_chunks = []
|
||||||
flat_keys = []
|
flat_keys = []
|
||||||
|
flat_ck = []
|
||||||
for item in items:
|
for item in items:
|
||||||
body = (item["resumen"] or "").strip()
|
body = (item["resumen"] or "").strip()
|
||||||
if body:
|
if body:
|
||||||
chunks = split_body_into_chunks(body)
|
chunks = split_body_into_chunks(body)
|
||||||
flat_chunks.extend(chunks)
|
flat_chunks.extend(chunks)
|
||||||
flat_keys.extend([(item["tr_id"], i) for i in range(len(chunks))])
|
flat_keys.extend([(item["tr_id"], i) for i in range(len(chunks))])
|
||||||
|
flat_ck.extend([cache_key(lang_from, lang_to, c) for c in chunks])
|
||||||
|
|
||||||
translated_bodies = []
|
chunk_cache = lookup_cache(flat_ck)
|
||||||
if flat_chunks:
|
if chunk_cache:
|
||||||
try:
|
LOG.info(f"Body cache hits: {len(chunk_cache)}/{len(flat_ck)}")
|
||||||
translated_bodies = translate_texts(lang_from, lang_to, flat_chunks)
|
|
||||||
except Exception as e:
|
|
||||||
LOG.error(f"Batch body translation error: {e}")
|
|
||||||
translated_bodies = flat_chunks
|
|
||||||
|
|
||||||
body_parts = defaultdict(list)
|
body_parts = defaultdict(list)
|
||||||
for (tr_id, _), tr in zip(flat_keys, translated_bodies):
|
miss_idx = []
|
||||||
if tr is None:
|
miss_texts = []
|
||||||
continue
|
miss_keys = []
|
||||||
body_parts[tr_id].append(tr)
|
for idx, ck in enumerate(flat_ck):
|
||||||
|
tr_id, _ = flat_keys[idx]
|
||||||
|
if ck in chunk_cache:
|
||||||
|
body_parts[tr_id].append(chunk_cache[ck])
|
||||||
|
else:
|
||||||
|
miss_idx.append(idx)
|
||||||
|
miss_texts.append(flat_chunks[idx])
|
||||||
|
miss_keys.append(ck)
|
||||||
|
|
||||||
|
if miss_texts:
|
||||||
|
try:
|
||||||
|
miss_translated = translate_texts(lang_from, lang_to, miss_texts)
|
||||||
|
store_cache(list(zip(miss_keys, miss_translated)))
|
||||||
|
except Exception as e:
|
||||||
|
LOG.error(f"Batch body translation error: {e}")
|
||||||
|
miss_translated = miss_texts
|
||||||
|
for j, tr in enumerate(miss_translated):
|
||||||
|
if tr:
|
||||||
|
body_parts[flat_keys[miss_idx[j]][0]].append(tr)
|
||||||
|
|
||||||
|
# --- BATCH COMMIT: single transaction per group ---
|
||||||
|
updates = []
|
||||||
for idx, item in enumerate(items):
|
for idx, item in enumerate(items):
|
||||||
tt = clean_text((translated_titles[idx] or "").strip())
|
tt = clean_text((translated_titles[idx] or "").strip())
|
||||||
parts = body_parts.get(item["tr_id"])
|
parts = body_parts.get(item["tr_id"])
|
||||||
|
|
@ -440,21 +528,23 @@ def process_batch(conn, rows):
|
||||||
if not tb:
|
if not tb:
|
||||||
tb = item["resumen"]
|
tb = item["resumen"]
|
||||||
|
|
||||||
# 2. INDIVIDUAL COMMIT: Save each item as it's done
|
updates.append((tt, tb, item["tr_id"]))
|
||||||
|
|
||||||
|
if updates:
|
||||||
try:
|
try:
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
cursor.execute(
|
cursor.executemany(
|
||||||
"""
|
"""
|
||||||
UPDATE traducciones
|
UPDATE traducciones
|
||||||
SET titulo_trad = %s, resumen_trad = %s, status = 'done', locked_at = NULL
|
SET titulo_trad = %s, resumen_trad = %s, status = 'done', locked_at = NULL
|
||||||
WHERE id = %s
|
WHERE id = %s
|
||||||
""",
|
""",
|
||||||
(tt, tb, item["tr_id"]),
|
updates,
|
||||||
)
|
)
|
||||||
conn.commit()
|
conn.commit()
|
||||||
cursor.close()
|
cursor.close()
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
LOG.error(f"Update error for ID {item['tr_id']}: {e}")
|
LOG.error(f"Batch update error (items stay pending for retry): {e}")
|
||||||
conn.rollback()
|
conn.rollback()
|
||||||
|
|
||||||
LOG.info(f"Finished group {lang_from} -> {lang_to}")
|
LOG.info(f"Finished group {lang_from} -> {lang_to}")
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue