diff --git a/.env.example b/.env.example index da291467..dbde23b1 100644 --- a/.env.example +++ b/.env.example @@ -55,7 +55,7 @@ POSTGRES_DB=vagas # Use esta URL quando backend/scraper rodarem fora do Docker e o banco/Valkey # estiverem expostos no host (veja LOCAL_DEVELOPMENT.md para expor as portas). -DATABASE_URL=postgresql://vagas:vagas@localhost:5432/vagas +DATABASE_URL=postgresql://vagas:vagas@localhost:5432/vagas?sslmode=disable VALKEY_URL=redis://localhost:6379/0 # Dentro do Docker Compose, o backend/scraper-go já recebem @@ -156,3 +156,8 @@ GREENHOUSE_COMPANIES_FILE=./internal/interfaces/greenhouseCompanies.json LEVER_ENABLED=false LEVER_COMPANIES_FILE=./internal/interfaces/leverCompanies.json LEVER_INCLUDE_ALL_JOBS=true + +# Catálogo durável do Processor (DATABASE_URL já existente): nove dias renováveis. +SCRAPER_CATALOG_LIFETIME=216h +# Cache de páginas de busca, limitado também pela próxima expiração de vaga. +JOB_SEARCH_CACHE_TTL_SECONDS=120 diff --git a/BACKEND.md b/BACKEND.md index 597023b3..2bf18e62 100644 --- a/BACKEND.md +++ b/BACKEND.md @@ -566,3 +566,71 @@ Para impedir perdas com índices estruturados incompletos, buscas filtradas usam Custo temporário: O(N) documentos candidatos por busca filtrada para obter total exato; IDs permanecem em memória. Ordenação por match conserva no máximo `offset + limit` IDs/scores e hidrata novamente a página escolhida. Valkey não oferece snapshot entre essas leituras: expiração/reclassificação concorrente pode mudar documentos durante uma busca. A busca simples legada ainda estima total descontando apenas órfãos observados na hidratação da página; esse comportamento preexistente foi preservado. Ficam para a próxima PAV: índices separados de famílias principais/relacionadas, reconstrução de índices, cache keys finais, invalidação por reclassificação e evolução de match score. Não houve mudança no Processor Go, autenticação, autorização ou rate limit (a busca não tinha limitador próprio; os limitadores de autenticação permanecem). + +## Busca e cache após PAV-125 + +A PAV-125 evolui a PAV-124 sem mudar o parser público de `family`/`familyMode`, +os IDs da PAV-123, autenticação, rate limit ou envelope HTTP. PostgreSQL mantém o +catálogo do Processor; a busca online continua usando Valkey. Consulte +[SCRAPER.md](SCRAPER.md#catálogo-durável-e-índices-de-famílias-pav-125) para migration, +backfill, rebuild, reconciliação e rollback. + +Quando existe `scraper:jobs:index-version`, o repository usa o namespace ativo. +Múltiplas famílias usam SUNIONSTORE; grupos de keywords são intersectados e a +atividade é conferida pelo ZSET de expiração. Temporárias têm TTL de 120 segundos +com renovação durante leitura e são removidas ao finalizar. IDs ficam no Valkey; +o backend hidrata lotes de até 200 documentos. Os demais predicados da PAV-124 +continuam sendo verificados antes de total/paginação, porque índices legados de +senioridade, modalidade e localização não representam exatamente todas as +inferências e aliases do contrato HTTP. Não há consulta SQL por request nem +pós-filtro de famílias depois da página. Ordenação explícita por match usa ZSET +temporário com scores, sem acumular documentos de todo o catálogo no Node. +Empates usam ID determinístico; a ordenação anterior de SETs não tinha garantia. +A busca simples conserva priorização por keywords do perfil e ordenação da página. + +Antes da primeira publicação de namespace, o caminho legado da PAV-124 permanece +como fallback de migração. Depois da publicação, falhas de índice não causam +fallback silencioso para dados antigos. A leitura de vagas por ID também acompanha +a versão ativa. Saved jobs continuam independentes e seus dados persistidos não +são modificados. + +O cache de páginas usa `jobs:search:v2:`. O fingerprint inclui filtros +normalizados, famílias ordenadas e deduplicadas, `familyMode`, página, limite, +ordenação, versões de contrato/taxonomia/índices, geração e contexto normalizado +de ranking. Texto, localização e perfil não aparecem diretamente nas chaves. +`JOB_SEARCH_CACHE_TTL_SECONDS=120` é configurável; o TTL é limitado pela próxima +expiração global de vaga. Publicação do cache verifica a geração atomicamente. +Uma geração nova torna entradas antigas inacessíveis; elas expiram naturalmente. +Há single-flight local por processo, sem reutilizar o lock do scraper. Falha de +cache permite executar a consulta normal. Não há lock distribuído de stampede. + +Create/update/reclassificação/desativação e expiração concluídos no Processor +versionam o cache após indexação; rebuild/rollback publicam namespace e geração +juntos. Alterações de regras devem incrementar `searchContractVersion`; mudanças +de taxonomia também integram o fingerprint. A limpeza administrativa +`DELETE /admin/jobs/cache` mantém autenticação e formato de resposta, mas, +com catálogo versionado, invalida cache por geração e retorna `deleted: 0` em vez +de apagar documentos/índices ativos. É uma mudança operacional deliberada para +proteger PostgreSQL como fonte de verdade; desativação durável pertence ao Processor. + +Produto e Design usam evidências positivas disponíveis: família nas preferências, +senioridade, modalidade/localização, competências, ferramentas e experiências. +Ausência de linguagem não reduz o score. Sem evidências não se inventa score. +`matchReasons` contém somente razões públicas, sem pesos ou dados privados. +As demais famílias conservam exatamente a fórmula anterior por tecnologias. +Preferências e notificações continuam exclusivamente no backend. + +No contrato atual de `UserPreferences`, `jobTypes` representa modalidades +(`Remoto`, `Híbrido`, `Presencial`), e `remoteOnly` indica preferência por remoto. +Nenhum deles fornece contrato de contratação. `searchLocation` é a localização +de busca; `keywords` são termos livres usados como evidência textual, sem inferir +uma família canônica. O modelo atual não oferece uma seleção autoritativa de +família nem contrato; esses campos de match permanecem sem preferência derivada. + +Limitações: a busca filtrada com cache frio ainda percorre todos os candidatos +necessários para obter total exato; consultas residuais amplas têm custo O(N). +Valkey não oferece snapshot de documentos entre todos os lotes: mudanças concorrentes +podem afetar uma resposta, embora uma geração modificada impeça publicar cache +obsoleto. O índice usa expiração em segundos; existe granularidade inferior a um +segundo em relação aos timestamps SQL. A meta operacional de p95 < 500 ms exige +medição com volume e concorrência representativos; testes locais não a comprovam. diff --git a/SCRAPER.md b/SCRAPER.md index 230d685e..fcb61d9b 100644 --- a/SCRAPER.md +++ b/SCRAPER.md @@ -461,3 +461,124 @@ docker exec vagas-valkey valkey-cli HGETALL scraper:run:state ``` Uma segunda execução manual deve retornar conflito sem iniciar adapters. Após o término, as duas chaves devem desaparecer. Nunca remova a chave manualmente apenas porque ela existe: primeiro confirme que não há processo correspondente ativo e registre valor e TTL. + +## Catálogo durável e índices de famílias (PAV-125) + +O catálogo de vagas passa a ter PostgreSQL como fonte de verdade. A migration +aditiva `backend/drizzle/0015_job_catalog.sql` segue o histórico Drizzle existente. +Não havia tabela de catálogo reutilizável: `saved_jobs` representa vagas salvas +por usuários; o script de seed anterior escreve documentos de demonstração no +Valkey. Nenhuma dessas estruturas substitui o catálogo do Processor. O seed +legado de vagas só continua disponível antes da publicação do catálogo; depois +disso ele é ignorado para não alterar índices fora do fluxo durável. Para migrar +dados de demonstração existentes, use o backfill explícito. + +`job_catalog` conserva o documento `domain.Job` em `payload` JSONB, o ID estável +já calculado por `jobstore.StableID`, `first_seen_at`, `last_seen_at`, `updated_at`, +`expires_at`, revisão monotônica e revisão de indexação confirmada. Não existe +FK para usuários ou saved jobs. A listagem administrativa conserva o endpoint +completo e o parâmetro `limit` por streaming de um snapshot SQL read-only, com +count e linhas consistentes, sem carregar todo o catálogo no Processor. O upsert em transação faz merge de fontes e +keywords como antes, preserva `first_seen_at` e renova `last_seen_at` e expiração. +Somente o resultado de um commit confirmado é enviado ao indexador. + +A atividade é inferida por `expires_at > agora`; registros antigos permanecem +no PostgreSQL. `SCRAPER_CATALOG_LIFETIME=216h` define a janela padrão de nove dias +(e pode ser configurado para `192h`, oito dias). Reclassificação não renova essa +janela. Documentos Valkey recebem TTL correspondente à expiração persistida. +A busca também intersecta candidatos com o ZSET de expiração; não depende apenas +do desaparecimento das chaves. Antes de cada execução do scheduler, a manutenção +remove associações de registros já inativos. Portanto, SETs podem conservar IDs +expirados até a próxima manutenção, mas a busca já os exclui. Há comando explícito +para antecipar essa remoção. Não há deleção física automática de histórico. + +O Processor continua responsável por classificação, persistência, indexação e +manutenção. O backend Node continua responsável por HTTP, validação, autenticação, +autorização, filtros e cache de busca. `DATABASE_URL` agora é obrigatório também +para o Processor; o startup verifica conectividade, existência da migration e +impede bootstrap parcial quando já existem vagas e ainda não foi publicado um +namespace. Nesse caso exige backfill/rebuild explícito antes de iniciar. +O overlay `docker-compose.migrate.yml` fornece a conexão interna ao PostgreSQL +e faz o Processor aguardar a migration, como já fazia o backend. Em execução local +a URL de exemplo usa `sslmode=disable`; ambientes externos devem preservar a +configuração TLS apropriada da sua conexão. + +### Índices e atualização + +Cada namespace `scraper:jobs:ns::` contém SETs de IDs: + +- `family:`: associação principal **ou** relacionada; +- `family:primary:`: somente principal; +- `family:related:`: somente relacionada, sem repetir a principal; +- `index`: catálogo indexado; `expires`: ZSET de expiração; +- `index-membership:`: JSON com revisão, taxonomia, principal, relacionadas + e chaves de associação anteriores; +- `job:`: documento único; `keys`: manifesto de limpeza controlada. + +Os índices fixos `scraper:jobs:family:`, `family:primary:` e +`family:related:` são preservados como projeções de compatibilidade. +A taxonomia vem de `internal/taxonomy/families.json`; `other` não cria índice de +família, e valores desconhecidos interrompem a indexação controlada. + +Lua valida todos os tipos e registros inversos antes de escrever. Remove apenas +o ID daquela vaga dos seus antigos SETs, adiciona novas associações e atualiza +membership, documento e geração de cache atomicamente por lote. Revisões impedem +retries e atualizações atrasadas de desfazer classificações novas. Não há uma +transação distribuída entre PostgreSQL e Valkey: um commit pode ficar pendente de +indexação se Valkey falhar. A execução retorna erro, o registro durável continua +recuperável, e a reconciliação com `--fix` ou o rebuild podem corrigir a projeção. +Retries são limitados; cancelamento interrompe espera e operações. + +### Implantação e operações explícitas + +Execute a migration pelo mecanismo já existente (`npm run db:migrate` no +backend). Pause a versão antiga do coletor antes do backfill: ela não participa +do fencing PostgreSQL introduzido nesta task. Execute o backfill e valide a +reconciliação antes de iniciar o novo Processor. Essas ações **não** são executadas +automaticamente na inicialização normal. + +O Dockerfile do Processor também fornece `/catalog`. Exemplos, com as mesmas +variáveis de conexão da aplicação: + +```sh +docker compose run --rm --entrypoint /catalog scraper-go --operation backfill --batch-size 200 +docker compose run --rm --entrypoint /catalog scraper-go --operation reconcile +docker compose run --rm --entrypoint /catalog scraper-go --operation reconcile --fix +docker compose run --rm --entrypoint /catalog scraper-go --operation rebuild --batch-size 200 +docker compose run --rm --entrypoint /catalog scraper-go --operation reclassify +docker compose run --rm --entrypoint /catalog scraper-go --operation expire +docker compose run --rm --entrypoint /catalog scraper-go --operation deactivate --ids ID1,ID2 +docker compose run --rm --entrypoint /catalog scraper-go --operation rollback --version VERSAO_ANTERIOR +docker compose run --rm --entrypoint /catalog scraper-go --operation cleanup --version VERSAO_INATIVA +``` + +Também é possível executar `go run ./cmd/catalog` dentro de `scraper-go`, com +`DATABASE_URL` e `VALKEY_URL` exportados no ambiente. + +O backfill faz SCAN e GET/PTTL em lotes, valida documentos/famílias, preserva os +IDs e a vida restante observada e faz `ON CONFLICT DO NOTHING`. Não inventa a data +original da coleta: `first_seen_at` representa a observação na migração. Documentos +inválidos, sem identidade ou sem TTL confiável são contabilizados e ignorados; +próximas coletas podem completar o catálogo. Depois dos commits, um novo namespace +é reconstruído a partir do PostgreSQL. + +Rebuild usa cursor de IDs e somente registros ativos do PostgreSQL. Constrói um +namespace isolado, compara documentos, membership, famílias, IDs e contagens, e +só então troca o ponteiro ativo e a geração de cache. Um advisory lock exclusivo +impede concorrência com commit/indexação do Processor. A busca continua lendo a +versão ativa anterior durante o rebuild. Uma manutenção pode esperar a coleta em +andamento terminar; a espera é cancelável. Namespaces incompletos têm TTL; a versão +anterior é mantida até cleanup explicitamente solicitado. + +Reconcile é read-only por padrão; `--fix` reconstrói uma versão validada. Detecta +vagas ausentes, IDs inexistentes/inativos, classificação, membership, documentos, +associações, taxonomia e contagens divergentes. Os números contam ocorrências de +divergência, não necessariamente IDs únicos. + +Rollback de namespace exige que a versão escolhida ainda corresponda ao catálogo +PostgreSQL; caso contrário, é recusado e deve-se fazer rebuild. O catálogo durável +não é apagado durante rollback de aplicação. Drizzle neste projeto usa migrations +forward-only: não existe down migration destrutiva automática. Não remover a tabela +para reverter código; preservar os dados e planejar eventual arquivamento separado. +Nenhuma operação usa FLUSH ou o comando KEYS; cleanup só alcança o manifesto do +namespace selecionado e recusa a versão ativa e chaves externas. diff --git a/backend/PAV-125_REPORT.md b/backend/PAV-125_REPORT.md new file mode 100644 index 00000000..55ee246b --- /dev/null +++ b/backend/PAV-125_REPORT.md @@ -0,0 +1,220 @@ +# PAV-125 — relatório de implementação + +Data: 2026-10-07. Implementação sobre a branch derivada da PAV-124, sem troca de branch, merge, rebase, commit, push ou PR. + +## Resultado e auditoria da base + +O catálogo processado passa a ter PostgreSQL como fonte de verdade. O Processor classifica, faz upsert em lote e somente depois do commit publica documentos e índices no Valkey. A busca pública continua usando Valkey, com união de famílias, filtros antes da paginação, total consistente, cache determinístico e score específico para Produto/Design. + +A busca conclusiva examinou schemas, migrations Drizzle, repositories/adapters, scripts, backend, scraper-go e referências Git locais disponíveis (59 referências). Não foi encontrado catálogo PostgreSQL compatível. `saved_jobs` representa vagas salvas por usuários; não é catálogo de ingestão e não foi convertido nem alterado. O seed existente produzia documentos no Valkey. Foram reutilizados o mecanismo Drizzle, o modelo `domain.Job`, o algoritmo `StableID`, o merge de duplicados, os índices de keywords e a taxonomia da PAV-123. + +Foram considerados o agente `.codex_local/agents/agent-dev-jobs-processor.md`, Engineering Spec e Engineering Workflow. `.codex_local/agent.base.md` não estava disponível; nenhuma regra foi inventada para substituí-lo. + +## Ownership e arquitetura + +| Componente | Responsabilidade | +| --- | --- | +| Backend/Node | HTTP, parser e validação da PAV-124, composição da consulta, paginação/total, cache de busca, perfil e match, autenticação/autorização/rate limit | +| Jobs Processor/Go | Ingestão/classificação, persistência transacional do catálogo, publicação após commit, reclassificação, ciclo ativo, backfill, rebuild e reconciliação | +| PostgreSQL | Estado durável, identidade, payload confirmado, ciclo temporal e revisões | +| Valkey | Projeção de documentos, SETs de IDs, registro inverso, expiração da projeção, namespaces e cache | + +O port durável é consumido por `jobstore`; o adapter SQL é composto no servidor. Produção exige `DATABASE_URL` e a migration aplicada. A implementação legada de `jobstore.New` permanece para compatibilidade dos testes e da transição; o servidor injeta exclusivamente `NewDurable` nas coletas manual e agendada. + +## Schema, identidade e ciclo + +Migration aditiva: `backend/drizzle/0015_job_catalog.sql`, com journal e snapshot Drizzle correspondentes. Cria apenas `job_catalog`, a sequência de revisões e índices SQL. Não modifica usuários ou saved jobs. + +| Campo | Finalidade | +| --- | --- | +| `id` text PK | ID estável já produzido pelo Processor | +| `payload` jsonb | Representação atual de Job: origem, conteúdo, filtros, classificação e insumos de match; sem uma segunda representação paralela | +| `first_seen_at` | Primeira persistência observada | +| `last_seen_at` | Última coleta persistida | +| `updated_at` | Última atualização persistida, inclusive reclassificação | +| `expires_at` | Participação no catálogo ativo | +| `revision` | Revisão monotônica da persistência, usada para impedir indexação antiga | +| `indexed_revision` | Checkpoint da publicação confirmada no Valkey | + +Há constraints para payload objeto, identidade do payload e ciclo temporal. A PK atende o cursor por ID; o índice `(expires_at, id)` atende atividade; o índice parcial de revisões pendentes permite observar trabalho ainda não publicado. Não foram criados índices SQL especulativos para cada filtro HTTP. + +Uma vaga está ativa se `expires_at > now`. `SCRAPER_CATALOG_LIFETIME` é uma duração configurável, padrão `216h` (9 dias). Recoleta faz upsert da mesma identidade, preserva primeira observação e renova última observação/expiração. Reclassificação não renova esse ciclo. Expiração e desativação explícita conservam a linha histórica; não há deleção física nem política nova de retenção histórica. + +SaveBatch deduplica e ordena IDs, bloqueia identidades em lote, lê registros anteriores em lote e executa um upsert transacional do lote. Contexto e rollback são preservados. `Persisted` contém somente linhas confirmadas pelo commit. Falha de commit não produz índice. O ID original e as regras atuais de merge foram preservados; a mudança de título/empresa/localização continua sujeita ao algoritmo de identidade existente. + +## Índices, registro inverso e atomicidade + +São SETs de IDs estáveis, sem repetição do objeto completo: + +- `scraper:jobs:family:`: associação principal OU relacionada; +- `scraper:jobs:family:primary:`: exclusivamente principal; +- `scraper:jobs:family:related:`: exclusivamente relacionada, excluindo a principal. + +As consultas usam equivalentes dentro do namespace ativo `scraper:jobs:ns::`. As 39 chaves de famílias de compatibilidade são mantidas na publicação. `fullstack`, `devops` e `platform` permanecem independentes. As 13 famílias vêm da taxonomia canônica; não foi criada lista manual adicional. `other` pode permanecer no diagnóstico persistido, mas não possui índice público. Família desconhecida provoca erro controlado, sem criação de chave arbitrária de família. + +O registro `index-membership:` contém revisão, versão da taxonomia, primary/related, expiração e associações anteriores. A atualização remove somente o ID da própria vaga nos índices anteriores e adiciona suas associações novas, sem varrer todas as famílias. + +Foi escolhido Lua em lotes limitados: o projeto já usa Valkey independente e a operação precisa publicar documentos, membership, conjuntos e geração juntos. O script faz preflight de tipos, metadados e geração antes de escrever, pois erro de execução Lua não reverte comandos anteriores. Revisões impedem atualização atrasada; repetir a mesma revisão não altera a geração. Testes exercitam erro estrutural em um lote sem mutação parcial. Não foi introduzida infraestrutura de transações distribuídas. + +PostgreSQL e Valkey não têm atomicidade conjunta: há uma janela entre commit e publicação. Falha nessa janela é retornada, conserva o dado durável e o checkpoint pendente, permite retry limitado e reparo explícito. A operação não declara sucesso após falha de indexação. A geração de busca é atualizada no mesmo script que conclui a publicação dos índices, nunca antes do commit. Erro posterior no checkpoint SQL também é retornado; os índices podem já estar publicados e o retry é seguro. + +## Rebuild, backfill e reconciliação + +O comando `/catalog` foi incluído na imagem Go. Também pode ser executado com `go run ./cmd/catalog` no módulo. Exemplos completos de configuração e execução estão em [SCRAPER.md](../SCRAPER.md). + +| Operação | Comportamento | +| --- | --- | +| `rebuild` | Lê apenas vagas SQL ativas, por cursor em batches; cria namespace novo; valida PostgreSQL × Valkey; publica ponteiro/compatibilidade/geração atomicamente; registra checkpoints após publicação | +| `reconcile` | Padrão read-only; relatório de ausência, sobras, membership, primary, related, any, documentos, valores inválidos e contagens | +| `reconcile --fix` | Correção explicitamente solicitada, por rebuild validado | +| `backfill` | SCAN dos documentos legados, leitura/PTTL em batches, validação, import SQL após commit e rebuild do SQL | +| `reclassify` | Reclassifica documentos persistidos, commits em lote, remove associações antigas e publica novas; repetição repara publicação pendente sem renovar expiração | +| `expire` | Remove da projeção ativa vagas SQL expiradas; é idempotente e também roda antes da coleta agendada | +| `deactivate --ids ...` | Desativação SQL explícita seguida de publicação após commit, conservando histórico | +| `rollback` | Só publica namespace anterior se ele ainda corresponde ao estado SQL atual | +| `cleanup --version ...` | Remove somente um namespace inativo, usando seu manifesto e batches; recusa o ativo | + +O fencing PostgreSQL usa lease compartilhado para processamento/expiração e exclusivo para manutenção. Assim, um rebuild não perde commits publicados durante sua construção. Cancelamento libera conexões/locks; não reutiliza o lock de execução do scraper para cache de busca. Relatórios contam ocorrências de divergência; vários problemas podem pertencer ao mesmo ID. + +O rebuild nunca apaga o conjunto ativo antes da validação. Namespace anterior permanece para rollback controlado; drafts possuem TTL. A limpeza é explícita e limitada ao namespace pertencente ao catálogo. Não foram introduzidos FLUSHALL, FLUSHDB ou o comando KEYS para limpeza. Sessões, filas, locks e estruturas alheias não são removidos. + +O backfill usa somente dados confiáveis: documento existente, ID da chave e TTL restante. Não inventa data histórica de coleta: os timestamps SQL de observação representam a importação; a expiração conserva a vida restante observada. Documento inválido, identidade divergente, taxonomia desconhecida, TTL ausente ou já expirado é contabilizado e ignorado. `ON CONFLICT DO NOTHING` evita duplicação, renovação artificial ou sobrescrita de dados de coletas novas. Próximas coletas completam/renovam os registros pelo fluxo normal. Backfill não roda no startup. + +## Busca, paginação e compatibilidade da PAV-124 + +O adapter faz SUNION entre famílias e SINTER com o grupo de keywords quando presente. O conjunto ativo é intersectado com o ZSET de expiração. IDs candidatos e ranking permanecem em estruturas temporárias no Valkey, com TTL e remoção ao terminar. Hidratação usa batches de até 200 documentos, sem query SQL por vaga. + +Os demais filtros continuam passando pelos predicados da PAV-124 antes de total/paginação. Os índices históricos desses filtros usam inferências/aliases diferentes e não foram usados para uma interseção que descartaria resultados válidos. Não há pós-filtro restrito à página. Total conta exatamente os matches dos mesmos predicados que selecionam a página. Ranking global guarda IDs/scores no Valkey e hidrata somente a página final; desempate usa ID determinístico. A ordenação padrão conserva a prioridade por keywords do perfil e a ordenação de match da página prevista no fluxo antigo. + +CSV, repetição de `family`, combinação dos formatos, normalização, limites, erros estáveis e `familyMode=any` padrão permanecem na PAV-124. `primary` usa exclusivamente o índice principal; `any` inclui relacionadas. Envelope HTTP, paginação, auth, rate limit, opções de filtros, IDs e saved jobs permanecem. O diagnóstico `source` pode indicar a estratégia de batches verificados. + +Não foram inventados novos query params para provider/text/modality: o fingerprint cobre todos os filtros que o contrato atual realmente interpreta, incluindo keywords, technology, company, type/model, level/seniority, localização/continente/país/estado/cidade, contrato, matchSort e paginação. Evoluir esse contrato exige evolução correspondente do fingerprint. + +A listagem administrativa Go conserva a opção de retornar todo o catálogo, com streaming de um snapshot SQL read-only e count no mesmo snapshot; não há um limite silencioso de mil registros nem carregamento de todo o catálogo em memória. Falha após iniciar essa resposta aborta o stream em vez de devolver JSON incompleto como sucesso. + +## Cache e invalidação + +Chaves `jobs:search:v2:` usam os filtros tipados normalizados, famílias deduplicadas/ordenadas, modo padrão, paginação, ordenação, versão do contrato, versão/hash da taxonomia, namespace/geração e contexto normalizado de ranking do perfil. CSV/repetição/ordem de famílias/default any são equivalentes; primary e any são distintos. Texto, localização privada e contexto de perfil entram apenas no hash, sem PII/token/e-mail ou texto completo nas chaves. + +`JOB_SEARCH_CACHE_TTL_SECONDS` é configurável, padrão 120 segundos; o TTL real é limitado pela próxima expiração ativa. O cache valida geração na leitura e faz compare-and-set Lua na escrita. Criação, atualização, reclassificação, desativação, expiração e rebuild mudam a geração depois da publicação. Mudanças de taxonomia/contrato alteram o fingerprint. Resultados antigos expiram sem busca global de chaves. Há single-flight local para solicitações simultâneas equivalentes; não foi criado lock distribuído de stampede. Falha exclusiva de cache não oculta o resultado válido da consulta. + +Mudança operacional registrada: depois da ativação do catálogo, `DELETE /admin/jobs/cache` invalida a geração e preserva a projeção; mantém o envelope com `deleted: 0`. Não pode mais apagar o catálogo durável por limpeza de cache. O caminho legado ainda existe antes da ativação, com guarda atômica contra uma ativação concorrente. O seed Node também deixa de escrever dados de demonstração no catálogo já ativo. + +## Match score + +Somente primary `product` e `product_design` usam o novo modelo de evidência positiva. Consideram família, senioridade, modalidade, localização, contrato e experiências/competências/ferramentas realmente disponíveis no perfil. Produto considera, por exemplo, discovery/roadmap/backlog/analytics; Design considera UX/UI, pesquisa, prototipação, acessibilidade e Figma. A ausência de linguagem/framework não entra no denominador nem provoca penalidade. HTML/CSS em Design têm contribuição auxiliar limitada. + +Sem evidência de perfil não se inventa um score. Scores são inteiros, determinísticos; competências são normalizadas/deduplicadas. `matchReasons` é opcional, com razões públicas sem pesos, descrição completa ou dados privados. As outras 11 famílias continuam delegando à fórmula anterior, testada sem alteração. Preferências são carregadas uma vez por busca, sem N+1 por vaga; auth e regras de usuário permanecem no Node. + +Revisão de semântica das preferências: o schema HTTP e o frontend definem `jobTypes` como `Remoto`, `Híbrido`, `Presencial`, portanto o mapeamento para `modalities` é correto. A leitura de valores persistidos também valida esse enum e ignora listas legadas incompatíveis; não converte CLT/PJ/full-time/part-time/contract. `remoteOnly` continua separado de contrato, e `searchLocation` representa localização de busca. Foi removida a inferência de família por `keywords`: são strings livres, sem garantia de IDs canônicos. Não existe campo contratual de preferência de contratação ou família no modelo atual, portanto `contract` e `family` não são derivados dele. Schemas HTTP, persistência de preferências e frontend não foram alterados. Sete casos adicionais cobrem os dois scores, modalidade versus contrato, dados legados e keywords sem autoridade de família. Após esta revisão, os 123 testes selecionados de perfil/match/busca/fingerprint/contrato HTTP passaram (5 arquivos); typecheck e `git diff --check` também passaram. A execução HTTP foi repetida fora do sandbox porque ele bloqueava portas locais com EPERM. + +## Validação executada + +Serviços descartáveis locais: PostgreSQL 16 em porta 55432 e Redis 7.0.15 em porta 56379. A migration foi aplicada somente a schemas isolados de teste. Nenhuma migration/backfill/operação destrutiva foi executada na aplicação real. Redis exercita os comandos compatíveis utilizados; uma validação com Valkey 8 de produção continua recomendada na implantação. + +| Check | Resultado | +| --- | --- | +| Backend `npm test -- --run`, com `PAV125_TEST_VALKEY_URL` | **76 arquivos, 826 testes aprovados**, incluindo unitários, contratos HTTP, integração Redis e compatibilidade da PAV-124 | +| Go `go test ./...`, com `PAV125_TEST_DATABASE_URL` e `PAV125_TEST_VALKEY_URL` | **Aprovado**, incluindo insert/upsert/ciclo, rollback/falha no commit sem indexação, batches, backfill, rebuild, reconciliação, reclassificação e streaming | +| Go `go test -race ./...`, mesmas dependências isoladas | **Aprovado**, sem races detectadas | +| Go `go vet ./...` | **Aprovado** | +| Backend `backend/node_modules/.bin/tsc --noEmit -p backend/tsconfig.json` | **Aprovado**, compilador do próprio módulo | +| Swagger/OpenAPI com `SwaggerParser.validate` | **Aprovado**, incluído na suíte backend | +| Compose com arquivos base/infra/migrate e `.env.example`, `config --quiet` | **Aprovado** | +| Build CLI Go com `CGO_ENABLED=0` | **Aprovado** | +| `git diff --check` | **Aprovado** | +| Lint | Backend não tem script/configuração de lint aplicável; ESLint da raiz ignora backend. Não há target/configuração adicional de lint Go definida; vet foi executado | + +Os testes de integração externos são opt-in por essas variáveis, para não tocar bancos arbitrários; sem elas são ignorados. Nesta validação foram habilitados. Os casos de índice também usam testes locais controlados para preflight, falha parcial, deduplicação, desconhecidos, other, revisão atrasada, idempotência, primary/related/any e cancelamento. O teste de busca com mais de mil candidatos verifica hidratação limitada a 200 e total/página/ranking corretos. + +Logs completos da execução local: `/tmp/pav125-backend-all.log`, `/tmp/pav125-go-all.log`, `/tmp/pav125-go-race.log` e `/tmp/pav125-go-vet.log`. Esses arquivos temporários não integram o repositório. + +## Desempenho, riscos e limites + +- Não foi medido p95 representativo de `/jobs/search`; **a meta de 500 ms não foi comprovada**. O teste de mil candidatos verifica limites funcionais e de hidratação, não representa benchmark de produção. +- Uma busca fria com filtros residuais/ranking ainda percorre candidatos em batches: memória Node limitada, mas custo O(N). O cache e os índices de família reduzem esse trabalho; não há garantia de latência sob grande catálogo. +- A consulta online usa um snapshot de IDs temporário e documentos hidratados em momentos diferentes. Atualizações concorrentes podem afetar uma leitura em andamento; não é snapshot transacional de documentos. CAS impede publicar cache de uma geração antiga; perda da hidratação final de ranking retorna erro em vez de diminuir silenciosamente a página. +- Expiração da projeção usa precisão de segundos. IDs expirados deixam de aparecer na busca imediatamente; limpeza física dos SETs acontece na manutenção antes da coleta agendada ou por `expire` explícito. Cache é evitado enquanto houver membros expirados pendentes. +- Manutenção exclusiva pode aguardar uma coleta em andamento e bloqueia novas publicações durante o rebuild; a API continua lendo o namespace ativo anterior. Histórico SQL e namespaces anteriores exigem políticas operacionais futuras de retenção/cleanup. +- Não há transação distribuída PostgreSQL/Valkey. Pendências são duráveis e reparáveis; correção da reconciliação permanece explícita. Não foi criado worker automático de outbox. +- Scripts multi-chave pressupõem o Valkey independente atual; migração futura para Cluster requer desenho de hash slots. +- O build da imagem Docker e benchmark em ambiente de produção não foram executados. Compose foi validado e o binário estático CLI foi compilado. + +## Implantação e rollback + +1. Pausar o coletor antigo para a migração inicial. +2. Aplicar a migration pelo mecanismo Drizzle existente e configurar PostgreSQL/Valkey/lifetime/TTL. +3. Executar `backfill` explicitamente se houver documentos legados; se o catálogo SQL já estiver preenchido, executar `rebuild`. +4. Conferir `reconcile` read-only e ativar o novo Processor/backend. O startup recusa catálogo existente sem namespace publicado, evitando bootstrap parcial silencioso. +5. Acompanhar falhas, revisões pendentes e relatório de consistência. Guardar o namespace anterior até a validação operacional; limpar somente por comando explícito. + +Rollback de índices usa `rollback` com validação contra o SQL atual; namespace desatualizado é recusado. Se não for compatível, fazer novo rebuild do SQL. Rollback de código não remove tabela/sequence/histórico: o projeto usa migrations forward-only, sem padrão de down migration. Retornar a um Processor antigo exige pausa e plano de compatibilidade dos dados; não se deve permitir novamente que Valkey seja a única fonte durável. Nenhum rollback destrutivo foi executado. + +## Escopo revisado e itens fora da task + +Diff de código e arquivos novos revisados. Não houve alteração desta implementação em `frontend/**`, `front_admin/**` ou `package-lock.json`. Não houve nova taxonomia divergente, pós-filtro apenas após paginação, índice público other, autenticação movida ao Go ou comando de limpeza proibido. + +Uma alteração concorrente em `.cspell/custom-dictionary-workspace.txt` apareceu no workspace durante o trabalho. Não foi produzida pela PAV-125 e foi preservada, sem revertê-la ou incluí-la no conjunto abaixo. + +Fora do escopo: métricas Prometheus novas completas, match score das outras famílias, alteração do algoritmo de IDs, Redis Cluster, política de deleção histórica, lock distribuído de stampede, worker de outbox, novos filtros públicos, frontend, correções de lockfile e deploy/migration/backfill de produção. + +## Arquivos alterados pela implementação + + + +- `.env.example` +- `BACKEND.md` +- `SCRAPER.md` +- `backend/PAV-125_REPORT.md` +- `backend/drizzle/0015_job_catalog.sql` +- `backend/drizzle/meta/0015_snapshot.json` +- `backend/drizzle/meta/_journal.json` +- `backend/src/db/schema/index.ts` +- `backend/src/db/schema/jobCatalog.ts` +- `backend/src/lib/cache.ts` +- `backend/src/modules/jobs/cache/jobSearchCache.ts` +- `backend/src/modules/jobs/cache/jobSearchFingerprint.ts` +- `backend/src/modules/jobs/cache/valkeySearchCache.adapter.ts` +- `backend/src/modules/jobs/repositories/jobSearch.repository.ts` +- `backend/src/modules/jobs/repositories/valkeyJobSearch.adapter.ts` +- `backend/src/modules/jobs/services/jobMatch.service.ts` +- `backend/src/modules/jobs/services/jobProfileMatch.service.ts` +- `backend/src/modules/jobs/services/searchJobs.service.ts` +- `backend/src/scripts/seedCatalogJobs.ts` +- `backend/src/swagger.ts` +- `backend/tests/integration/jobs.valkey.test.ts` +- `backend/tests/integration/routes/searchJobs.routes.test.ts` +- `backend/tests/unit/app.test.ts` +- `backend/tests/unit/libs/cache.test.ts` +- `backend/tests/unit/modules/jobs/jobMatch.service.test.ts` +- `backend/tests/unit/modules/jobs/jobProfileMatch.service.test.ts` +- `backend/tests/unit/modules/jobs/jobSearch.repository.test.ts` +- `backend/tests/unit/modules/jobs/jobSearchCache.test.ts` +- `backend/tests/unit/modules/jobs/jobSearchFingerprint.test.ts` +- `backend/tests/unit/modules/jobs/searchJobs.service.test.ts` +- `docker-compose.migrate.yml` +- `docker-compose.yml` +- `scraper-go/Dockerfile` +- `scraper-go/cmd/catalog/main.go` +- `scraper-go/cmd/server/admin_handlers.go` +- `scraper-go/cmd/server/catalog_stream_test.go` +- `scraper-go/cmd/server/handlers.go` +- `scraper-go/cmd/server/server.go` +- `scraper-go/go.mod` +- `scraper-go/go.sum` +- `scraper-go/internal/catalog/store.go` +- `scraper-go/internal/catalogops/maintenance.go` +- `scraper-go/internal/catalogops/maintenance_integration_test.go` +- `scraper-go/internal/config/config.go` +- `scraper-go/internal/config/config_test.go` +- `scraper-go/internal/cronjob/cronjob.go` +- `scraper-go/internal/domain/job.go` +- `scraper-go/internal/jobindex/index.go` +- `scraper-go/internal/jobindex/index_test.go` +- `scraper-go/internal/jobstore/jobstore.go` +- `scraper-go/internal/pipeline/catalog_integration_test.go` +- `scraper-go/internal/pipeline/index.go` +- `scraper-go/internal/pipeline/pipeline.go` +- `scraper-go/internal/pipeline/process.go` +- `scraper-go/internal/pipeline/product_test.go` +- `scraper-go/internal/pipeline/scrape.go` diff --git a/backend/drizzle/0015_job_catalog.sql b/backend/drizzle/0015_job_catalog.sql new file mode 100644 index 00000000..a84c3682 --- /dev/null +++ b/backend/drizzle/0015_job_catalog.sql @@ -0,0 +1,19 @@ +CREATE SEQUENCE "job_catalog_revision_seq"; +--> statement-breakpoint +CREATE TABLE "job_catalog" ( + "id" text PRIMARY KEY NOT NULL, + "payload" jsonb NOT NULL, + "first_seen_at" timestamptz DEFAULT now() NOT NULL, + "last_seen_at" timestamptz DEFAULT now() NOT NULL, + "updated_at" timestamptz DEFAULT now() NOT NULL, + "expires_at" timestamptz NOT NULL, + "revision" bigint DEFAULT nextval('job_catalog_revision_seq') NOT NULL, + "indexed_revision" bigint DEFAULT 0 NOT NULL, + CONSTRAINT "job_catalog_payload_object" CHECK (jsonb_typeof(payload) = 'object'), + CONSTRAINT "job_catalog_payload_id" CHECK (payload->>'id' = id), + CONSTRAINT "job_catalog_cycle" CHECK (expires_at >= last_seen_at) +); +--> statement-breakpoint +CREATE INDEX "job_catalog_expires_id_idx" ON "job_catalog" ("expires_at", "id"); +--> statement-breakpoint +CREATE INDEX "job_catalog_pending_idx" ON "job_catalog" ("id") WHERE indexed_revision < revision; diff --git a/backend/drizzle/meta/0015_snapshot.json b/backend/drizzle/meta/0015_snapshot.json new file mode 100644 index 00000000..66d6c174 --- /dev/null +++ b/backend/drizzle/meta/0015_snapshot.json @@ -0,0 +1,1415 @@ +{ + "id": "47dfcd63-db53-4490-813b-1a346cb60daf", + "prevId": "849617ba-d9b9-4539-8b4a-bda2c9de83ce", + "version": "7", + "dialect": "postgresql", + "tables": { + "public.accounts": { + "name": "accounts", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "provider": { + "name": "provider", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "provider_account_id": { + "name": "provider_account_id", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "access_token": { + "name": "access_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "refresh_token": { + "name": "refresh_token", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "token_type": { + "name": "token_type", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "scope": { + "name": "scope", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "accounts_provider_unique": { + "name": "accounts_provider_unique", + "columns": [ + { + "expression": "provider", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "provider_account_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "accounts_user_id_users_id_fk": { + "name": "accounts_user_id_users_id_fk", + "tableFrom": "accounts", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.application_events": { + "name": "application_events", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "saved_job_id": { + "name": "saved_job_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "type": { + "name": "type", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true + }, + "from_status": { + "name": "from_status", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true + }, + "to_status": { + "name": "to_status", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true + }, + "metadata": { + "name": "metadata", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "application_events_saved_job_id_created_at_idx": { + "name": "application_events_saved_job_id_created_at_idx", + "columns": [ + { + "expression": "saved_job_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "application_events_user_id_users_id_fk": { + "name": "application_events_user_id_users_id_fk", + "tableFrom": "application_events", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "application_events_saved_job_id_saved_jobs_id_fk": { + "name": "application_events_saved_job_id_saved_jobs_id_fk", + "tableFrom": "application_events", + "tableTo": "saved_jobs", + "columnsFrom": [ + "saved_job_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.application_notes": { + "name": "application_notes", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "saved_job_id": { + "name": "saved_job_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "content": { + "name": "content", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "application_notes_user_job_idx": { + "name": "application_notes_user_job_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "saved_job_id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "application_notes_user_id_users_id_fk": { + "name": "application_notes_user_id_users_id_fk", + "tableFrom": "application_notes", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "application_notes_saved_job_id_saved_jobs_id_fk": { + "name": "application_notes_saved_job_id_saved_jobs_id_fk", + "tableFrom": "application_notes", + "tableTo": "saved_jobs", + "columnsFrom": [ + "saved_job_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.audit_logs": { + "name": "audit_logs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "serial", + "primaryKey": true, + "notNull": true + }, + "actor_id": { + "name": "actor_id", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "actor_role": { + "name": "actor_role", + "type": "user_role", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "action": { + "name": "action", + "type": "varchar(100)", + "primaryKey": false, + "notNull": true + }, + "target_type": { + "name": "target_type", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "target_id": { + "name": "target_id", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "metadata": { + "name": "metadata", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "ip": { + "name": "ip", + "type": "varchar(45)", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "audit_logs_actor_id_users_id_fk": { + "name": "audit_logs_actor_id_users_id_fk", + "tableFrom": "audit_logs", + "tableTo": "users", + "columnsFrom": [ + "actor_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "no action", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.credentials": { + "name": "credentials", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "email": { + "name": "email", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "email_hash": { + "name": "email_hash", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "password_hash": { + "name": "password_hash", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "credentials_user_id_users_id_fk": { + "name": "credentials_user_id_users_id_fk", + "tableFrom": "credentials", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "credentials_user_id_unique": { + "name": "credentials_user_id_unique", + "nullsNotDistinct": false, + "columns": [ + "user_id" + ] + }, + "credentials_email_unique": { + "name": "credentials_email_unique", + "nullsNotDistinct": false, + "columns": [ + "email" + ] + }, + "credentials_email_hash_unique": { + "name": "credentials_email_hash_unique", + "nullsNotDistinct": false, + "columns": [ + "email_hash" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.keywords": { + "name": "keywords", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "keyword": { + "name": "keyword", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "source": { + "name": "source", + "type": "text", + "primaryKey": false, + "notNull": true, + "default": "'user'" + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "keywords_user_keyword_unique": { + "name": "keywords_user_keyword_unique", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "keyword", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "keywords_user_id_users_id_fk": { + "name": "keywords_user_id_users_id_fk", + "tableFrom": "keywords", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.permission_rules": { + "name": "permission_rules", + "schema": "", + "columns": { + "resource": { + "name": "resource", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true + }, + "action": { + "name": "action", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true + }, + "min_role": { + "name": "min_role", + "type": "user_role", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "reason": { + "name": "reason", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": { + "permission_rules_resource_action_pk": { + "name": "permission_rules_resource_action_pk", + "columns": [ + "resource", + "action" + ] + } + }, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.saved_jobs": { + "name": "saved_jobs", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "job_link": { + "name": "job_link", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "job_title": { + "name": "job_title", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "company": { + "name": "company", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "location": { + "name": "location", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "source": { + "name": "source", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "keyword": { + "name": "keyword", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "varchar(50)", + "primaryKey": false, + "notNull": true, + "default": "'saved'" + }, + "applied_at": { + "name": "applied_at", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "notes": { + "name": "notes", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "saved_jobs_user_id_users_id_fk": { + "name": "saved_jobs_user_id_users_id_fk", + "tableFrom": "saved_jobs", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_notifications": { + "name": "user_notifications", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "channel": { + "name": "channel", + "type": "notification_channel", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'notification'" + }, + "type": { + "name": "type", + "type": "notification_type", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "message": { + "name": "message", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "entity_type": { + "name": "entity_type", + "type": "varchar(50)", + "primaryKey": false, + "notNull": false + }, + "entity_id": { + "name": "entity_id", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "metadata": { + "name": "metadata", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'{}'::jsonb" + }, + "read_at": { + "name": "read_at", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "user_notifications_user_created_at_idx": { + "name": "user_notifications_user_created_at_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "created_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "user_notifications_user_read_at_idx": { + "name": "user_notifications_user_read_at_idx", + "columns": [ + { + "expression": "user_id", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "read_at", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "user_notifications_user_id_users_id_fk": { + "name": "user_notifications_user_id_users_id_fk", + "tableFrom": "user_notifications", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.user_preferences": { + "name": "user_preferences", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "user_id": { + "name": "user_id", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "keywords": { + "name": "keywords", + "type": "text[]", + "primaryKey": false, + "notNull": false, + "default": "'{}'" + }, + "search_location": { + "name": "search_location", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "search_language": { + "name": "search_language", + "type": "varchar(10)", + "primaryKey": false, + "notNull": false + }, + "remote_only": { + "name": "remote_only", + "type": "boolean", + "primaryKey": false, + "notNull": false, + "default": false + }, + "job_types": { + "name": "job_types", + "type": "text[]", + "primaryKey": false, + "notNull": false, + "default": "'{}'" + }, + "email_notifications": { + "name": "email_notifications", + "type": "boolean", + "primaryKey": false, + "notNull": false, + "default": false + }, + "career_checklist": { + "name": "career_checklist", + "type": "jsonb", + "primaryKey": false, + "notNull": false, + "default": "'[]'::jsonb" + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": {}, + "foreignKeys": { + "user_preferences_user_id_users_id_fk": { + "name": "user_preferences_user_id_users_id_fk", + "tableFrom": "user_preferences", + "tableTo": "users", + "columnsFrom": [ + "user_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "user_preferences_user_id_unique": { + "name": "user_preferences_user_id_unique", + "nullsNotDistinct": false, + "columns": [ + "user_id" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.users": { + "name": "users", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true, + "default": "gen_random_uuid()" + }, + "first_name": { + "name": "first_name", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "first_name_encrypted": { + "name": "first_name_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "last_name": { + "name": "last_name", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "last_name_encrypted": { + "name": "last_name_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "display_name": { + "name": "display_name", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "display_name_encrypted": { + "name": "display_name_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "username": { + "name": "username", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "email": { + "name": "email", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "email_encrypted": { + "name": "email_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "email_hash": { + "name": "email_hash", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "email_verified": { + "name": "email_verified", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "avatar_url": { + "name": "avatar_url", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "avatar_url_encrypted": { + "name": "avatar_url_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "phone": { + "name": "phone", + "type": "varchar(20)", + "primaryKey": false, + "notNull": false + }, + "phone_encrypted": { + "name": "phone_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "cpf": { + "name": "cpf", + "type": "varchar(14)", + "primaryKey": false, + "notNull": false + }, + "cpf_encrypted": { + "name": "cpf_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "cpf_hash": { + "name": "cpf_hash", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "technologies": { + "name": "technologies", + "type": "text[]", + "primaryKey": false, + "notNull": false, + "default": "'{}'" + }, + "technologies_encrypted": { + "name": "technologies_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "technology_experiences_encrypted": { + "name": "technology_experiences_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "level": { + "name": "level", + "type": "varchar(50)", + "primaryKey": false, + "notNull": false + }, + "level_encrypted": { + "name": "level_encrypted", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "role": { + "name": "role", + "type": "user_role", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'user'" + }, + "is_blocked": { + "name": "is_blocked", + "type": "boolean", + "primaryKey": false, + "notNull": true, + "default": false + }, + "created_at": { + "name": "created_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "last_login_at": { + "name": "last_login_at", + "type": "timestamp", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "users_username_unique": { + "name": "users_username_unique", + "columns": [ + { + "expression": "username", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + }, + "users_email_unique": { + "name": "users_email_unique", + "columns": [ + { + "expression": "email", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + }, + "users_email_hash_unique": { + "name": "users_email_hash_unique", + "columns": [ + { + "expression": "email_hash", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": true, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.job_catalog": { + "name": "job_catalog", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true + }, + "payload": { + "name": "payload", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "first_seen_at": { + "name": "first_seen_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "last_seen_at": { + "name": "last_seen_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "updated_at": { + "name": "updated_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "expires_at": { + "name": "expires_at", + "type": "timestamp with time zone", + "primaryKey": false, + "notNull": true + }, + "revision": { + "name": "revision", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": "nextval('job_catalog_revision_seq')" + }, + "indexed_revision": { + "name": "indexed_revision", + "type": "bigint", + "primaryKey": false, + "notNull": true, + "default": 0 + } + }, + "indexes": { + "job_catalog_expires_id_idx": { + "name": "job_catalog_expires_id_idx", + "columns": [ + { + "expression": "expires_at", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "job_catalog_pending_idx": { + "name": "job_catalog_pending_idx", + "columns": [ + { + "expression": "id", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "where": "\"job_catalog\".\"indexed_revision\" < \"job_catalog\".\"revision\"", + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": { + "job_catalog_payload_object": { + "name": "job_catalog_payload_object", + "value": "jsonb_typeof(\"job_catalog\".\"payload\") = 'object'" + }, + "job_catalog_payload_id": { + "name": "job_catalog_payload_id", + "value": "\"job_catalog\".\"payload\"->>'id' = \"job_catalog\".\"id\"" + }, + "job_catalog_cycle": { + "name": "job_catalog_cycle", + "value": "\"job_catalog\".\"expires_at\" >= \"job_catalog\".\"last_seen_at\"" + } + }, + "isRLSEnabled": false + } + }, + "enums": { + "public.notification_channel": { + "name": "notification_channel", + "schema": "public", + "values": [ + "notification", + "message" + ] + }, + "public.notification_type": { + "name": "notification_type", + "schema": "public", + "values": [ + "job_saved", + "job_applied", + "job_status_changed", + "high_match", + "mentor", + "system" + ] + }, + "public.user_role": { + "name": "user_role", + "schema": "public", + "values": [ + "user", + "support", + "admin", + "super_admin" + ] + } + }, + "schemas": {}, + "sequences": { + "public.job_catalog_revision_seq": { + "name": "job_catalog_revision_seq", + "schema": "public", + "increment": "1", + "startWith": "1", + "minValue": "1", + "maxValue": "9223372036854775807", + "cache": "1", + "cycle": false + } + }, + "roles": {}, + "policies": {}, + "views": {}, + "_meta": { + "columns": {}, + "schemas": {}, + "tables": {} + } +} \ No newline at end of file diff --git a/backend/drizzle/meta/_journal.json b/backend/drizzle/meta/_journal.json index 797f1641..73790969 100644 --- a/backend/drizzle/meta/_journal.json +++ b/backend/drizzle/meta/_journal.json @@ -106,6 +106,13 @@ "when": 1788029383364, "tag": "0014_hard_korath", "breakpoints": true + }, + { + "idx": 15, + "version": "7", + "when": 1791360000000, + "tag": "0015_job_catalog", + "breakpoints": true } ] -} \ No newline at end of file +} diff --git a/backend/src/db/schema/index.ts b/backend/src/db/schema/index.ts index 07c1f0a4..b66ab5f5 100644 --- a/backend/src/db/schema/index.ts +++ b/backend/src/db/schema/index.ts @@ -9,3 +9,4 @@ export * from "./savedJobs"; export * from "./userNotifications"; export * from "./userPreferences"; export * from "./users"; +export * from "./jobCatalog"; diff --git a/backend/src/db/schema/jobCatalog.ts b/backend/src/db/schema/jobCatalog.ts new file mode 100644 index 00000000..6f92748a --- /dev/null +++ b/backend/src/db/schema/jobCatalog.ts @@ -0,0 +1,54 @@ +import { + bigint, + check, + pgSequence, + index, + jsonb, + pgTable, + text, + timestamp, +} from "drizzle-orm/pg-core"; + +import { sql } from "drizzle-orm"; + +export const jobCatalogRevisionSequence = pgSequence( + "job_catalog_revision_seq", +); + +// Exact Processor Job payload, including classification. No user/saved-job FK: +// stable Processor IDs remain independent of users and collection runs. +export const jobCatalog = pgTable( + "job_catalog", + { + id: text("id").primaryKey(), + payload: jsonb("payload").$type>().notNull(), + firstSeenAt: timestamp("first_seen_at", { withTimezone: true }) + .defaultNow() + .notNull(), + lastSeenAt: timestamp("last_seen_at", { withTimezone: true }) + .defaultNow() + .notNull(), + updatedAt: timestamp("updated_at", { withTimezone: true }) + .defaultNow() + .notNull(), + expiresAt: timestamp("expires_at", { withTimezone: true }).notNull(), + revision: bigint("revision", { mode: "number" }) + .default(sql`nextval('job_catalog_revision_seq')`) + .notNull(), + indexedRevision: bigint("indexed_revision", { mode: "number" }) + .default(0) + .notNull(), + }, + (table) => [ + index("job_catalog_expires_id_idx").on(table.expiresAt, table.id), + index("job_catalog_pending_idx") + .on(table.id) + .where(sql`${table.indexedRevision} < ${table.revision}`), + check( + "job_catalog_payload_object", + sql`jsonb_typeof(${table.payload}) = 'object'`, + ), + check("job_catalog_payload_id", sql`${table.payload}->>'id' = ${table.id}`), + check("job_catalog_cycle", sql`${table.expiresAt} >= ${table.lastSeenAt}`), + ], +); diff --git a/backend/src/lib/cache.ts b/backend/src/lib/cache.ts index 8f1b8354..d4fc6bc6 100644 --- a/backend/src/lib/cache.ts +++ b/backend/src/lib/cache.ts @@ -142,7 +142,6 @@ export async function cacheAbsoluteSCard(absoluteKey: string): Promise { } } - export async function cacheSearchKeywords( keywords: string[], ): Promise { @@ -157,7 +156,9 @@ export async function cacheSearchKeywords( try { const result = - keys.length === 1 ? await client.sMembers(keys[0]) : await client.sUnion(keys); + keys.length === 1 + ? await client.sMembers(keys[0]) + : await client.sUnion(keys); recordCacheOperation("search_keywords", result.length > 0 ? "hit" : "miss"); return result; } catch (error) { @@ -231,7 +232,7 @@ function keywordIndexKeyVariants(keyword: string): string[] { ]); } -function keywordSearchKeys(keywords: string[]): string[] { +export function keywordSearchKeys(keywords: string[]): string[] { return [ ...new Set(keywords.flatMap((keyword) => keywordIndexKeyVariants(keyword))), ].filter((key) => key !== "scraper:jobs:keyword:"); @@ -389,7 +390,10 @@ export async function cacheGetJobsByIdsDetailed( if (ids.length === 0) return { jobs: [], missingIds: [] }; - const keys = ids.map((id) => `scraper:job:${id}`); + const version = await client.get("scraper:jobs:index-version"); + const keys = ids.map((id) => + version ? `scraper:jobs:ns:${version}:job:${id}` : `scraper:job:${id}`, + ); let results: Array; try { results = await client.mGet(keys); @@ -459,7 +463,17 @@ async function cacheDeleteByPattern(pattern: string): Promise { cursor = result[0]; const keys = result[1] ?? []; if (keys.length > 0) { - deleted += await client.del(keys); + const count = Number( + await client.eval( + `if redis.call('GET',KEYS[1]) then return -1 end;local n=0;for i=2,#KEYS do n=n+redis.call('DEL',KEYS[i]) end;return n`, + { keys: ["scraper:jobs:index-version", ...keys], arguments: [] }, + ), + ); + if (count < 0) + throw new Error( + "Job catalog activated during legacy cache clear; retry", + ); + deleted += count; } } while (cursor !== "0"); @@ -470,6 +484,13 @@ export async function cacheClearJobs(): Promise<{ deleted: number; patterns: string[]; }> { + const client = await getCache(); + if (await client.get("scraper:jobs:index-version")) { + // Catalog projections are owned by the Processor. Clearing HTTP search + // cache must not delete an active validated namespace or durable jobs. + await client.incr("jobs:search:generation"); + return { deleted: 0, patterns: ["jobs:search:generation"] }; + } const patterns = ["scraper:job:*", "scraper:jobs:*"]; let deleted = 0; diff --git a/backend/src/modules/jobs/cache/jobSearchCache.ts b/backend/src/modules/jobs/cache/jobSearchCache.ts new file mode 100644 index 00000000..6b8a5022 --- /dev/null +++ b/backend/src/modules/jobs/cache/jobSearchCache.ts @@ -0,0 +1,87 @@ +import { jobSearchCacheKey } from "./jobSearchFingerprint"; +import type { PaginationParams } from "../../../lib/pagination"; +import type { ParsedJobSearchQuery } from "../types/jobSearch.types"; + +export type SearchPage = { jobs: unknown[]; total: number }; +export interface SearchCacheStore { + generation(): Promise; + read(key: string): Promise; + // The adapter must compare generation atomically before writing. + writeIfGeneration( + key: string, + page: SearchPage, + generation: string, + ttlSeconds: number, + ): Promise; +} + +/** Local single-flight combines simultaneous identical queries without reusing + * the scraper execution lock. Failures never remain in the in-flight map. + * Cache lookup/publication failures degrade to the normal repository query. + */ +export class JobSearchCache { + private readonly inFlight = new Map>(); + + constructor( + private readonly store: SearchCacheStore, + private readonly ttlSeconds = 120, + ) { + if (!Number.isInteger(ttlSeconds) || ttlSeconds < 1 || ttlSeconds > 86400) { + throw new Error( + "JOB_SEARCH_CACHE_TTL_SECONDS must be an integer between 1 and 86400", + ); + } + } + + async search( + filters: ParsedJobSearchQuery, + pagination: PaginationParams, + rankingContext: unknown, + query: () => Promise, + ): Promise { + let generation: string | null; + try { + generation = await this.store.generation(); + } catch { + return query(); + } + // null means indexes are unavailable or being rebuilt; do not cache. + if (generation === null) return query(); + const key = jobSearchCacheKey( + filters, + pagination, + generation, + rankingContext, + ); + const existing = this.inFlight.get(key); + if (existing) return structuredClone(await existing); + + const pending = (async () => { + try { + const cached = await this.store.read(key); + if (cached && (await this.store.generation()) === generation) + return cached; + } catch { + // A cache outage must not hide a successful persistence query. + } + const page = await query(); + try { + await this.store.writeIfGeneration( + key, + page, + generation, + this.ttlSeconds, + ); + } catch { + // The query remains useful; the next request can repopulate the cache. + } + return page; + })(); + this.inFlight.set(key, pending); + try { + return structuredClone(await pending); + } finally { + this.inFlight.delete(key); + } + } +} diff --git a/backend/src/modules/jobs/cache/jobSearchFingerprint.ts b/backend/src/modules/jobs/cache/jobSearchFingerprint.ts new file mode 100644 index 00000000..dcb3916b --- /dev/null +++ b/backend/src/modules/jobs/cache/jobSearchFingerprint.ts @@ -0,0 +1,55 @@ +import { createHash } from "node:crypto"; +import type { PaginationParams } from "../../../lib/pagination"; +import type { ParsedJobSearchQuery } from "../types/jobSearch.types"; +import { + taxonomyVersion, + professionalFamilies, +} from "../types/professionalTaxonomy"; + +// Bump when query, index or ranking semantics change, independently of taxonomy. +export const searchContractVersion = "v2"; +const text = (value: string) => value.trim().toLowerCase().replace(/\s+/g, " "); +const values = (items: string[]) => + [...new Set(items.map(text).filter(Boolean))].sort(); + +/** Accept only already validated search filters. No raw query, PII or text is + * exposed in the returned key. Profile-dependent ordering must include its + * normalized ranking inputs; never share a personalized result anonymously. + */ +export function jobSearchCacheKey( + filters: ParsedJobSearchQuery, + pagination: PaginationParams, + generation: string, + rankingContext: unknown = null, +): string { + const normalized = { + contractVersion: searchContractVersion, + taxonomyVersion, + taxonomyHash: createHash("sha256") + .update(JSON.stringify(professionalFamilies)) + .digest("hex"), + generation, + families: [...new Set(filters.families)].sort(), + familyMode: filters.familyMode, + keywords: values(filters.keywords), + technology: values(filters.technology), + company: values(filters.company), + type: values(filters.type), + level: text(filters.level), + seniority: text(filters.seniority), + location: text(filters.location), + continent: text(filters.continent), + country: text(filters.country), + state: text(filters.state), + city: text(filters.city), + contract: text(filters.contract), + ordering: filters.matchSort, + page: pagination.page, + limit: pagination.limit, + rankingContext, + }; + const fingerprint = createHash("sha256") + .update(JSON.stringify(normalized)) + .digest("hex"); + return `jobs:search:${searchContractVersion}:${fingerprint}`; +} diff --git a/backend/src/modules/jobs/cache/valkeySearchCache.adapter.ts b/backend/src/modules/jobs/cache/valkeySearchCache.adapter.ts new file mode 100644 index 00000000..79653817 --- /dev/null +++ b/backend/src/modules/jobs/cache/valkeySearchCache.adapter.ts @@ -0,0 +1,75 @@ +import { getCache } from "../../../lib/cache"; +import { taxonomyVersion } from "../types/professionalTaxonomy"; +import { + JobSearchCache, + type SearchCacheStore, + type SearchPage, +} from "./jobSearchCache"; + +const generationScript = ` +local version=redis.call('GET',KEYS[1]);if not version then return false end +local next=redis.call('ZRANGE','scraper:jobs:ns:'..version..':expires',0,0,'WITHSCORES') +local expiry=next[2] or '0' +-- Expired projection members require reconciliation; bypass caching meanwhile. +if tonumber(expiry)>0 and tonumber(expiry)<=tonumber(ARGV[1]) then return false end +return version..':'..(redis.call('GET',KEYS[2]) or '0')..':'..expiry..':'..ARGV[2] +`; +const writeScript = ` +local expected,raw,ttl,now,taxonomy=ARGV[1],ARGV[2],tonumber(ARGV[3]),tonumber(ARGV[4]),ARGV[5] +local version=redis.call('GET',KEYS[1]);if not version then return 0 end +local next=redis.call('ZRANGE','scraper:jobs:ns:'..version..':expires',0,0,'WITHSCORES') +local expiry=next[2] or '0' +local actual=version..':'..(redis.call('GET',KEYS[2]) or '0')..':'..expiry..':'..taxonomy +if actual~=expected then return 0 end +if tonumber(expiry)>0 then ttl=math.min(ttl,tonumber(expiry)-now) end +if ttl<1 then return 0 end +redis.call('SET',KEYS[3],raw,'EX',ttl);return 1 +`; +const generationKeys = ["scraper:jobs:index-version", "jobs:search:generation"]; + +export class ValkeySearchCacheStore implements SearchCacheStore { + async generation(): Promise { + const client = await getCache(); + const result = await client.eval(generationScript, { + keys: generationKeys, + arguments: [String(Math.floor(Date.now() / 1000)), taxonomyVersion], + }); + return typeof result === "string" ? result : null; + } + async read(key: string): Promise { + const raw = await (await getCache()).get(key); + if (!raw) return null; + const page = JSON.parse(raw); + return Array.isArray(page.jobs) && Number.isInteger(page.total) + ? page + : null; + } + async writeIfGeneration( + key: string, + page: SearchPage, + generation: string, + ttlSeconds: number, + ): Promise { + return ( + Number( + await ( + await getCache() + ).eval(writeScript, { + keys: [...generationKeys, key], + arguments: [ + generation, + JSON.stringify(page), + String(ttlSeconds), + String(Math.floor(Date.now() / 1000)), + taxonomyVersion, + ], + }), + ) === 1 + ); + } +} + +export const searchPageCache = new JobSearchCache( + new ValkeySearchCacheStore(), + Number(process.env.JOB_SEARCH_CACHE_TTL_SECONDS ?? 120), +); diff --git a/backend/src/modules/jobs/repositories/jobSearch.repository.ts b/backend/src/modules/jobs/repositories/jobSearch.repository.ts index de57e4d4..c9020b1f 100644 --- a/backend/src/modules/jobs/repositories/jobSearch.repository.ts +++ b/backend/src/modules/jobs/repositories/jobSearch.repository.ts @@ -1,3 +1,5 @@ +import { searchPageCache } from "../cache/valkeySearchCache.adapter"; +import { openIndexedSearch } from "./valkeyJobSearch.adapter"; import { cacheAbsoluteSMembers, cacheGetJobsByIds, @@ -19,16 +21,42 @@ export class JobSearchRepository { filters: ParsedJobSearchQuery, pagination: PaginationParams, enrich?: (jobs: unknown[]) => Promise, + rankingContext: unknown = null, + priorityKeywords: string[] = [], ): Promise<{ jobs: unknown[]; total: number }> { - const ids = [ - ...new Set( - filters.keywords.length - ? await cacheSearchKeywords(filters.keywords) - : await cacheAbsoluteSMembers("scraper:jobs:index"), - ), - ]; + return searchPageCache.search( + filters, + pagination, + { + rankingContext, + priorityKeywords: [ + ...new Set(priorityKeywords.map((x) => x.trim().toLowerCase())), + ].sort(), + }, + () => this.query(filters, pagination, enrich, priorityKeywords), + ); + } + + private async query( + filters: ParsedJobSearchQuery, + pagination: PaginationParams, + enrich?: (jobs: unknown[]) => Promise, + priorityKeywords: string[] = [], + ): Promise<{ jobs: unknown[]; total: number }> { + const indexed = await openIndexedSearch(filters, priorityKeywords); + const ids = indexed + ? [] + : [ + ...new Set( + filters.keywords.length + ? await cacheSearchKeywords(filters.keywords) + : await cacheAbsoluteSMembers("scraper:jobs:index"), + ), + ]; const offset = (pagination.page - 1) * pagination.limit; - const capacity = Math.min(ids.length, offset + pagination.limit); + const capacity = indexed + ? offset + pagination.limit + : Math.min(ids.length, offset + pagination.limit); let total = 0; const page: unknown[] = []; let ranked: RankedJob[] = []; @@ -36,36 +64,55 @@ export class JobSearchRepository { (filters.matchSort === "asc" ? a.score - b.score : b.score - a.score) || a.rank - b.rank; - for (let cursor = 0; cursor < ids.length; cursor += BATCH_SIZE) { - const matches = filterJobs( - await cacheGetJobsByIds(ids.slice(cursor, cursor + BATCH_SIZE)), - filters, - ); - if (filters.matchSort && enrich) { - const jobs = await enrich(matches); - const candidates = jobs.map((job, index) => ({ - id: String((job as { id: string }).id), - rank: total + index, - score: (job as { matchScore?: number }).matchScore ?? 0, - })); - ranked = [...ranked, ...candidates].sort(compare).slice(0, capacity); - } else { - for (const job of matches) { - if (total >= offset && page.length < pagination.limit) page.push(job); - total++; + async function* legacyBatches() { + for (let cursor = 0; cursor < ids.length; cursor += BATCH_SIZE) + yield await cacheGetJobsByIds(ids.slice(cursor, cursor + BATCH_SIZE)); + } + try { + for await (const batch of indexed ? indexed.batches() : legacyBatches()) { + const matches = filterJobs(batch, filters); + if (filters.matchSort && enrich) { + const jobs = await enrich(matches); + const candidates = jobs.map((job, index) => ({ + id: String((job as { id: string }).id), + rank: total + index, + score: (job as { matchScore?: number }).matchScore ?? 0, + })); + if (indexed) await indexed.rank(candidates, filters.matchSort); + else + ranked = [...ranked, ...candidates] + .sort(compare) + .slice(0, capacity); + } else { + for (const job of matches) { + if (total >= offset && page.length < pagination.limit) + page.push(job); + total++; + } + continue; } - continue; + total += matches.length; + } + if (indexed && filters.matchSort && enrich) { + page.push( + ...(await indexed.hydrate( + await indexed.rankedIds(offset, pagination.limit, total), + )), + ); } - total += matches.length; + } finally { + await indexed?.close(); } return { jobs: filters.matchSort && enrich - ? await cacheGetJobsByIds( - ranked - .slice(offset, offset + pagination.limit) - .map((item) => item.id), - ) + ? indexed + ? page + : await cacheGetJobsByIds( + ranked + .slice(offset, offset + pagination.limit) + .map((item) => item.id), + ) : page, total, }; diff --git a/backend/src/modules/jobs/repositories/valkeyJobSearch.adapter.ts b/backend/src/modules/jobs/repositories/valkeyJobSearch.adapter.ts new file mode 100644 index 00000000..f0b5604e --- /dev/null +++ b/backend/src/modules/jobs/repositories/valkeyJobSearch.adapter.ts @@ -0,0 +1,155 @@ +import { randomUUID } from "node:crypto"; +import { getCache, keywordSearchKeys } from "../../../lib/cache"; +import type { ParsedJobSearchQuery } from "../types/jobSearch.types"; +import { normalizeJobTaxonomy } from "../types/professionalTaxonomy"; + +export const activeIndexKey = "scraper:jobs:index-version"; +export const searchGenerationKey = "jobs:search:generation"; + +// Snapshot only candidate IDs on the server, use lexical ZSET pagination and +// stream batches. All residual predicates run before final pagination/count. +// Expiration is checked with the catalog projection's expiresAt, not key TTL. +const candidatesScript = ` +local prefix,groups,target,now=ARGV[1],cjson.decode(ARGV[2]),KEYS[2],tonumber(ARGV[3]) +if redis.call('GET',KEYS[1])~=ARGV[4] then return redis.error_reply('index version changed') end +local temp={};local inputs={} +local priority=cjson.decode(ARGV[5]);local preferred=target..':preferred' +if #priority>0 then redis.call('SUNIONSTORE',preferred,unpack(priority));redis.call('EXPIRE',preferred,120) end +for i,group in ipairs(groups) do + local key=target..':g'..i + redis.call('SUNIONSTORE',key,unpack(group));redis.call('EXPIRE',key,120) + table.insert(temp,key);table.insert(inputs,key) +end +local combined=target..':set' +if #inputs==0 then redis.call('SUNIONSTORE',combined,prefix..'index') +else redis.call('SINTERSTORE',combined,unpack(inputs)) end +redis.call('EXPIRE',combined,120) +local live=target..':live' +redis.call('ZRANGESTORE',live,prefix..'expires','('..now,'+inf','BYSCORE');redis.call('EXPIRE',live,120) +redis.call('ZINTERSTORE',target,2,combined,live,'WEIGHTS',0,0);redis.call('EXPIRE',target,120) +if #priority>0 then + local overlap=target..':overlap' + redis.call('ZINTERSTORE',overlap,2,target,preferred,'WEIGHTS',0,-1);redis.call('EXPIRE',overlap,120) + redis.call('ZUNIONSTORE',target,2,target,overlap,'AGGREGATE','MIN');redis.call('EXPIRE',target,120) + redis.call('DEL',overlap) +end +redis.call('DEL',combined,preferred,live) +for _,key in ipairs(temp) do redis.call('DEL',key) end +return redis.call('ZCARD',target) +`; + +function keywordKeys(prefix: string, keywords: string[]): string[] { + return keywordSearchKeys(keywords).map( + (key) => prefix + key.slice("scraper:jobs:".length), + ); +} + +export type IndexedSearch = { + batches(): AsyncGenerator; + hydrate(ids: string[]): Promise; + rank( + scores: { id: string; score: number }[], + direction: "asc" | "desc", + ): Promise; + rankedIds(offset: number, limit: number, total: number): Promise; + close(): Promise; +}; + +export async function openIndexedSearch( + filters: ParsedJobSearchQuery, + priorityKeywords: string[] = [], +): Promise { + const client = await getCache(); + const version = await client.get(activeIndexKey); + // Before the first validated rebuild, preserve PAV-124's exact fallback. + if (!version) return null; + const prefix = `scraper:jobs:ns:${version}:`; + const key = `jobs:search:candidates:${randomUUID()}`; + const groups: string[][] = []; + if (filters.families.length) + groups.push( + filters.families.map( + (family) => + `${prefix}family:${filters.familyMode === "primary" ? "primary:" : ""}${family}`, + ), + ); + if (filters.keywords.length) + groups.push(keywordKeys(prefix, filters.keywords)); + // Existing non-family index inference differs from the HTTP predicate in + // some aliases/locations. Verify these filters in bounded hydrated batches + // instead of dropping valid jobs with an unsafe index intersection. + let count: number; + try { + count = Number( + await client.eval(candidatesScript, { + keys: [activeIndexKey, key], + arguments: [ + prefix, + JSON.stringify(groups), + String(Math.floor(Date.now() / 1000)), + version, + JSON.stringify(keywordKeys(prefix, priorityKeywords)), + ], + }), + ); + } catch (error) { + await client.del(key).catch(() => {}); + throw error; + } + return { + async *batches() { + for (let offset = 0; ; offset += 200) { + const ids = await client.zRange(key, offset, offset + 199); + if (!ids.length) { + if (offset < count) + throw new Error("Search candidate snapshot expired"); + return; + } + await client.expire(key, 120); + const raws = await client.mGet(ids.map((id) => `${prefix}job:${id}`)); + const jobs: unknown[] = []; + for (const raw of raws) + if (raw) { + try { + jobs.push(normalizeJobTaxonomy(JSON.parse(raw))); + } catch { + /* invalid projection is reconciled by the Processor */ + } + } + yield jobs; + } + }, + async rank(scores, direction) { + if (!scores.length) return; + await client.zAdd( + `${key}:rank`, + scores.map((item) => ({ + value: item.id, + score: direction === "desc" ? -item.score : item.score, + })), + ); + await client.expire(`${key}:rank`, 120); + }, + async rankedIds(offset, limit, total) { + if (total && (await client.zCard(`${key}:rank`)) !== total) + throw new Error("Search ranking snapshot expired"); + return client.zRange(`${key}:rank`, offset, offset + limit - 1); + }, + async hydrate(ids) { + if (!ids.length) return []; + const raws = await client.mGet(ids.map((id) => `${prefix}job:${id}`)); + return raws.map((raw) => { + if (!raw) + throw new Error("Search projection changed during ranking hydration"); + return normalizeJobTaxonomy(JSON.parse(raw)); + }); + }, + async close() { + await client.del([key, `${key}:rank`]); + }, + }; +} + +export async function hasActiveJobIndex(): Promise { + return Boolean(await (await getCache()).get(activeIndexKey)); +} diff --git a/backend/src/modules/jobs/services/jobMatch.service.ts b/backend/src/modules/jobs/services/jobMatch.service.ts index 5f8224b2..7d47fdf1 100644 --- a/backend/src/modules/jobs/services/jobMatch.service.ts +++ b/backend/src/modules/jobs/services/jobMatch.service.ts @@ -26,6 +26,7 @@ export type MatchedJob = MatchableJob & { matchScore?: number; matchSource?: "backend_profile"; matchedTechnologies?: string[]; + matchReasons?: string[]; }; function normalizeMatchText(value: string) { @@ -159,3 +160,128 @@ export function jobNotificationIdentity(job: MatchableJob | SavedJob) { : ""; return url.trim() || String(job.id ?? ""); } + +export type MatchPreferences = { + family?: string; + seniority?: string; + modality?: string; + modalities?: string[]; + location?: string; + contract?: string; + skills?: TechnologyExperience[]; +}; + +/** Only Product/Design use this additive evidence model. Missing development + * languages never enter its denominator. Other families retain the old formula. + * The current user model supplies level and named experiences; optional future + * preferences are evaluated only when actually supplied by the backend. + */ +export function scoreProfessionalJob( + job: MatchableJob, + technologies: TechnologyExperience[], + preferences: MatchPreferences = {}, +): MatchedJob { + const classification = job.classification as + | { primaryFamily?: string; relatedFamilies?: string[]; seniority?: string } + | undefined; + const family = classification?.primaryFamily; + if (family !== "product" && family !== "product_design") + return scoreJobWithTechnologies(job, technologies); + const text = jobMatchText(job); + const vocabulary = + family === "product" + ? /\b(product|produto|discovery|roadmap|backlog|priorizacao|analytics|amplitude|mixpanel|jira|sql|agile|scrum|kanban|stakeholder|experimentos|estrategia)\b/ + : /\b(ux|ui|design|figma|sketch|adobe|pesquisa|research|prototipacao|prototype|acessibilidade|accessibility|wireframe|usabilidade|html|css)\b/; + const relevant = [...technologies, ...(preferences.skills ?? [])].filter( + (t) => vocabulary.test(normalizeMatchText(t.name)), + ); + const normalizedSkills = new Map(); + for (const skill of relevant) { + const key = normalizeMatchText(skill.name); + const old = normalizedSkills.get(key); + normalizedSkills.set(key, { + name: + old && old.name.localeCompare(skill.name.trim()) <= 0 + ? old.name + : skill.name.trim(), + years: Math.max(old?.years ?? 0, skill.years), + }); + } + const matched = [...normalizedSkills.values()] + .filter((skill) => + matchAliases(skill.name).some((alias) => textMatchesAlias(text, alias)), + ) + .sort((a, b) => a.name.localeCompare(b.name)); + const reasons: string[] = []; + let evidence = 0; + if ( + preferences.family && + (preferences.family === family || + classification?.relatedFamilies?.includes(preferences.family)) + ) { + evidence += 15; + reasons.push("família profissional compatível"); + } + const predicates: [string | undefined, string | undefined | null, string][] = + [ + [ + preferences.seniority, + classification?.seniority ?? job.level, + "senioridade compatível", + ], + [preferences.modality, job.modality ?? job.type, "modalidade compatível"], + [preferences.location, job.location, "localização compatível"], + [preferences.contract, text, "contrato compatível"], + ]; + for (const [wanted, actual, reason] of predicates) + if ( + wanted && + actual && + textMatchesAlias(normalizeMatchText(actual), normalizeMatchText(wanted)) + ) { + evidence += 7; + reasons.push(reason); + } + if ( + !preferences.modality && + preferences.modalities?.some((mode) => + textMatchesAlias( + normalizeMatchText(job.modality ?? job.type ?? text) + .replace(/\bremote\b/g, "remoto") + .replace(/\bhybrid\b/g, "hibrido"), + normalizeMatchText(mode), + ), + ) + ) { + evidence += 7; + reasons.push("modalidade compatível"); + } + if (matched.length) { + evidence += Math.min( + 30, + matched.reduce( + (sum, t) => + sum + + (family === "product_design" && + /^(html|css)$/.test(normalizeMatchText(t.name)) + ? 2 + : 8 + Math.min(5, Math.max(0, t.years))), + 0, + ), + ); + reasons.push( + family === "product" + ? "competências e ferramentas de Produto relacionadas" + : "competências e ferramentas de Design relacionadas", + ); + } + // Without profile evidence, do not invent a compatibility estimate. + if (!evidence) return job; + return { + ...job, + matchScore: Math.min(99, Math.round(55 + evidence)), + matchSource: "backend_profile", + matchedTechnologies: matched.map((t) => t.name), + matchReasons: reasons, + }; +} diff --git a/backend/src/modules/jobs/services/jobProfileMatch.service.ts b/backend/src/modules/jobs/services/jobProfileMatch.service.ts index a8c4793c..91a4c8c3 100644 --- a/backend/src/modules/jobs/services/jobProfileMatch.service.ts +++ b/backend/src/modules/jobs/services/jobProfileMatch.service.ts @@ -1,25 +1,61 @@ import { logWarn } from "../../../logger"; import { NotificationsService } from "../../notifications/notifications.service"; import { UsersService } from "../../users/users.service"; +import { updatePreferencesSchema } from "../../users/schemas/user.schemas"; import { MatchTechnology } from "../types/jobSearch.types"; import { getUserMatchTechnologies, MatchableJob, MatchedJob, - scoreJobWithTechnologies, + scoreProfessionalJob, + type MatchPreferences, } from "./jobMatch.service"; export class JobProfileMatchService { - async getUserTechnologies(userId?: string): Promise { + async getUserTechnologies( + userId?: string, + capture?: (preferences: MatchPreferences) => void, + ): Promise { if (!userId) return []; try { - const user = await new UsersService().getUserById(userId); + const usersService = new UsersService(); + const user = await usersService.getUserById(userId); + if (capture && user) { + const preferences: MatchPreferences = { + seniority: user.level ?? undefined, + }; + try { + const saved = await usersService.getPreferences(userId); + if (saved) { + preferences.location = saved.searchLocation ?? undefined; + preferences.modality = saved.remoteOnly ? "remoto" : undefined; + // Despite its name, jobTypes is the HTTP modality enum. Validate + // persisted text[] too; legacy contract values must not become + // modalities. This model has no authoritative contract/family field. + const modalities = updatePreferencesSchema.shape.jobTypes.safeParse( + saved.jobTypes ?? [], + ); + preferences.modalities = modalities.success + ? (modalities.data ?? []) + : []; + // Keywords are free search terms, not canonical family selections. + preferences.skills = (saved.keywords ?? []).map((name) => ({ + name, + years: 1, + })); + } + } catch { + logWarn( + "Não foi possível carregar preferências para cálculo de match", + ); + } + if (Object.values(preferences).some(Boolean)) capture(preferences); + } return getUserMatchTechnologies(user); } catch (error) { logWarn("Não foi possível carregar perfil para cálculo de match", { - userId, - error: (error as Error).message, + code: "PROFILE_READ_FAILED", }); return []; @@ -30,14 +66,15 @@ export class JobProfileMatchService { userId: string | undefined, jobs: MatchableJob[], technologies: MatchTechnology[], - options: { notifyHighMatches?: boolean } = {}, + options: { + notifyHighMatches?: boolean; + preferences?: MatchPreferences; + } = {}, ): Promise { - if (technologies.length === 0) { + if (!technologies.length && !options.preferences) return jobs as MatchedJob[]; - } - const matchedJobs = jobs.map((job) => - scoreJobWithTechnologies(job, technologies), + scoreProfessionalJob(job, technologies, options.preferences), ); if (options.notifyHighMatches !== false) { @@ -61,9 +98,7 @@ export class JobProfileMatchService { .map((job) => notifications.createHighMatchIfMissing(userId, job).catch((error) => { logWarn("Não foi possível registrar notificação de alto match", { - error: (error as Error).message, - userId, - job: job.title ?? job.jobTitle ?? job.id, + code: "MATCH_NOTIFICATION_FAILED", }); }), ), diff --git a/backend/src/modules/jobs/services/searchJobs.service.ts b/backend/src/modules/jobs/services/searchJobs.service.ts index dd4fe5d3..5d47b400 100644 --- a/backend/src/modules/jobs/services/searchJobs.service.ts +++ b/backend/src/modules/jobs/services/searchJobs.service.ts @@ -1,3 +1,4 @@ +import { hasActiveJobIndex } from "../repositories/valkeyJobSearch.adapter"; import { JobSearchRepository } from "../repositories/jobSearch.repository"; import { cacheAbsoluteSMembers, @@ -18,7 +19,7 @@ import type { SearchJobsInput, SearchJobsResult, } from "../types/jobSearch.types"; -import type { MatchableJob } from "./jobMatch.service"; +import type { MatchPreferences, MatchableJob } from "./jobMatch.service"; import { JobProfileMatchService } from "./jobProfileMatch.service"; async function legacyResolveIds( @@ -161,8 +162,15 @@ export class SearchJobsService { const filters = parseJobSearchQuery(input.query); const pagination = parsePagination(input.query); const hasFilters = hasStructuredFilters(filters); + let preferences: MatchPreferences | undefined; const matchTechnologies = - await this.profileMatchService.getUserTechnologies(input.userId); + await this.profileMatchService.getUserTechnologies( + input.userId, + (value) => { + preferences = value; + }, + ); + const preferenceOptions = preferences ? { preferences } : {}; let ids: string[] = []; let source = @@ -170,7 +178,17 @@ export class SearchJobsService { ? `valkey_filtered_by_keywords:${filters.keywords.join("+")}` : "valkey_global_index"; - if (hasFilters || hasPostOnlyFilters(filters) || filters.matchSort) { + const indexedDefault = + !hasFilters && + !hasPostOnlyFilters(filters) && + !filters.matchSort && + (await hasActiveJobIndex()); + if ( + hasFilters || + hasPostOnlyFilters(filters) || + filters.matchSort || + indexedDefault + ) { const result = await this.repository.search( filters, pagination, @@ -180,18 +198,28 @@ export class SearchJobsService { input.userId, jobs as MatchableJob[], matchTechnologies, - { notifyHighMatches: false }, + { notifyHighMatches: false, ...preferenceOptions }, ) : undefined, + { + preferences, + technologies: [...matchTechnologies] + .map((t) => ({ name: t.name.trim().toLowerCase(), years: t.years })) + .sort((a, b) => a.name.localeCompare(b.name) || a.years - b.years), + }, + indexedDefault ? matchTechnologies.map((t) => t.name) : [], ); const jobs = await this.profileMatchService.enrich( input.userId, result.jobs as MatchableJob[], matchTechnologies, + ...(preferences ? [preferenceOptions] : []), ); const totalPages = Math.ceil(result.total / pagination.limit); return toSearchResult( - jobs, + indexedDefault && matchTechnologies.length + ? sortJobsByMatch(jobs, "desc") + : jobs, { ...pagination, total: result.total, @@ -216,6 +244,7 @@ export class SearchJobsService { input.userId, pageJobs as MatchableJob[], matchTechnologies, + ...(preferences ? [preferenceOptions] : []), ); if (relevance.matchedIds === 0) { diff --git a/backend/src/scripts/seedCatalogJobs.ts b/backend/src/scripts/seedCatalogJobs.ts index 0ec87d1a..83ac4f32 100644 --- a/backend/src/scripts/seedCatalogJobs.ts +++ b/backend/src/scripts/seedCatalogJobs.ts @@ -412,6 +412,10 @@ export async function seedCatalogJobs(): Promise { } try { + if (await client.get("scraper:jobs:index-version")) { + console.log("! Catálogo durável ativo: seed legado de vagas ignorado; indexação pertence ao Processor."); + return; + } for (const spec of SEED_CATALOG_JOBS) { const id = catalogJobId(spec); diff --git a/backend/src/swagger.ts b/backend/src/swagger.ts index 4d898108..cde64dc4 100644 --- a/backend/src/swagger.ts +++ b/backend/src/swagger.ts @@ -241,6 +241,9 @@ const options: swaggerJsdoc.Options = { }, source: { type: "string", example: "LinkedIn" }, keyword: { type: "string", example: "node" }, + matchScore: { type: "integer", minimum: 0, maximum: 100, description: "Compatibilidade quando existem evidências no perfil; não é garantida em toda vaga." }, + matchedTechnologies: { type: "array", items: { type: "string" } }, + matchReasons: { type: "array", items: { type: "string" }, description: "Razões públicas opcionais para Produto/Design, sem pesos ou dados privados." }, }, }, JobSearchResponse: { @@ -604,7 +607,7 @@ const options: swaggerJsdoc.Options = { summary: "Busca vagas", security: auth, description: - "OR entre famílias; AND com os demais filtros. primary considera somente classification.primaryFamily; any (default) considera principal ou relacionada. Full Stack (fullstack) é independente de backend/frontend; devops e platform são independentes. IDs são case-sensitive; labels, aliases e other não são aceitos. Valores vazios são removidos, mas family presente sem IDs retorna 400. Duplicidades são removidas antes do máximo de 13 famílias únicas e IDs são ordenados. familyMode sem family é validado e não altera a busca. Total e paginação usam o mesmo predicado antes de paginar.", + "Índices Valkey versionados com união de famílias e verificação dos demais predicados antes de total/paginação. Match de Produto/Design usa evidências disponíveis sem exigir linguagens; matchReasons é opcional. OR entre famílias; AND com os demais filtros. primary considera somente classification.primaryFamily; any (default) considera principal ou relacionada. Full Stack (fullstack) é independente de backend/frontend; devops e platform são independentes. IDs são case-sensitive; labels, aliases e other não são aceitos. Valores vazios são removidos, mas family presente sem IDs retorna 400. Duplicidades são removidas antes do máximo de 13 famílias únicas e IDs são ordenados. familyMode sem family é validado e não altera a busca. Total e paginação usam o mesmo predicado antes de paginar.", parameters: [ { in: "query", diff --git a/backend/tests/integration/jobs.valkey.test.ts b/backend/tests/integration/jobs.valkey.test.ts new file mode 100644 index 00000000..0959a08d --- /dev/null +++ b/backend/tests/integration/jobs.valkey.test.ts @@ -0,0 +1,362 @@ +import { + afterAll, + beforeAll, + beforeEach, + describe, + expect, + it, + vi, +} from "vitest"; +import { createClient } from "redis"; +const mocks = vi.hoisted(() => ({ getCache: vi.fn() })); +vi.mock("../../src/lib/cache", async (importOriginal) => ({ + ...(await importOriginal()), + getCache: mocks.getCache, +})); +import { JobSearchRepository } from "../../src/modules/jobs/repositories/jobSearch.repository"; +import { parseJobSearchQuery } from "../../src/modules/jobs/parsers/jobSearchQuery.parser"; +import { ValkeySearchCacheStore } from "../../src/modules/jobs/cache/valkeySearchCache.adapter"; +import { JobSearchCache } from "../../src/modules/jobs/cache/jobSearchCache"; +import { jobSearchCacheKey } from "../../src/modules/jobs/cache/jobSearchFingerprint"; +import { openIndexedSearch } from "../../src/modules/jobs/repositories/valkeyJobSearch.adapter"; + +// Explicit opt-in, dedicated disposable instance only. Never erase a database. +const url = process.env.PAV125_TEST_VALKEY_URL; +describe.skipIf(!url)("PAV-125 real Valkey search and cache", () => { + const client = createClient({ url }); + const version = `v2-test-${process.pid}`; + const prefix = `scraper:jobs:ns:${version}:`; + const owned = new Set(); + const jobs = [ + { + id: "a", + title: "Backend senior", + modality: "remoto", + location: "Brasil", + description: "CLT", + classification: { + primaryFamily: "backend", + relatedFamilies: ["platform"], + seniority: "senior", + }, + }, + { + id: "b", + title: "Leadership senior", + modality: "remoto", + location: "Brasil", + description: "CLT", + classification: { + primaryFamily: "leadership", + relatedFamilies: ["backend"], + seniority: "senior", + }, + }, + { + id: "c", + title: "Fullstack junior", + modality: "presencial", + location: "Portugal", + description: "PJ", + classification: { + primaryFamily: "fullstack", + relatedFamilies: [], + seniority: "junior", + }, + }, + { + id: "d", + title: "DevOps senior", + modality: "remoto", + location: "Brasil", + description: "CLT", + classification: { + primaryFamily: "devops", + relatedFamilies: ["platform"], + seniority: "senior", + }, + }, + { + id: "e", + title: "Platform senior", + modality: "remoto", + location: "Brasil", + description: "CLT", + classification: { + primaryFamily: "platform", + relatedFamilies: [], + seniority: "senior", + }, + }, + { + id: "p", + title: "Product senior", + modality: "remoto", + location: "Brasil", + description: "CLT Jira", + classification: { + primaryFamily: "product", + relatedFamilies: [], + seniority: "senior", + }, + }, + { + id: "q", + title: "Product Design senior", + modality: "remoto", + location: "Brasil", + description: "CLT Figma", + classification: { + primaryFamily: "product_design", + relatedFamilies: [], + seniority: "senior", + }, + }, + ]; + beforeAll(async () => { + await client.connect(); + mocks.getCache.mockResolvedValue(client); + }); + afterAll(async () => { + if (client.isOpen) { + for (const key of owned) await client.del(key); + await client.quit(); + } + }); + beforeEach(async () => { + for (const key of owned) await client.del(key); + owned.clear(); + owned.add("scraper:jobs:index-version"); + owned.add("jobs:search:generation"); + await client.set("scraper:jobs:index-version", version); + await client.set("jobs:search:generation", "1"); + const end = Math.floor(Date.now() / 1000) + 3600; + for (const job of jobs) { + owned.add(prefix + "job:" + job.id); + await client.set(prefix + "job:" + job.id, JSON.stringify(job)); + owned.add(prefix + "index"); + await client.sAdd(prefix + "index", job.id); + owned.add(prefix + "expires"); + await client.zAdd(prefix + "expires", { score: end, value: job.id }); + const families = [ + job.classification.primaryFamily, + ...job.classification.relatedFamilies, + ]; + for (const family of families) { + owned.add(prefix + "family:" + family); + await client.sAdd(prefix + "family:" + family, job.id); + } + const primary = + prefix + "family:primary:" + job.classification.primaryFamily; + owned.add(primary); + await client.sAdd(primary, job.id); + for (const family of job.classification.relatedFamilies) { + owned.add(prefix + "family:related:" + family); + await client.sAdd(prefix + "family:related:" + family, job.id); + } + } + }); + async function search(query: Record, page = 1, limit = 20) { + return new JobSearchRepository().search(parseJobSearchQuery(query), { + page, + limit, + }); + } + function ids(result: { jobs: unknown[] }) { + return result.jobs.map((j) => (j as { id: string }).id); + } + it.each([ + ["backend", "primary", ["a"]], + ["backend", "any", ["a", "b"]], + ["backend,fullstack", "primary", ["a", "c"]], + ["backend,fullstack", "any", ["a", "b", "c"]], + ["fullstack", "primary", ["c"]], + ["backend,frontend,fullstack", "primary", ["a", "c"]], + ["devops", "primary", ["d"]], + ["platform", "primary", ["e"]], + ["devops,platform", "primary", ["d", "e"]], + ["platform", "any", ["a", "d", "e"]], + ["product", "any", ["p"]], + ["product_design", "primary", ["q"]], + ])("union %s / %s", async (family, familyMode, expected) => { + const result = await search({ family, familyMode }); + expect(ids(result)).toEqual(expected); + expect(result.total).toBe(expected.length); + }); + it("intersects before page and count", async () => { + const query = { + family: "backend,fullstack", + familyMode: "any", + seniority: "senior", + type: "remoto", + location: "Brasil", + contract: "clt", + }; + const first = await search(query, 1, 1), + second = await search(query, 2, 1); + expect(ids(first)).toEqual(["a"]); + expect(ids(second)).toEqual(["b"]); + expect(first.total).toBe(2); + expect(second.total).toBe(2); + expect( + (await search({ ...query, family: "product,product_design" })).total, + ).toBe(2); + }); + + it("bounds hydration for 1001 candidates and ranks without retaining catalog payloads", async () => { + const multi = client.multi(); + const family = prefix + "family:backend", + primary = prefix + "family:primary:backend"; + owned.add(family); + owned.add(primary); + for (let i = 0; i < 1001; i++) { + const id = `load-${String(i).padStart(4, "0")}`; + const key = prefix + "job:" + id; + owned.add(key); + multi.set( + key, + JSON.stringify({ + id, + title: "Backend", + classification: { primaryFamily: "backend" }, + }), + ); + multi.sAdd(family, id); + multi.sAdd(primary, id); + multi.sAdd(prefix + "index", id); + multi.zAdd(prefix + "expires", { + value: id, + score: Math.floor(Date.now() / 1000) + 3600, + }); + } + await multi.exec(); + const hydration = vi.spyOn(client, "mGet"); + try { + const result = await search( + { family: "backend", familyMode: "primary" }, + 21, + 50, + ); + expect(result.total).toBe(1002); + expect(result.jobs).toHaveLength(2); + expect(hydration.mock.calls.every(([keys]) => keys.length <= 200)).toBe( + true, + ); + } finally { + hydration.mockRestore(); + } + }); + it("normalizes equivalent queries and separates primary cache", async () => { + const store = new ValkeySearchCacheStore(); + const generation = (await store.generation())!; + const cache = new JobSearchCache(store, 90); + const page = { page: 1, limit: 20 }; + const q = vi.fn().mockResolvedValue({ jobs: [], total: 0 }); + const a = parseJobSearchQuery({ family: "backend,fullstack" }), + b = parseJobSearchQuery({ + family: ["fullstack", "backend,backend"], + familyMode: "any", + }); + const key = jobSearchCacheKey(a, page, generation); + owned.add(key); + await cache.search(a, page, null, q); + await cache.search(b, page, null, q); + expect(q).toHaveBeenCalledTimes(1); + expect(await client.ttl(key)).toBeGreaterThan(0); + expect( + jobSearchCacheKey( + parseJobSearchQuery({ family: "backend", familyMode: "primary" }), + page, + generation, + ), + ).not.toBe( + jobSearchCacheKey( + parseJobSearchQuery({ family: "backend" }), + page, + generation, + ), + ); + }); + it("rejects stale-generation writes and limits TTL by expiry", async () => { + const store = new ValkeySearchCacheStore(); + let generation = (await store.generation())!; + await client.incr("jobs:search:generation"); + const key = "jobs:search:v2:integration-stale"; + owned.add(key); + expect( + await store.writeIfGeneration( + key, + { jobs: [], total: 0 }, + generation, + 120, + ), + ).toBe(false); + expect(await client.exists(key)).toBe(0); + await client.zAdd(prefix + "expires", { + value: "a", + score: Math.floor(Date.now() / 1000) + 5, + }); + generation = (await store.generation())!; + expect( + await store.writeIfGeneration( + key, + { jobs: [], total: 0 }, + generation, + 120, + ), + ).toBe(true); + expect(await client.ttl(key)).toBeLessThanOrEqual(5); + await client.zAdd(prefix + "expires", { + value: "a", + score: Math.floor(Date.now() / 1000) - 1, + }); + expect(await store.generation()).toBe(null); + }); + it("filters committed expiration and cleans temporary candidates", async () => { + await client.zAdd(prefix + "expires", { + value: "a", + score: Math.floor(Date.now() / 1000) - 1, + }); + const result = await search({ family: "backend" }); + expect(ids(result)).toEqual(["b"]); + expect(result.total).toBe(1); + const indexed = (await openIndexedSearch( + parseJobSearchQuery({ family: "backend" }), + ))!; + await indexed.close(); + // SCAN audit, no KEYS command or broad cleanup. + const candidates = []; + for await (const keys of client.scanIterator({ + MATCH: "jobs:search:candidates:*", + COUNT: 200, + })) + candidates.push(...keys); + expect(candidates).toEqual([]); + }); + it("preserves profile keyword priority and global match ordering", async () => { + const kw = prefix + "keyword:figma"; + owned.add(kw); + await client.sAdd(kw, "q"); + const result = await new JobSearchRepository().search( + parseJobSearchQuery({}), + { page: 1, limit: 1 }, + undefined, + null, + ["figma"], + ); + expect(ids(result)).toEqual(["q"]); + expect(result.total).toBe(7); + const sorted = await new JobSearchRepository().search( + parseJobSearchQuery({ family: "backend", matchSort: "desc" }), + { page: 1, limit: 1 }, + async (batch) => + batch.map((j) => ({ + ...(j as object), + matchScore: (j as { id: string }).id === "b" ? 90 : 50, + })), + "profile", + ); + expect(ids(sorted)).toEqual(["b"]); + expect(sorted.total).toBe(2); + }); +}); diff --git a/backend/tests/integration/routes/searchJobs.routes.test.ts b/backend/tests/integration/routes/searchJobs.routes.test.ts index 62b4763b..5ccc025d 100644 --- a/backend/tests/integration/routes/searchJobs.routes.test.ts +++ b/backend/tests/integration/routes/searchJobs.routes.test.ts @@ -1,3 +1,5 @@ +vi.mock("../../../src/modules/jobs/repositories/valkeyJobSearch.adapter", () => ({ openIndexedSearch: vi.fn().mockResolvedValue(null), hasActiveJobIndex: vi.fn().mockResolvedValue(false) })); +vi.mock("../../../src/modules/jobs/cache/valkeySearchCache.adapter", () => ({ searchPageCache: {search: async (_f: unknown, _p: unknown, _r: unknown, query: () => Promise) => query()} })); import request from "supertest"; import { beforeEach, describe, expect, it, vi } from "vitest"; @@ -26,6 +28,7 @@ vi.mock("iron-session", () => ({ vi.mock("../../../src/modules/users/users.service", () => ({ UsersService: class { getUserById = profileMocks.getUserById; + getPreferences = vi.fn().mockResolvedValue(undefined); }, })); diff --git a/backend/tests/unit/app.test.ts b/backend/tests/unit/app.test.ts index 3abe8c11..bb4a7546 100644 --- a/backend/tests/unit/app.test.ts +++ b/backend/tests/unit/app.test.ts @@ -1,3 +1,5 @@ +vi.mock("../../src/modules/jobs/repositories/valkeyJobSearch.adapter", () => ({ openIndexedSearch: vi.fn().mockResolvedValue(null), hasActiveJobIndex: vi.fn().mockResolvedValue(false) })); +vi.mock("../../src/modules/jobs/cache/valkeySearchCache.adapter", () => ({ searchPageCache: {search: async (_f: unknown, _p: unknown, _r: unknown, query: () => Promise) => query()} })); import request from "supertest"; import { beforeEach, describe, expect, it, vi } from "vitest"; diff --git a/backend/tests/unit/libs/cache.test.ts b/backend/tests/unit/libs/cache.test.ts index 70b01a67..b7604645 100644 --- a/backend/tests/unit/libs/cache.test.ts +++ b/backend/tests/unit/libs/cache.test.ts @@ -53,6 +53,8 @@ vi.mock("redis", () => { sendCommand: vi.fn(), expire: vi.fn(), mGet: vi.fn(), + incr: vi.fn(), + eval: vi.fn(), }; return { createClient: vi.fn(() => mockClient), @@ -66,6 +68,7 @@ describe("Valkey Cache Lib", () => { vi.stubEnv("VALKEY_URL", "redis://localhost:6379"); mockClientInstance = createClient(); vi.clearAllMocks(); + mockClientInstance.get.mockReset(); }); afterEach(async () => { @@ -463,6 +466,22 @@ describe("Valkey Cache Lib", () => { }); describe("cacheClearJobs", () => { + it("does not delete projection keys if activation races legacy cleanup", async () => { + mockClientInstance.sendCommand.mockResolvedValueOnce(["0",["scraper:jobs:ns:bootstrap:index"]]); + mockClientInstance.eval.mockResolvedValueOnce(-1); + await expect(cacheClearJobs()).rejects.toThrow("activated during legacy cache clear"); + expect(mockClientInstance.del).not.toHaveBeenCalled(); + }); + + it("versions search cache without removing active catalog projections", async () => { + mockClientInstance.get.mockResolvedValueOnce("v2-test"); + mockClientInstance.incr.mockResolvedValueOnce(2); + await expect(cacheClearJobs()).resolves.toEqual({deleted:0,patterns:["jobs:search:generation"]}); + expect(mockClientInstance.incr).toHaveBeenCalledWith("jobs:search:generation"); + expect(mockClientInstance.del).not.toHaveBeenCalled(); + expect(mockClientInstance.sendCommand).not.toHaveBeenCalled(); + }); + it("deve remover payloads e índices de vagas por SCAN em lotes", async () => { mockClientInstance.sendCommand .mockResolvedValueOnce(["0", ["scraper:job:1", "scraper:job:2"]]) @@ -470,7 +489,7 @@ describe("Valkey Cache Lib", () => { "0", ["scraper:jobs:index", "scraper:jobs:keyword:node"], ]); - mockClientInstance.del.mockResolvedValueOnce(2).mockResolvedValueOnce(2); + mockClientInstance.eval.mockResolvedValueOnce(2).mockResolvedValueOnce(2); const result = await cacheClearJobs(); @@ -490,14 +509,8 @@ describe("Valkey Cache Lib", () => { "COUNT", "500", ]); - expect(mockClientInstance.del).toHaveBeenNthCalledWith(1, [ - "scraper:job:1", - "scraper:job:2", - ]); - expect(mockClientInstance.del).toHaveBeenNthCalledWith(2, [ - "scraper:jobs:index", - "scraper:jobs:keyword:node", - ]); + expect(mockClientInstance.eval).toHaveBeenNthCalledWith(1, expect.any(String), {keys:["scraper:jobs:index-version","scraper:job:1","scraper:job:2"],arguments:[]}); + expect(mockClientInstance.eval).toHaveBeenNthCalledWith(2, expect.any(String), {keys:["scraper:jobs:index-version","scraper:jobs:index","scraper:jobs:keyword:node"],arguments:[]}); expect(result).toEqual({ deleted: 4, patterns: ["scraper:job:*", "scraper:jobs:*"], diff --git a/backend/tests/unit/modules/jobs/jobMatch.service.test.ts b/backend/tests/unit/modules/jobs/jobMatch.service.test.ts index 42cb6e5b..3ed3bc28 100644 --- a/backend/tests/unit/modules/jobs/jobMatch.service.test.ts +++ b/backend/tests/unit/modules/jobs/jobMatch.service.test.ts @@ -4,7 +4,8 @@ import { getUserMatchTechnologies, jobNotificationIdentity, scoreJobWithTechnologies, -} from "../../../../src/modules/jobs/jobMatch.service"; + scoreProfessionalJob, +} from "../../../../src/modules/jobs/services/jobMatch.service"; function basePublicUser(overrides: Partial = {}): PublicUser { return { @@ -108,3 +109,41 @@ describe("jobMatch.service", () => { expect(jobNotificationIdentity({ id: "job-3" })).toBe("job-3"); }); }); + +describe("PAV-125 professional match", () => { + const product = {id:"p",title:"Product Manager",description:"Discovery roadmap SQL Jira",classification:{primaryFamily:"product",seniority:"senior"},modality:"remoto",location:"Brasil"}; + const design = {...product,id:"d",title:"Product Designer",description:"UX UI pesquisa Figma design systems acessibilidade HTML CSS",classification:{primaryFamily:"product_design",seniority:"senior"}}; + it.each([product,design])("does not penalize absent development languages ($id)", job => { + const preferences = {family:job.classification.primaryFamily,seniority:"senior",modality:"remoto",location:"Brasil"}; + expect(scoreProfessionalJob(job,[],preferences).matchScore).toBeGreaterThan(80); + expect(scoreProfessionalJob(job,[{name:"Java",years:10}],preferences).matchScore).toBe(scoreProfessionalJob(job,[],preferences).matchScore); + }); + it("uses product skills/tools and experience", () => { + const matched=scoreProfessionalJob(product,[{name:"Jira",years:3},{name:"Discovery",years:2}],{family:"product"}); + expect(matched.matchedTechnologies).toEqual(["Discovery","Jira"]); + expect(matched.matchScore).toBeGreaterThan(85); + expect(matched.matchReasons).toContain("competências e ferramentas de Produto relacionadas"); + }); + it("uses design tools without requiring HTML/CSS", () => { + const technologies=[{name:"Figma",years:4},{name:"UX",years:2}]; + expect(scoreProfessionalJob(design,technologies).matchScore).toBeGreaterThan(70); + expect(scoreProfessionalJob(design,[...technologies].reverse())).toEqual(scoreProfessionalJob(design,technologies)); + }); + it("does not invent a score without profile evidence", () => { + expect(scoreProfessionalJob(product,[])).toBe(product); + expect(scoreProfessionalJob(product,[{name:"Java",years:10}])).toBe(product); + }); + it.each(["backend","frontend","fullstack","mobile","data","devops","platform","qa","security","software","leadership"])("preserves %s's exact existing formula", family => { + const job={...product,classification:{primaryFamily:family},description:"Go React PostgreSQL"}; + const technologies=[{name:"Go",years:3},{name:"React",years:2}]; + expect(scoreProfessionalJob(job,technologies,{family:"product",seniority:"senior"})).toEqual(scoreJobWithTechnologies(job,technologies)); + }); +}); + +it("keeps HTML/CSS auxiliary for Design and canonicalizes duplicate skills", () => { + const job={title:"Product Designer UX Figma HTML CSS",classification:{primaryFamily:"product_design"}}; + expect(scoreProfessionalJob(job,[{name:"HTML",years:20},{name:"CSS",years:20}]).matchScore).toBeLessThan(65); + const skills=[{name:"Figma",years:0.5},{name:"Figma",years:3.5}]; + expect(scoreProfessionalJob(job,skills)).toEqual(scoreProfessionalJob(job,[...skills].reverse())); + expect(Number.isInteger(scoreProfessionalJob(job,skills).matchScore)).toBe(true); +}); diff --git a/backend/tests/unit/modules/jobs/jobProfileMatch.service.test.ts b/backend/tests/unit/modules/jobs/jobProfileMatch.service.test.ts index 2ce436b4..4a329f7a 100644 --- a/backend/tests/unit/modules/jobs/jobProfileMatch.service.test.ts +++ b/backend/tests/unit/modules/jobs/jobProfileMatch.service.test.ts @@ -2,6 +2,7 @@ import { beforeEach, describe, expect, it, vi } from "vitest"; const profileMocks = vi.hoisted(() => ({ getUserById: vi.fn(), + getPreferences: vi.fn(), createHighMatchIfMissing: vi.fn(), logWarn: vi.fn(), })); @@ -9,6 +10,7 @@ const profileMocks = vi.hoisted(() => ({ vi.mock("../../../../src/modules/users/users.service", () => ({ UsersService: class { getUserById = profileMocks.getUserById; + getPreferences = profileMocks.getPreferences; }, })); @@ -35,6 +37,7 @@ describe("JobProfileMatchService", () => { technologyExperiences: [{ name: "Go", years: 2 }], }); profileMocks.createHighMatchIfMissing.mockResolvedValue(undefined); + profileMocks.getPreferences.mockResolvedValue(undefined); }); it("retorna perfil vazio sem consultar usuário quando não há userId", async () => { @@ -52,7 +55,7 @@ describe("JobProfileMatchService", () => { await expect(service.getUserTechnologies("user-1")).resolves.toEqual([]); expect(profileMocks.logWarn).toHaveBeenCalledWith( "Não foi possível carregar perfil para cálculo de match", - expect.objectContaining({ userId: "user-1", error: "database down" }), + expect.objectContaining({ code: "PROFILE_READ_FAILED" }), ); }); @@ -106,9 +109,179 @@ describe("JobProfileMatchService", () => { expect(profileMocks.logWarn).toHaveBeenCalledWith( "Não foi possível registrar notificação de alto match", expect.objectContaining({ - userId: "user-1", - error: "notification store down", + code: "MATCH_NOTIFICATION_FAILED", }), ); }); }); + +it("loads Product preferences once and scores without programming languages", async () => { + vi.clearAllMocks(); + profileMocks.getUserById.mockResolvedValueOnce({ + level: "Sênior", + technologies: [], + technologyExperiences: [], + }); + profileMocks.getPreferences.mockResolvedValueOnce({ + keywords: ["product", "Jira", "roadmap"], + remoteOnly: true, + jobTypes: ["Remoto"], + searchLocation: "Brasil", + }); + const service = new JobProfileMatchService(); + const capture = vi.fn(); + const technologies = await service.getUserTechnologies("candidate", capture); + expect(technologies).toEqual([]); + expect(capture).toHaveBeenCalledWith( + expect.objectContaining({ + seniority: "Sênior", + modality: "remoto", + location: "Brasil", + }), + ); + const [job] = await service.enrich( + "candidate", + [ + { + id: "p", + title: "Product Manager", + description: "Jira roadmap", + modality: "remoto", + location: "Brasil", + classification: { primaryFamily: "product", seniority: "senior" }, + }, + ], + technologies, + { preferences: capture.mock.calls[0][0], notifyHighMatches: false }, + ); + expect(job.matchScore).toBeGreaterThan(80); + expect(profileMocks.getPreferences).toHaveBeenCalledTimes(1); + expect(profileMocks.createHighMatchIfMissing).not.toHaveBeenCalled(); +}); + +describe("UserPreferences semantic mapping", () => { + beforeEach(() => { + vi.resetAllMocks(); + profileMocks.getUserById.mockResolvedValue({ + level: null, + technologies: [], + technologyExperiences: [], + }); + }); + + it.each(["product", "product_design"])( + "does not infer family from free keywords (%s)", + async (family) => { + profileMocks.getPreferences.mockResolvedValue({ + keywords: [family, "Jira", "Figma"], + remoteOnly: false, + jobTypes: [], + searchLocation: null, + }); + const capture = vi.fn(); + const service = new JobProfileMatchService(); + await service.getUserTechnologies("candidate", capture); + const preferences = capture.mock.calls[0][0]; + expect(preferences.family).toBeUndefined(); + expect( + preferences.skills.map((skill: { name: string }) => skill.name), + ).toContain(family); + const [job] = await service.enrich( + "candidate", + [ + { + title: "Product UX Jira Figma", + classification: { primaryFamily: family }, + }, + ], + [], + { preferences, notifyHighMatches: false }, + ); + expect(job.matchReasons).not.toContain("família profissional compatível"); + }, + ); + + it.each(["product", "product_design"])( + "maps jobTypes to modality, never contract (%s)", + async (family) => { + profileMocks.getPreferences.mockResolvedValue({ + keywords: [], + remoteOnly: false, + jobTypes: ["Híbrido", "Presencial"], + searchLocation: "São Paulo, SP", + }); + const capture = vi.fn(); + const service = new JobProfileMatchService(); + await service.getUserTechnologies("candidate", capture); + const preferences = capture.mock.calls[0][0]; + expect(preferences).toMatchObject({ + modalities: ["Híbrido", "Presencial"], + location: "São Paulo, SP", + }); + expect(preferences.modality).toBeUndefined(); + expect(preferences.contract).toBeUndefined(); + const jobs = [ + { + title: "Product UX", + modality: "híbrido", + description: "CLT", + classification: { primaryFamily: family }, + }, + { + title: "Product UX", + modality: "remoto", + description: "CLT", + classification: { primaryFamily: family }, + }, + ]; + const [hybrid, remote] = await service.enrich("candidate", jobs, [], { + preferences, + notifyHighMatches: false, + }); + expect(hybrid.matchReasons).toEqual(["modalidade compatível"]); + expect(remote.matchScore).toBeUndefined(); + }, + ); + + it("keeps remoteOnly separate from contract and rejects legacy hiring types", async () => { + profileMocks.getPreferences.mockResolvedValue({ + keywords: ["CLT", "PJ"], + remoteOnly: true, + jobTypes: ["CLT", "PJ", "full-time", "part-time", "contract"], + searchLocation: null, + }); + const capture = vi.fn(); + await new JobProfileMatchService().getUserTechnologies( + "candidate", + capture, + ); + const preferences = capture.mock.calls[0][0]; + expect(preferences.modality).toBe("remoto"); + expect(preferences.modalities).toEqual([]); + expect(preferences.contract).toBeUndefined(); + expect(preferences.family).toBeUndefined(); + }); + + it.each(["product", "product_design"])( + "scores explicit contract independently of modality (%s)", + async (family) => { + const [job] = await new JobProfileMatchService().enrich( + "candidate", + [ + { + title: "Product UX", + modality: "presencial", + description: "CLT", + classification: { primaryFamily: family }, + }, + ], + [], + { + preferences: { contract: "CLT", modality: "remoto" }, + notifyHighMatches: false, + }, + ); + expect(job.matchReasons).toEqual(["contrato compatível"]); + }, + ); +}); diff --git a/backend/tests/unit/modules/jobs/jobSearch.repository.test.ts b/backend/tests/unit/modules/jobs/jobSearch.repository.test.ts index 7805727a..2023d3ed 100644 --- a/backend/tests/unit/modules/jobs/jobSearch.repository.test.ts +++ b/backend/tests/unit/modules/jobs/jobSearch.repository.test.ts @@ -1,3 +1,5 @@ +vi.mock("../../../../src/modules/jobs/repositories/valkeyJobSearch.adapter", () => ({ openIndexedSearch: vi.fn().mockResolvedValue(null), hasActiveJobIndex: vi.fn().mockResolvedValue(false) })); +vi.mock("../../../../src/modules/jobs/cache/valkeySearchCache.adapter", () => ({ searchPageCache: {search: async (_f: unknown, _p: unknown, _r: unknown, query: () => Promise) => query()} })); import { beforeEach, describe, expect, it, vi } from "vitest"; vi.mock("../../../../src/lib/cache", () => ({ cacheAbsoluteSMembers: vi.fn(), diff --git a/backend/tests/unit/modules/jobs/jobSearchCache.test.ts b/backend/tests/unit/modules/jobs/jobSearchCache.test.ts new file mode 100644 index 00000000..5e0120c6 --- /dev/null +++ b/backend/tests/unit/modules/jobs/jobSearchCache.test.ts @@ -0,0 +1,121 @@ +import { describe, expect, it, vi } from "vitest"; +import { + JobSearchCache, + type SearchCacheStore, +} from "../../../../src/modules/jobs/cache/jobSearchCache"; +import { parseJobSearchQuery } from "../../../../src/modules/jobs/parsers/jobSearchQuery.parser"; +const filters = parseJobSearchQuery({ family: "product" }); +const pagination = { page: 1, limit: 20 }; +const page = { total: 1, jobs: [{ id: "product-1" }] }; +function store() { + return { + generation: vi.fn().mockResolvedValue("1"), + read: vi.fn().mockResolvedValue(null), + writeIfGeneration: vi.fn().mockResolvedValue(true), + } satisfies SearchCacheStore; +} +describe("generation-aware search cache", () => { + it("uses a hit without querying and isolates mutable results", async () => { + const storage = store(); + storage.read.mockResolvedValue(page); + const cache = new JobSearchCache(storage); + const query = vi.fn(); + const result = await cache.search(filters, pagination, null, query); + result.jobs.pop(); + expect(page.jobs).toHaveLength(1); + expect(query).not.toHaveBeenCalled(); + }); + it("publishes misses with a configurable TTL and observed generation", async () => { + const storage = store(); + const query = vi.fn().mockResolvedValue(page); + await new JobSearchCache(storage, 300).search( + filters, + pagination, + null, + query, + ); + expect(storage.writeIfGeneration).toHaveBeenCalledWith( + expect.stringMatching(/^jobs:search:v2:/), + page, + "1", + 300, + ); + expect(query).toHaveBeenCalledOnce(); + }); + it("does not serve an old generation during invalidation", async () => { + const storage = store(); + storage.read.mockResolvedValue(page); + storage.generation.mockResolvedValueOnce("1").mockResolvedValueOnce("2"); + const query = vi.fn().mockResolvedValue({ total: 0, jobs: [] }); + expect( + await new JobSearchCache(storage).search( + filters, + pagination, + null, + query, + ), + ).toEqual({ total: 0, jobs: [] }); + expect(query).toHaveBeenCalledOnce(); + }); + it("bypasses cache while indexes are unavailable", async () => { + const storage = store(); + storage.generation.mockResolvedValue(null); + const query = vi.fn().mockResolvedValue(page); + await new JobSearchCache(storage).search(filters, pagination, null, query); + expect(storage.read).not.toHaveBeenCalled(); + expect(storage.writeIfGeneration).not.toHaveBeenCalled(); + }); + it("combines simultaneous queries and never retains failures", async () => { + const storage = store(); + const cache = new JobSearchCache(storage); + let resolve: (value: typeof page) => void = () => {}; + const query = vi.fn( + () => + new Promise((done) => { + resolve = done; + }), + ); + const first = cache.search(filters, pagination, null, query); + const second = cache.search(filters, pagination, null, query); + await vi.waitFor(() => expect(query).toHaveBeenCalledOnce()); + resolve(page); + expect(await first).toEqual(await second); + const failing = vi + .fn() + .mockRejectedValueOnce(new Error("query failed")) + .mockResolvedValueOnce(page); + await expect( + cache.search(filters, pagination, null, failing), + ).rejects.toThrow("query failed"); + await expect( + cache.search(filters, pagination, null, failing), + ).resolves.toEqual(page); + }); + it("tolerates cache read/write outages without hiding query failures", async () => { + const storage = store(); + storage.read.mockRejectedValue(new Error("cache down")); + storage.writeIfGeneration.mockRejectedValue(new Error("cache down")); + await expect( + new JobSearchCache(storage).search( + filters, + pagination, + null, + async () => page, + ), + ).resolves.toEqual(page); + storage.generation.mockRejectedValue(new Error("cache down")); + await expect( + new JobSearchCache(storage).search( + filters, + pagination, + null, + async () => page, + ), + ).resolves.toEqual(page); + }); + it.each([0, -1, 1.5, 86401, NaN])("rejects invalid TTL %s", (ttl) => { + expect(() => new JobSearchCache(store(), ttl)).toThrow( + "JOB_SEARCH_CACHE_TTL_SECONDS", + ); + }); +}); diff --git a/backend/tests/unit/modules/jobs/jobSearchFingerprint.test.ts b/backend/tests/unit/modules/jobs/jobSearchFingerprint.test.ts new file mode 100644 index 00000000..44c19ca9 --- /dev/null +++ b/backend/tests/unit/modules/jobs/jobSearchFingerprint.test.ts @@ -0,0 +1,81 @@ +import { describe, expect, it } from "vitest"; +import { jobSearchCacheKey } from "../../../../src/modules/jobs/cache/jobSearchFingerprint"; +import { parseJobSearchQuery } from "../../../../src/modules/jobs/parsers/jobSearchQuery.parser"; +const key = ( + query: Record, + pagination = { page: 1, limit: 20 }, + generation = "1", + profile: unknown = null, +) => + jobSearchCacheKey( + parseJobSearchQuery(query as never), + pagination, + generation, + profile, + ); + +describe("PAV-125 deterministic search fingerprint", () => { + it.each([ + { family: "backend,fullstack" }, + { family: "fullstack,backend" }, + { family: ["backend", "fullstack"] }, + { family: ["fullstack,backend", "backend"], familyMode: "any" }, + ])("equivalent normalized families %j", (query) => { + expect(key(query)).toBe( + key({ family: "backend,fullstack", familyMode: "any" }), + ); + }); + it("separates primary and any", () => { + expect(key({ family: "backend", familyMode: "primary" })).not.toBe( + key({ family: "backend" }), + ); + }); + it.each([ + { keywords: "private search" }, + { technology: "go" }, + { company: "acme" }, + { type: "remoto" }, + { level: "senior" }, + { seniority: "senior" }, + { location: "private address" }, + { country: "Brasil" }, + { continent: "Europa" }, + { state: "SP" }, + { city: "São Paulo" }, + { contract: "pj" }, + { matchSort: "asc" }, + { matchSort: "desc" }, + ])("includes existing filter %j", (query) => + expect(key(query)).not.toBe(key({})), + ); + it("includes page, limit, generation and ranking inputs", () => { + const base = key({}); + expect(key({}, { page: 2, limit: 20 })).not.toBe(base); + expect(key({}, { page: 1, limit: 21 })).not.toBe(base); + expect(key({}, undefined, "2")).not.toBe(base); + expect(key({}, undefined, "1", { skills: ["Figma"] })).not.toBe(base); + }); + it("normalizes current aliases without exposing private values", () => { + expect(key({ model: " remoto ", contractType: "PJ", sort: "desc" })).toBe( + key({ type: "remoto", contract: "pj", matchSort: "desc" }), + ); + const result = key( + { + keywords: "private@email.test", + location: "Private address", + company: "Acme", + }, + undefined, + "1", + { userId: "sensitive-user" }, + ); + expect(result).toMatch(/^jobs:search:v2:[a-f0-9]{64}$/); + for (const privateValue of [ + "private@email.test", + "Private address", + "Acme", + "sensitive-user", + ]) + expect(result).not.toContain(privateValue); + }); +}); diff --git a/backend/tests/unit/modules/jobs/searchJobs.service.test.ts b/backend/tests/unit/modules/jobs/searchJobs.service.test.ts index f36a5e41..bf023630 100644 --- a/backend/tests/unit/modules/jobs/searchJobs.service.test.ts +++ b/backend/tests/unit/modules/jobs/searchJobs.service.test.ts @@ -1,3 +1,5 @@ +vi.mock("../../../../src/modules/jobs/repositories/valkeyJobSearch.adapter", () => ({ openIndexedSearch: vi.fn().mockResolvedValue(null), hasActiveJobIndex: vi.fn().mockResolvedValue(false) })); +vi.mock("../../../../src/modules/jobs/cache/valkeySearchCache.adapter", () => ({ searchPageCache: {search: async (_f: unknown, _p: unknown, _r: unknown, query: () => Promise) => query()} })); import { beforeEach, describe, expect, it, vi } from "vitest"; vi.mock("../../../../src/lib/cache", () => ({ @@ -229,7 +231,7 @@ describe("SearchJobsService - repository boundary", () => { const repository = { search: vi.fn().mockResolvedValue({ jobs: [{ id: "a" }], total: 21 }) }; const svc = new SearchJobsService(profileService, repository); const result = await svc.execute({ userId: "u1", query: {} }); - expect(repository.search).toHaveBeenCalledWith(filters, defaultPagination, undefined); + expect(repository.search).toHaveBeenCalledWith(filters, defaultPagination, undefined, {preferences: undefined, technologies: []}, []); expect(result).toMatchObject({ total: 21, page: 1, limit: 10, totalPages: 3, hasNext: true }); expect(mockCacheSearchJobIds).not.toHaveBeenCalled(); }); diff --git a/docker-compose.migrate.yml b/docker-compose.migrate.yml index 837e57af..b79f34cf 100644 --- a/docker-compose.migrate.yml +++ b/docker-compose.migrate.yml @@ -19,6 +19,13 @@ services: - vagas-net restart: "no" + scraper-go: + environment: + DATABASE_URL: postgresql://${POSTGRES_USER}:${POSTGRES_PASSWORD}@postgres:5432/${POSTGRES_DB}?sslmode=disable + depends_on: + migrate: + condition: service_completed_successfully + backend: env_file: - ./.env diff --git a/docker-compose.yml b/docker-compose.yml index 5a5a1a85..b64ac9eb 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -21,6 +21,7 @@ services: - SCRAPER_CLASSIFICATION_BATCH_SIZE=${SCRAPER_CLASSIFICATION_BATCH_SIZE-100} - SCRAPER_PERSIST_BATCH_SIZE=${SCRAPER_PERSIST_BATCH_SIZE-100} - SCRAPER_INDEX_BATCH_SIZE=${SCRAPER_INDEX_BATCH_SIZE-250} + - SCRAPER_CATALOG_LIFETIME=${SCRAPER_CATALOG_LIFETIME-216h} - GOMAXPROCS=${GOMAXPROCS:-2} - GOMEMLIMIT=${GOMEMLIMIT:-1500MiB} - GUPY_ENABLED=${GUPY_ENABLED:-true} @@ -59,6 +60,7 @@ services: - ./.env environment: - PORT=3001 + - JOB_SEARCH_CACHE_TTL_SECONDS=${JOB_SEARCH_CACHE_TTL_SECONDS-120} - PROMETHEUS_URL=http://prometheus:9090 - SCRAPER_URL=http://scraper-go:8081 - GO_SCRAPER_URL=http://scraper-go:8081 diff --git a/scraper-go/Dockerfile b/scraper-go/Dockerfile index 86ab8780..3ee8efa8 100644 --- a/scraper-go/Dockerfile +++ b/scraper-go/Dockerfile @@ -13,6 +13,8 @@ RUN CGO_ENABLED=0 GOOS=linux go build \ -ldflags="-s -w" \ -o /go-scraper ./cmd/server +RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags="-s -w" -o /catalog ./cmd/catalog + # ── Runtime stage ─────────────────────────────────────────── FROM alpine:3.20 @@ -21,6 +23,7 @@ RUN apk --no-cache add ca-certificates wget WORKDIR /app COPY --from=builder /go-scraper /go-scraper +COPY --from=builder /catalog /catalog COPY --from=builder /app/internal/keywords/keywords.json ./internal/keywords/keywords.json COPY --from=builder /app/internal/keywords/generator.json ./internal/keywords/generator.json diff --git a/scraper-go/cmd/catalog/main.go b/scraper-go/cmd/catalog/main.go new file mode 100644 index 00000000..a611aeb9 --- /dev/null +++ b/scraper-go/cmd/catalog/main.go @@ -0,0 +1,137 @@ +// Explicit catalog operations. Reconcile is read-only unless --fix is supplied. +package main + +import ( + "context" + "database/sql" + "encoding/json" + "flag" + "fmt" + "log/slog" + "os" + "os/signal" + "strings" + "syscall" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/catalog" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/catalogops" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/config" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobindex" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/pipeline" + "github.com/redis/go-redis/v9" +) + +func main() { + if err := run(); err != nil { + slog.Error("catalog maintenance failed", "error", err) + os.Exit(1) + } +} +func run() error { + operation := flag.String("operation", "reconcile", "reconcile, rebuild, backfill, reclassify, expire, deactivate, rollback, cleanup") + fix := flag.Bool("fix", false, "explicitly repair reconciliation divergences by rebuilding") + batch := flag.Int("batch-size", 200, "bounded batch size (1..2500)") + version := flag.String("version", "", "inactive namespace to clean") + ids := flag.String("ids", "", "comma-separated IDs to deactivate") + flag.Parse() + if *batch < 1 || *batch > 2500 { + return fmt.Errorf("invalid batch size") + } + cfg, err := config.LoadRuntimeConfig() + if err != nil { + return err + } + ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer cancel() + if strings.TrimSpace(os.Getenv("DATABASE_URL")) == "" { + return fmt.Errorf("DATABASE_URL is required") + } + db, err := sql.Open("postgres", os.Getenv("DATABASE_URL")) + if err != nil { + return err + } + defer db.Close() + db.SetMaxOpenConns(8) + pingCtx, pingCancel := context.WithTimeout(ctx, 30*time.Second) + err = db.PingContext(pingCtx) + pingCancel() + if err != nil { + return err + } + opts, err := redis.ParseURL(os.Getenv("VALKEY_URL")) + if err != nil { + return err + } + rdb := redis.NewClient(opts) + defer rdb.Close() + store := catalog.New(db, cfg.CatalogLifetime) + index := jobindex.New(rdb) + configuredKeywords := keywords.LoadDefaultKeywords() + if raw, e := rdb.Get(ctx, "scraper:keywords").Result(); e == nil { + var configured []string + if json.Unmarshal([]byte(raw), &configured) == nil { + configuredKeywords = keywords.GenerateSearchKeywords(configured) + } + } + ops := catalogops.Maintenance{Store: store, Index: index, BatchSize: *batch, Keywords: configuredKeywords} + switch *operation { + case "reclassify": + n, e := ops.Reclassify(ctx) + if e != nil { + return e + } + fmt.Printf("reclassified=%d\n", n) + case "expire": + n, e := ops.Expire(ctx) + if e != nil { + return e + } + fmt.Printf("expired=%d\n", n) + case "rebuild": + v, e := ops.Rebuild(ctx) + if e != nil { + return e + } + fmt.Println(v) + case "backfill": + n, bad, e := ops.Backfill(ctx) + if e != nil { + return e + } + fmt.Printf("imported=%d invalid=%d\n", n, bad) + case "reconcile": + report, e := ops.Reconcile(ctx, *fix) + if e != nil { + return e + } + raw, _ := json.Marshal(report) + fmt.Println(string(raw)) + case "rollback": + return ops.Rollback(ctx, *version) + case "cleanup": + return ops.Cleanup(ctx, *version) + case "deactivate": + if *ids == "" { + return fmt.Errorf("--ids is required") + } + release, e := store.ProcessingLease(ctx) + if e != nil { + return e + } + defer release() + jobs, e := store.Deactivate(ctx, strings.Split(*ids, ",")) + if e != nil { + return e + } + if _, e = index.Apply(ctx, jobs, func(j domain.Job) []string { return pipeline.IndexKeys(j, nil) }); e != nil { + return e + } + return store.MarkIndexed(ctx, jobs) + default: + return fmt.Errorf("unknown operation") + } + return nil +} diff --git a/scraper-go/cmd/server/admin_handlers.go b/scraper-go/cmd/server/admin_handlers.go index ff6b9c70..e30fa94a 100644 --- a/scraper-go/cmd/server/admin_handlers.go +++ b/scraper-go/cmd/server/admin_handlers.go @@ -4,6 +4,9 @@ import ( "context" "encoding/json" "errors" + "fmt" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "log/slog" "net/http" "strconv" "time" @@ -116,12 +119,42 @@ func handleScraperStatus(scheduler *cronjob.Scheduler) http.HandlerFunc { } } -// handleGetJobs retorna todas as vagas do Valkey via GET /admin/jobs +// handleGetJobs preserves the full administrative listing with SQL streaming. // Para o frontend, o Node.js lê direto do Valkey — esse endpoint é para uso administrativo. func handleGetJobs(js *jobstore.Store) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { limit, _ := strconv.Atoi(r.URL.Query().Get("limit")) + if js.Durable() { + started, first := false, true + err := js.StreamActive(r.Context(), limit, func(total int64) error { + w.Header().Set("Content-Type", "application/json") + started = true + _, err := fmt.Fprintf(w, `{"total":%d,"jobs":[`, total) + return err + }, func(job domain.Job) error { + if !first { + if _, err := w.Write([]byte(",")); err != nil { + return err + } + } + first = false + return json.NewEncoder(w).Encode(job) + }) + if err != nil { + slog.Error("administrative catalog stream failed") + if started { + panic(http.ErrAbortHandler) + } + http.Error(w, "erro ao buscar vagas", http.StatusInternalServerError) + return + } + if _, err := w.Write([]byte("]}")); err != nil { + panic(http.ErrAbortHandler) + } + return + } + jobs, err := js.GetSample(r.Context(), limit) if err != nil { http.Error(w, "erro ao buscar vagas", http.StatusInternalServerError) diff --git a/scraper-go/cmd/server/catalog_stream_test.go b/scraper-go/cmd/server/catalog_stream_test.go new file mode 100644 index 00000000..7492ed7f --- /dev/null +++ b/scraper-go/cmd/server/catalog_stream_test.go @@ -0,0 +1,50 @@ +package main + +import ( + "context" + "encoding/json" + "fmt" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" + "github.com/stretchr/testify/require" + "net/http/httptest" + "testing" +) + +type streamedCatalog struct{ jobstore.Catalog } + +func (streamedCatalog) StreamActive(ctx context.Context, limit int, begin func(int64) error, emit func(domain.Job) error) error { + if err := begin(1005); err != nil { + return err + } + n := 1005 + if limit > 0 && limit < n { + n = limit + } + for i := 0; i < n; i++ { + if err := emit(domain.Job{ID: fmt.Sprint(i), Title: "Backend"}); err != nil { + return err + } + } + return nil +} +func TestAdministrativeCatalogPreservesFullAndLimitedListing(t *testing.T) { + for _, tc := range []struct { + query string + count int + }{{"", 1005}, {"?limit=2", 2}} { + t.Run(tc.query, func(t *testing.T) { + w := httptest.NewRecorder() + r := httptest.NewRequest("GET", "/admin/jobs"+tc.query, nil) + handleGetJobs(jobstore.NewDurable(nil, streamedCatalog{}))(w, r) + require.Equal(t, 200, w.Code) + var result struct { + Total int + Jobs []domain.Job + } + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &result)) + require.Equal(t, 1005, result.Total) + require.Len(t, result.Jobs, tc.count) + }) + } +} diff --git a/scraper-go/cmd/server/handlers.go b/scraper-go/cmd/server/handlers.go index 6364480c..ceb4b020 100644 --- a/scraper-go/cmd/server/handlers.go +++ b/scraper-go/cmd/server/handlers.go @@ -11,6 +11,7 @@ import ( "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cache" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/config" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/pipeline" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" @@ -29,6 +30,7 @@ func handleScrape( rdb *redis.Client, runLock *runlock.Manager, runtimeCfg config.RuntimeConfig, + stores ...*jobstore.Store, ) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { var req domain.ScrapeRequest @@ -50,6 +52,9 @@ func handleScrape( defer cancel() searchConfig := searchConfigFromRuntime(req, runtimeCfg) + if len(stores) > 0 { + searchConfig.Store = stores[0] + } slogScrapeStart("public_endpoint", runtimeCfg.MaxConcurrency, req.MaxConcurrency, searchConfig.MaxConcurrency, len(searchConfig.Keywords), len(adapterList)) start := time.Now() diff --git a/scraper-go/cmd/server/server.go b/scraper-go/cmd/server/server.go index a863722a..637e7ba0 100644 --- a/scraper-go/cmd/server/server.go +++ b/scraper-go/cmd/server/server.go @@ -2,17 +2,22 @@ package main import ( "context" + "database/sql" "errors" "log/slog" "net/http" "os" "os/signal" + "strings" "syscall" "time" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cache" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/catalog" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/catalogops" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/config" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cronjob" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobindex" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" @@ -42,7 +47,58 @@ func run(adapterList []ports.JobSource, runtimeCfg config.RuntimeConfig) { // ── Módulos ── kwStore := keywords.NewStore(c) - jobStore := jobstore.New(rdb) + if strings.TrimSpace(os.Getenv("DATABASE_URL")) == "" { + slog.Error("DATABASE_URL is required for the durable job catalog") + os.Exit(1) + } + db, err := sql.Open("postgres", os.Getenv("DATABASE_URL")) + if err != nil { + slog.Error("catalog PostgreSQL configuration failed", "error", err) + os.Exit(1) + } + db.SetMaxOpenConns(8) + db.SetMaxIdleConns(4) + defer db.Close() + dbCtx, dbCancel := context.WithTimeout(context.Background(), 30*time.Second) + err = db.PingContext(dbCtx) + if err == nil { + var ready bool + err = db.QueryRowContext(dbCtx, "SELECT to_regclass('job_catalog') IS NOT NULL").Scan(&ready) + if err == nil && !ready { + err = errors.New("job_catalog migration must be applied before Processor startup") + } + } + dbCancel() + if err != nil { + slog.Error("catalog PostgreSQL unavailable", "error", err) + os.Exit(1) + } + catalogStore := catalog.New(db, runtimeCfg.CatalogLifetime) + + // Do not silently publish a partial bootstrap over an unmigrated catalog. + readinessCtx, readinessCancel := context.WithTimeout(context.Background(), 30*time.Second) + activeVersion, versionErr := rdb.Get(readinessCtx, jobindex.ActiveKey).Result() + if versionErr != nil && versionErr != redis.Nil { + readinessCancel() + slog.Error("catalog index readiness unavailable") + os.Exit(1) + } + if activeVersion == "" { + persisted, countErr := catalogStore.Count(readinessCtx) + legacy, legacyErr := rdb.SCard(readinessCtx, "scraper:jobs:index").Result() + if countErr != nil || legacyErr != nil { + readinessCancel() + slog.Error("catalog migration readiness unavailable") + os.Exit(1) + } + if persisted > 0 || legacy > 0 { + readinessCancel() + slog.Error("explicit catalog backfill/rebuild is required before Processor startup") + os.Exit(1) + } + } + readinessCancel() + jobStore := jobstore.NewDurable(rdb, catalogStore) runLock, err := runlock.New(runlock.NewValkeyStore(rdb), runlock.Config{ TTL: runtimeCfg.RunLockTTL, RenewInterval: runtimeCfg.RunLockRenewInterval, @@ -62,6 +118,12 @@ func run(adapterList []ports.JobSource, runtimeCfg config.RuntimeConfig) { schedulerCfg.IndexBatchSize = runtimeCfg.IndexBatchSize scheduler := cronjob.New(schedulerCfg, kwStore, jobStore, adapterList, rdb, runLock) + scheduler.BeforeRun = func(ctx context.Context) error { + ops := catalogops.Maintenance{Store: catalogStore, Index: jobindex.New(rdb), BatchSize: runtimeCfg.IndexBatchSize} + _, err := ops.Expire(ctx) + return err + } + scheduler.OnComplete = func(kws []string, scraped, saved int, duration time.Duration) { printSummary(len(adapterList), kws, scraped, duration) } @@ -73,7 +135,7 @@ func run(adapterList []ports.JobSource, runtimeCfg config.RuntimeConfig) { mux := http.NewServeMux() // Públicas - mux.Handle("POST /scrape", handleScrape(adapterList, kwStore, c, rdb, runLock, runtimeCfg)) + mux.Handle("POST /scrape", handleScrape(adapterList, kwStore, c, rdb, runLock, runtimeCfg, jobStore)) mux.Handle("GET /health", handleHealth(c)) mux.Handle("GET /metrics", promhttp.Handler()) mux.Handle("GET /api/keywords", handleGetKeywords(kwStore)) diff --git a/scraper-go/go.mod b/scraper-go/go.mod index 67f86282..41e7c5de 100644 --- a/scraper-go/go.mod +++ b/scraper-go/go.mod @@ -32,6 +32,7 @@ require ( require ( github.com/andybalholm/cascadia v1.3.3 // indirect + github.com/lib/pq v1.10.9 github.com/redis/go-redis/v9 v9.19.0 golang.org/x/net v0.52.0 // indirect golang.org/x/sync v0.20.0 diff --git a/scraper-go/go.sum b/scraper-go/go.sum index ccc580a5..9578b843 100644 --- a/scraper-go/go.sum +++ b/scraper-go/go.sum @@ -30,6 +30,8 @@ github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc= github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw= +github.com/lib/pq v1.10.9 h1:YXG7RB+JIjhP29X+OtkiDnYaXQwpS4JEWq7dtCCRUEw= +github.com/lib/pq v1.10.9/go.mod h1:AlVN5x4E4T544tWzH6hKfbfQvm3HdbOxrmggDNAPY9o= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= diff --git a/scraper-go/internal/catalog/store.go b/scraper-go/internal/catalog/store.go new file mode 100644 index 00000000..5b190be5 --- /dev/null +++ b/scraper-go/internal/catalog/store.go @@ -0,0 +1,360 @@ +// Package catalog implements the Processor's PostgreSQL persistence port. +package catalog + +import ( + "context" + "database/sql" + "database/sql/driver" + "encoding/json" + "fmt" + "sort" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/config" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" + "github.com/lib/pq" +) + +const DefaultLifetime = config.DefaultCatalogLifetime +const MaintenanceLock int64 = 125001 + +type Store struct { + DB *sql.DB + Lifetime time.Duration + Now func() time.Time +} + +func New(db *sql.DB, lifetime time.Duration) *Store { + return &Store{DB: db, Lifetime: lifetime, Now: time.Now} +} + +func (s *Store) SaveBatch(ctx context.Context, jobs []domain.Job) (jobstore.SaveResult, error) { + return s.save(ctx, jobs, false) +} + +// Import preserves IDs and the remaining lifetime observed in legacy Valkey. +// ON CONFLICT DO NOTHING prevents an old backfill from overwriting new collection. +func (s *Store) Import(ctx context.Context, jobs []domain.Job) (jobstore.SaveResult, error) { + return s.save(ctx, jobs, true) +} +func (s *Store) save(ctx context.Context, jobs []domain.Job, importing bool) (jobstore.SaveResult, error) { + result := jobstore.SaveResult{} + if len(jobs) > 2500 { + return result, fmt.Errorf("catalog batch exceeds 2500") + } + if s.Lifetime <= 0 { + return result, fmt.Errorf("catalog lifetime must be positive") + } + byID := map[string]domain.Job{} + for _, job := range jobs { + id := jobstore.StableID(&job) + if importing { + id = job.ID + } + if id == "" || (job.Title == "" && job.URL == "") { + result.Invalid++ + continue + } + job.ID = id + if old, ok := byID[id]; ok { + job = jobstore.MergeStored(old, job) + } + byID[id] = job + } + if len(byID) == 0 { + return result, nil + } + ids := make([]string, 0, len(byID)) + for id := range byID { + ids = append(ids, id) + } + sort.Strings(ids) + tx, err := s.DB.BeginTx(ctx, nil) + if err != nil { + return result, fmt.Errorf("catalog begin: %w", err) + } + defer tx.Rollback() + // Fence absent IDs as well, so concurrent inserts preserve merge semantics. + if _, err = tx.ExecContext(ctx, `SELECT pg_advisory_xact_lock(hashtextextended(id,125)) FROM (SELECT unnest($1::text[]) AS id ORDER BY 1) AS ordered_ids`, pq.Array(ids)); err != nil { + return result, err + } + // Sorted locking avoids deadlocks between overlapping collection batches. + rows, err := tx.QueryContext(ctx, `SELECT id,payload FROM job_catalog WHERE id=ANY($1) ORDER BY id FOR UPDATE`, pq.Array(ids)) + if err != nil { + return result, fmt.Errorf("catalog read batch: %w", err) + } + existing := map[string]domain.Job{} + for rows.Next() { + var id string + var raw []byte + if err = rows.Scan(&id, &raw); err != nil { + rows.Close() + return result, err + } + var j domain.Job + if err = json.Unmarshal(raw, &j); err != nil { + rows.Close() + return result, fmt.Errorf("catalog decode: %w", err) + } + existing[id] = j + } + err = rows.Err() + rows.Close() + if err != nil { + return result, err + } + now := s.Now().UTC() + payloads := make([]json.RawMessage, 0, len(ids)) + expiry := make([]string, 0, len(ids)) + for _, id := range ids { + job := byID[id] + if old, ok := existing[id]; ok && !importing { + job = jobstore.MergeStored(old, job) + } + end := now.Add(s.Lifetime) + if importing { + end = job.CatalogExpiresAt + if end.IsZero() || !end.After(now) { + result.Invalid++ + continue + } + } + raw, e := json.Marshal(job) + if e != nil { + return result, e + } + payloads = append(payloads, raw) + expiry = append(expiry, end.Format(time.RFC3339Nano)) + } + if len(payloads) == 0 { + return jobstore.SaveResult{Invalid: result.Invalid}, nil + } + raw, _ := json.Marshal(payloads) + conflict := `DO UPDATE SET payload=EXCLUDED.payload,last_seen_at=EXCLUDED.last_seen_at,updated_at=EXCLUDED.updated_at,expires_at=EXCLUDED.expires_at,revision=nextval('job_catalog_revision_seq')` + if importing { + conflict = `DO NOTHING` + } + rows, err = tx.QueryContext(ctx, `INSERT INTO job_catalog(id,payload,first_seen_at,last_seen_at,updated_at,expires_at) + SELECT p->>'id',p,$2,$2,$2,e::timestamptz FROM jsonb_array_elements($1::jsonb) WITH ORDINALITY AS t(p,n) JOIN unnest($3::text[]) WITH ORDINALITY AS x(e,n) USING(n) + ON CONFLICT(id) `+conflict+` RETURNING payload,revision,expires_at`, string(raw), now, pq.Array(expiry)) + if err != nil { + return result, fmt.Errorf("catalog upsert: %w", err) + } + persisted, err := decodeRows(rows) + if err != nil { + return result, err + } + if err = tx.Commit(); err != nil { + return jobstore.SaveResult{Invalid: result.Invalid}, fmt.Errorf("catalog commit: %w", err) + } + for _, j := range persisted { + if _, ok := existing[j.ID]; ok { + result.Updated++ + } else { + result.Inserted++ + } + } + result.Persisted = persisted + return result, nil +} +func decodeRows(rows *sql.Rows) ([]domain.Job, error) { + defer rows.Close() + jobs := []domain.Job{} + for rows.Next() { + var raw []byte + var job domain.Job + if err := rows.Scan(&raw, &job.CatalogRevision, &job.CatalogExpiresAt); err != nil { + return nil, err + } + if err := json.Unmarshal(raw, &job); err != nil { + return nil, err + } + jobs = append(jobs, job) + } + return jobs, rows.Err() +} +func (s *Store) GetByIDs(ctx context.Context, ids []string) ([]domain.Job, error) { + rows, err := s.DB.QueryContext(ctx, `SELECT payload,revision,expires_at FROM job_catalog WHERE id=ANY($1) AND expires_at>$2 ORDER BY id`, pq.Array(ids), s.Now().UTC()) + if err != nil { + return nil, err + } + return decodeRows(rows) +} + +// Batch uses a stable keyset cursor and the rebuild's fixed activity instant. +func (s *Store) Batch(ctx context.Context, after string, limit int, at time.Time) ([]domain.Job, error) { + if limit < 1 || limit > 2500 { + return nil, fmt.Errorf("catalog batch limit must be 1..2500") + } + rows, err := s.DB.QueryContext(ctx, `SELECT payload,revision,expires_at FROM job_catalog WHERE id>$1 AND expires_at>$2 ORDER BY id LIMIT $3`, after, at, limit) + if err != nil { + return nil, err + } + return decodeRows(rows) +} +func (s *Store) Sample(ctx context.Context, limit int) ([]domain.Job, error) { + if limit < 1 || limit > 2500 { + return nil, fmt.Errorf("bounded sample limit must be 1..2500; use streaming for full catalog") + } + return s.Batch(ctx, "", limit, s.Now().UTC()) +} +func (s *Store) Count(ctx context.Context) (int64, error) { + var n int64 + err := s.DB.QueryRowContext(ctx, `SELECT count(*) FROM job_catalog WHERE expires_at>$1`, s.Now().UTC()).Scan(&n) + return n, err +} +func (s *Store) MarkIndexed(ctx context.Context, jobs []domain.Job) error { + ids := make([]string, len(jobs)) + revisions := make([]int64, len(jobs)) + for i, j := range jobs { + ids[i] = j.ID + revisions[i] = j.CatalogRevision + } + _, err := s.DB.ExecContext(ctx, `UPDATE job_catalog c SET indexed_revision=GREATEST(indexed_revision,t.r) FROM unnest($1::text[],$2::bigint[]) AS t(id,r) WHERE c.id=t.id AND c.revision=t.r AND c.indexed_revision$2 RETURNING payload,revision,expires_at`, pq.Array(ids), s.Now().UTC()) + if err != nil { + return nil, err + } + jobs, err := decodeRows(rows) + if err != nil { + return nil, err + } + if err = tx.Commit(); err != nil { + return nil, err + } + return jobs, nil +} + +// InactiveBatch preserves historical rows; expiration maintenance only removes +// their Valkey projection. The committed expiry itself is the lifecycle rule. +func (s *Store) InactiveBatch(ctx context.Context, after string, limit int, at time.Time) ([]domain.Job, error) { + if limit < 1 || limit > 2500 { + return nil, fmt.Errorf("catalog batch limit must be 1..2500") + } + rows, err := s.DB.QueryContext(ctx, `SELECT payload,revision,expires_at FROM job_catalog WHERE id>$1 AND expires_at<=$2 ORDER BY id LIMIT $3`, after, at, limit) + if err != nil { + return nil, err + } + return decodeRows(rows) +} + +// ReclassifyBatch updates processed payloads without pretending the job was +// seen in a new collection. Lifetime and collection timestamps are preserved. +func (s *Store) ReclassifyBatch(ctx context.Context, jobs []domain.Job) ([]domain.Job, error) { + if len(jobs) > 2500 { + return nil, fmt.Errorf("catalog batch exceeds 2500") + } + if len(jobs) == 0 { + return nil, nil + } + raw, err := json.Marshal(jobs) + if err != nil { + return nil, err + } + tx, err := s.DB.BeginTx(ctx, nil) + if err != nil { + return nil, err + } + defer tx.Rollback() + rows, err := tx.QueryContext(ctx, `UPDATE job_catalog c SET payload=t.p,updated_at=$2,revision=nextval('job_catalog_revision_seq') FROM jsonb_array_elements($1::jsonb) AS t(p) WHERE c.id=t.p->>'id' RETURNING c.payload,c.revision,c.expires_at`, string(raw), s.Now().UTC()) + if err != nil { + return nil, err + } + persisted, err := decodeRows(rows) + if err != nil { + return nil, err + } + if err = tx.Commit(); err != nil { + return nil, fmt.Errorf("catalog commit: %w", err) + } + return persisted, nil +} + +// StreamActive preserves the existing administrative listing contract. A +// read-only SQL snapshot supplies count and rows; database/sql streams rows +// incrementally instead of building an unbounded []Job in the Processor. +func (s *Store) StreamActive(ctx context.Context, limit int, begin func(int64) error, emit func(domain.Job) error) error { + tx, err := s.DB.BeginTx(ctx, &sql.TxOptions{Isolation: sql.LevelRepeatableRead, ReadOnly: true}) + if err != nil { + return err + } + defer tx.Rollback() + at := s.Now().UTC() + var total int64 + if err = tx.QueryRowContext(ctx, `SELECT count(*) FROM job_catalog WHERE expires_at>$1`, at).Scan(&total); err != nil { + return err + } + query := `SELECT payload,revision,expires_at FROM job_catalog WHERE expires_at>$1 ORDER BY id` + args := []any{at} + if limit > 0 { + query += ` LIMIT $2` + args = append(args, limit) + } + rows, err := tx.QueryContext(ctx, query, args...) + if err != nil { + return err + } + defer rows.Close() + if err = begin(total); err != nil { + return err + } + for rows.Next() { + var raw []byte + var j domain.Job + if err = rows.Scan(&raw, &j.CatalogRevision, &j.CatalogExpiresAt); err != nil { + return err + } + if err = json.Unmarshal(raw, &j); err != nil { + return err + } + if err = emit(j); err != nil { + return err + } + } + if err = rows.Err(); err != nil { + return err + } + return tx.Commit() +} diff --git a/scraper-go/internal/catalogops/maintenance.go b/scraper-go/internal/catalogops/maintenance.go new file mode 100644 index 00000000..2d2f3d07 --- /dev/null +++ b/scraper-go/internal/catalogops/maintenance.go @@ -0,0 +1,651 @@ +// Package catalogops runs explicit, cancellable catalog maintenance; never at startup. +package catalogops + +import ( + "context" + "crypto/rand" + "encoding/hex" + "encoding/json" + "fmt" + "log/slog" + "reflect" + "strings" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/catalog" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/classifier" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobindex" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/pipeline" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/taxonomy" + "github.com/redis/go-redis/v9" +) + +type Reader interface { + Batch(context.Context, string, int, time.Time) ([]domain.Job, error) + GetByIDs(context.Context, []string) ([]domain.Job, error) +} +type Maintenance struct { + Store *catalog.Store + Index *jobindex.Manager + BatchSize int + Keywords []string +} + +func (m *Maintenance) size() int { + if m.BatchSize < 1 { + return 200 + } + return m.BatchSize +} +func (m *Maintenance) keys(j domain.Job) []string { + return pipeline.IndexKeys(j, append(append(append([]string{}, m.Keywords...), j.Keywords...), j.Keyword)) +} + +type Report struct { + Active int `json:"active"` + Missing int `json:"missing"` + Stale int `json:"stale"` + Membership int `json:"membership"` + Primary int `json:"primary"` + Related int `json:"related"` + Any int `json:"any"` + Invalid int `json:"invalid"` + Documents int `json:"documents"` + Counts int `json:"counts"` +} + +func (r Report) Divergent() bool { + return r.Missing+r.Stale+r.Membership+r.Primary+r.Related+r.Any+r.Invalid+r.Documents+r.Counts > 0 +} + +func (m *Maintenance) Rebuild(ctx context.Context) (string, error) { + release, err := m.Store.MaintenanceLease(ctx) + if err != nil { + return "", err + } + defer release() + return m.rebuildLocked(ctx) +} +func (m *Maintenance) rebuildLocked(ctx context.Context) (string, error) { + bytes := make([]byte, 12) + if _, err := rand.Read(bytes); err != nil { + return "", err + } + version := "v2-" + hex.EncodeToString(bytes) + previous, err := m.Index.Active(ctx) + if err != nil { + return "", err + } + at := m.Store.Now().UTC() + cursor := "" + total := 0 + // Even an empty catalog has a validated namespace manifest. + if err = m.Index.RDB.SAdd(ctx, jobindex.Prefix(version)+"keys", jobindex.Prefix(version)+"index", jobindex.Prefix(version)+"expires").Err(); err != nil { + return "", err + } + if err = m.Index.RDB.Expire(ctx, jobindex.Prefix(version)+"keys", m.Store.Lifetime).Err(); err != nil { + return "", err + } + for { + jobs, e := m.Store.Batch(ctx, cursor, m.size(), at) + if e != nil { + return "", e + } + if len(jobs) == 0 { + break + } + if _, e = m.Index.ApplyVersion(ctx, version, jobs, m.keys, true); e != nil { + slog.Error("catalog rebuild invalid batch", "version", version, "batch_size", len(jobs)) + return "", e + } + cursor = jobs[len(jobs)-1].ID + total += len(jobs) + slog.Info("catalog rebuild batch", "version", version, "batch_size", len(jobs), "processed", total) + } + report, err := Compare(ctx, m.Store, m.Index, version, m.size(), m.keys) + if err != nil { + return "", err + } + if report.Divergent() { + return "", fmt.Errorf("rebuild validation failed: %+v", report) + } + familyKeys := []string{} + for _, family := range taxonomy.Families() { + for _, kind := range []string{"family:", "family:primary:", "family:related:"} { + familyKeys = append(familyKeys, kind+family.ID) + } + } + raw, _ := json.Marshal(familyKeys) + _, err = publish.Run(ctx, m.Index.RDB, []string{jobindex.ActiveKey, jobindex.GenerationKey}, previous, version, jobindex.Prefix(version), string(raw)).Result() + if err != nil { + return "", fmt.Errorf("publish catalog namespace: %w", err) + } + + // Record only revisions belonging to the successfully published projection. + cursor = "" + for { + jobs, e := m.Store.Batch(ctx, cursor, m.size(), at) + if e != nil { + return "", e + } + if len(jobs) == 0 { + break + } + if e = m.Store.MarkIndexed(ctx, jobs); e != nil { + return "", fmt.Errorf("rebuild indexed checkpoint: %w", e) + } + cursor = jobs[len(jobs)-1].ID + } + slog.Info("catalog rebuild published", "version", version, "previous_version", previous, "active", report.Active) + return version, nil +} + +var publish = redis.NewScript(` +local previous,version,prefix,families=ARGV[1],ARGV[2],ARGV[3],cjson.decode(ARGV[4]) +local function typed(k,want) local t=redis.call('TYPE',k).ok;if t~='none' and t~=want then error('WRONGTYPE rebuild preflight') end end +typed(KEYS[1],'string');typed(KEYS[2],'string');typed('scraper:jobs:previous-index-version','string') +if (redis.call('GET',KEYS[1]) or 'bootstrap')~=previous then return redis.error_reply('active namespace changed') end +local g=redis.call('GET',KEYS[2]);if g and (not string.match(g,'^%d+$') or tonumber(g)>=9007199254740991) then error('invalid generation') end +typed(prefix..'keys','set');typed(prefix..'index','set');typed(prefix..'expires','zset');typed('scraper:jobs:index','set');typed('scraper:jobs:expires','zset') +for _,k in ipairs(families) do typed(prefix..k,'set');typed('scraper:jobs:'..k,'set') end +for _,k in ipairs(families) do redis.call('SUNIONSTORE','scraper:jobs:'..k,prefix..k) end +redis.call('SUNIONSTORE','scraper:jobs:index',prefix..'index') +redis.call('ZUNIONSTORE','scraper:jobs:expires',1,prefix..'expires') +redis.call('SET','scraper:jobs:previous-index-version',previous) +redis.call('PERSIST',prefix..'keys');redis.call('PERSIST',prefix..'index');redis.call('PERSIST',prefix..'expires'); +redis.call('SET',KEYS[1],version);redis.call('INCR',KEYS[2]);return 1 +`) + +func (m *Maintenance) Reconcile(ctx context.Context, fix bool) (Report, error) { + release, err := m.Store.MaintenanceLease(ctx) + if err != nil { + return Report{}, err + } + defer release() + version, err := m.Index.Active(ctx) + if err != nil { + return Report{}, err + } + report, err := Compare(ctx, m.Store, m.Index, version, m.size(), m.keys) + if err != nil { + return report, err + } + if fix && report.Divergent() { + _, err = m.rebuildLocked(ctx) + } + return report, err +} + +// Compare streams SQL rows and each family Set. Only one batch of payloads/IDs +// is retained. Redis membership probes are pipelined, not a round trip per job. +func Compare(ctx context.Context, source Reader, manager *jobindex.Manager, version string, size int, keys func(domain.Job) []string) (Report, error) { + report := Report{} + prefix := jobindex.Prefix(version) + after := "" + at := manager.Now().UTC() + for { + jobs, err := source.Batch(ctx, after, size, at) + if err != nil { + return report, err + } + if len(jobs) == 0 { + break + } + pipe := manager.RDB.Pipeline() + metas := make([]*redis.StringCmd, len(jobs)) + docs := make([]*redis.StringCmd, len(jobs)) + checks := make([][]*redis.BoolCmd, len(jobs)) + expected := make([]jobindex.Plan, len(jobs)) + for i, j := range jobs { + plan, e := jobindex.Build(j, keys(j)) + if e != nil { + report.Invalid++ + continue + } + expected[i] = plan + report.Active++ + metas[i] = pipe.Get(ctx, prefix+"index-membership:"+j.ID) + docs[i] = pipe.Get(ctx, prefix+"job:"+j.ID) + checks[i] = append(checks[i], pipe.SIsMember(ctx, prefix+"index", j.ID)) + for _, k := range plan.Keys { + checks[i] = append(checks[i], pipe.SIsMember(ctx, prefix+k, j.ID)) + } + } + if _, err = pipe.Exec(ctx); err != nil && err != redis.Nil { + return report, err + } + for i, j := range jobs { + if metas[i] == nil { + continue + } + if metas[i].Err() == redis.Nil { + report.Missing++ + } else if metas[i].Err() != nil { + return report, metas[i].Err() + } else { + var p jobindex.Plan + if json.Unmarshal([]byte(metas[i].Val()), &p) != nil || p.Revision != j.CatalogRevision || p.Expires != j.CatalogExpiresAt.Unix() || p.Taxonomy != taxonomy.Version() || !reflect.DeepEqual(p.Keys, expected[i].Keys) { + report.Membership++ + } + if p.Primary != expected[i].Primary { + report.Primary++ + } + if !reflect.DeepEqual(p.Related, expected[i].Related) { + report.Related++ + } + } + if docs[i].Err() == redis.Nil { + report.Documents++ + } else if docs[i].Err() != nil { + return report, docs[i].Err() + } else { + var actual domain.Job + if json.Unmarshal([]byte(docs[i].Val()), &actual) != nil { + report.Documents++ + } else { + raw, _ := json.Marshal(j) + actualRaw, _ := json.Marshal(actual) + if string(raw) != string(actualRaw) { + report.Documents++ + } + } + } + for n, c := range checks[i] { + if !c.Val() { + if n == 0 { + report.Missing++ + } else { + report.Membership++ + } + } + } + } + after = jobs[len(jobs)-1].ID + } + indexCount, err := manager.RDB.ZCount(ctx, prefix+"expires", fmt.Sprint(at.Unix()+1), "+inf").Result() + if err != nil { + return report, err + } + if indexCount != int64(report.Active) { + report.Counts++ + } + // Includes global Set, so stale records without any family are found too. + sets := []string{"index"} + for _, f := range taxonomy.Families() { + for _, kind := range []string{"family:", "family:primary:", "family:related:"} { + sets = append(sets, kind+f.ID) + } + } + for _, key := range sets { + var cursor uint64 + for { + ids, next, e := manager.RDB.SScan(ctx, prefix+key, cursor, "*", int64(size)).Result() + if e != nil { + return report, e + } + cursor = next + // COUNT is only a hint; cap downstream SQL/Redis work explicitly. + for start := 0; start < len(ids); start += size { + end := min(start+size, len(ids)) + chunk := ids[start:end] + jobs, e := source.GetByIDs(ctx, chunk) + if e != nil { + return report, e + } + found := map[string]domain.Job{} + for _, j := range jobs { + found[j.ID] = j + } + for _, id := range chunk { + j, ok := found[id] + if !ok { + report.Stale++ + continue + } + if key == "index" { + continue + } + plan, e := jobindex.Build(j, keys(j)) + if e != nil { + report.Invalid++ + continue + } + belongs := false + for _, k := range plan.Keys { + if k == key { + belongs = true + } + } + if !belongs { + if strings.HasPrefix(key, "family:primary:") { + report.Primary++ + } else if strings.HasPrefix(key, "family:related:") { + report.Related++ + } else { + report.Any++ + } + } + } + } + if cursor == 0 { + break + } + } + } + var cursor uint64 + for { + registered, next, e := manager.RDB.SScan(ctx, prefix+"keys", cursor, "*", int64(size)).Result() + if e != nil { + return report, e + } + for _, k := range registered { + suffix := strings.TrimPrefix(k, prefix) + if strings.HasPrefix(suffix, "family:") { + f := strings.TrimPrefix(suffix, "family:") + f = strings.TrimPrefix(f, "primary:") + f = strings.TrimPrefix(f, "related:") + if !taxonomy.IsPublic(f) { + report.Invalid++ + } + } + } + cursor = next + if cursor == 0 { + break + } + } + // Detect unknown family keys even if created outside the managed manifest. + cursor = 0 + for { + names, next, e := manager.RDB.Scan(ctx, cursor, prefix+"family:*", int64(size)).Result() + if e != nil { + return report, e + } + for _, name := range names { + family := strings.TrimPrefix(name, prefix+"family:") + family = strings.TrimPrefix(family, "primary:") + family = strings.TrimPrefix(family, "related:") + if !taxonomy.IsPublic(family) { + report.Invalid++ + } + } + cursor = next + if cursor == 0 { + break + } + } + + return report, nil +} + +// Cleanup never uses KEYS and never deletes an active namespace. The previous +// version is retained until an operator explicitly selects it for cleanup. +func (m *Maintenance) Cleanup(ctx context.Context, version string) error { + release, err := m.Store.MaintenanceLease(ctx) + if err != nil { + return err + } + defer release() + if !strings.HasPrefix(version, "v2-") && version != jobindex.Bootstrap { + return fmt.Errorf("invalid cleanup namespace") + } + prefix := jobindex.Prefix(version) + var cursor uint64 + for { + keys, next, err := m.Index.RDB.SScan(ctx, prefix+"keys", cursor, "*", int64(m.size())).Result() + if err != nil { + return err + } + for _, key := range keys { + if !strings.HasPrefix(key, prefix) { + return fmt.Errorf("namespace manifest contains external key") + } + } + if len(keys) > 0 { + args := []any{version} + for _, key := range keys { + args = append(args, key) + } + if _, err = cleanup.Run(ctx, m.Index.RDB, []string{jobindex.ActiveKey}, args...).Result(); err != nil { + return err + } + } + cursor = next + if cursor == 0 { + break + } + } + _, err = cleanup.Run(ctx, m.Index.RDB, []string{jobindex.ActiveKey}, version, prefix+"keys").Result() + return err +} + +var cleanup = redis.NewScript(`if (redis.call('GET',KEYS[1]) or 'bootstrap')==ARGV[1] then return redis.error_reply('cannot clean active namespace') end;for i=2,#ARGV do redis.call('DEL',ARGV[i]) end;return 1`) + +// Backfill uses SCAN of document keys, pipelined GET/PTTL and SQL batch import. +// It does not renew old TTLs or overwrite a row already present in PostgreSQL. +func (m *Maintenance) Backfill(ctx context.Context) (int, int, error) { + release, err := m.Store.MaintenanceLease(ctx) + if err != nil { + return 0, 0, err + } + defer release() + total, invalid := 0, 0 + var cursor uint64 + for { + names, next, e := m.Index.RDB.Scan(ctx, cursor, "scraper:job:*", int64(m.size())).Result() + if e != nil { + return total, invalid, e + } + cursor = next + for start := 0; start < len(names); start += m.size() { + end := min(start+m.size(), len(names)) + chunk := names[start:end] + pipe := m.Index.RDB.Pipeline() + raws := map[string]*redis.StringCmd{} + ttls := map[string]*redis.DurationCmd{} + for _, key := range chunk { + if strings.HasSuffix(key, ":idx") { + continue + } + raws[key] = pipe.Get(ctx, key) + ttls[key] = pipe.PTTL(ctx, key) + } + if _, e = pipe.Exec(ctx); e != nil && e != redis.Nil { + return total, invalid, e + } + jobs := []domain.Job{} + for key, raw := range raws { + var j domain.Job + if raw.Err() != nil || json.Unmarshal([]byte(raw.Val()), &j) != nil || j.ID != strings.TrimPrefix(key, "scraper:job:") || ttls[key].Val() <= 0 { + invalid++ + continue + } + if c := j.Classification; c != nil { + c.PrimaryFamily = taxonomy.Normalize(c.PrimaryFamily) + if c.PrimaryFamily == "" { + invalid++ + continue + } + valid := true + for _, f := range c.RelatedFamilies { + if !taxonomy.IsPublic(taxonomy.Normalize(f)) { + valid = false + } + } + if !valid { + invalid++ + continue + } + c.RelatedFamilies = taxonomy.Related(c.PrimaryFamily, c.RelatedFamilies) + } + j.CatalogExpiresAt = m.Store.Now().Add(ttls[key].Val()) + jobs = append(jobs, j) + } + result, e := m.Store.Import(ctx, jobs) + if e != nil { + return total, invalid, e + } + total += len(result.Persisted) + invalid += result.Invalid + slog.Info("catalog backfill batch", "batch_size", len(jobs), "inserted", len(result.Persisted), "invalid", invalid) + } + if cursor == 0 { + break + } + } + if _, err = m.rebuildLocked(ctx); err != nil { + return total, invalid, err + } + return total, invalid, nil +} + +// Expire applies committed PostgreSQL expirations in bounded batches. It never +// deletes historical SQL rows, and repeated runs do not bump cache generation. +func (m *Maintenance) Expire(ctx context.Context) (int, error) { + release, err := m.Store.ProcessingLease(ctx) + if err != nil { + return 0, err + } + defer release() + after := "" + at := m.Store.Now() + changed := 0 + for { + jobs, e := m.Store.InactiveBatch(ctx, after, m.size(), at) + if e != nil { + return changed, e + } + if len(jobs) == 0 { + break + } + // Ignore archived rows that are already absent from the active projection. + // This also keeps an invalid historical classification from blocking + // collection after an explicitly repaired/rebuilt namespace. + version, e := m.Index.Active(ctx) + if e != nil { + return changed, e + } + pipe := m.Index.RDB.Pipeline() + members := make([]*redis.BoolCmd, len(jobs)) + for i, j := range jobs { + members[i] = pipe.SIsMember(ctx, jobindex.Prefix(version)+"index", j.ID) + } + if _, e = pipe.Exec(ctx); e != nil { + return changed, e + } + indexed := []domain.Job{} + for i, j := range jobs { + if members[i].Val() { + indexed = append(indexed, j) + } + } + n, e := m.Index.Apply(ctx, indexed, m.keys) + if e != nil { + return changed, e + } + + changed += n + after = jobs[len(jobs)-1].ID + } + slog.Info("catalog expiration complete", "changed", changed, "batch_size", m.size()) + return changed, nil +} + +// Reclassify is explicit, independent of external collection, and never renews +// lastSeenAt/expiresAt. Its exclusive fence prevents stale payload overwrites. +func (m *Maintenance) Reclassify(ctx context.Context) (int, error) { + release, err := m.Store.MaintenanceLease(ctx) + if err != nil { + return 0, err + } + defer release() + after := "" + at := m.Store.Now() + changed := 0 + for { + jobs, e := m.Store.Batch(ctx, after, m.size(), at) + if e != nil { + return changed, e + } + if len(jobs) == 0 { + break + } + cursor := jobs[len(jobs)-1].ID + updates := []domain.Job{} + for _, j := range jobs { + classification := classifier.Classify(j) + oldClassification, _ := json.Marshal(j.Classification) + newClassification, _ := json.Marshal(classification) + if string(oldClassification) != string(newClassification) { + j.Classification = &classification + updates = append(updates, j) + } + } + persisted, e := m.Store.ReclassifyBatch(ctx, updates) + if e != nil { + return changed, e + } + byID := map[string]domain.Job{} + for _, j := range persisted { + byID[j.ID] = j + } + for i, j := range jobs { + if update, ok := byID[j.ID]; ok { + jobs[i] = update + } + } + if _, e = m.Index.Apply(ctx, jobs, m.keys); e != nil { + return changed, e + } + if e = m.Store.MarkIndexed(ctx, jobs); e != nil { + return changed, e + } + changed += len(persisted) + after = cursor + slog.Info("catalog reclassification batch", "batch_size", len(jobs), "changed", len(persisted)) + } + return changed, nil +} + +// Rollback only activates an already existing namespace if it still matches +// the PostgreSQL truth. Changed catalog rows require a new rebuild instead. +func (m *Maintenance) Rollback(ctx context.Context, version string) error { + release, err := m.Store.MaintenanceLease(ctx) + if err != nil { + return err + } + defer release() + if version == "" { + version, err = m.Index.RDB.Get(ctx, "scraper:jobs:previous-index-version").Result() + if err != nil { + return err + } + } + if !strings.HasPrefix(version, "v2-") && version != jobindex.Bootstrap { + return fmt.Errorf("invalid rollback namespace") + } + if m.Index.RDB.Exists(ctx, jobindex.Prefix(version)+"keys").Val() == 0 { + return fmt.Errorf("rollback namespace absent") + } + report, err := Compare(ctx, m.Store, m.Index, version, m.size(), m.keys) + if err != nil { + return err + } + if report.Divergent() { + return fmt.Errorf("rollback namespace diverges from PostgreSQL; rebuild required") + } + previous, err := m.Index.Active(ctx) + if err != nil { + return err + } + familyKeys := []string{} + for _, f := range taxonomy.Families() { + for _, kind := range []string{"family:", "family:primary:", "family:related:"} { + familyKeys = append(familyKeys, kind+f.ID) + } + } + raw, _ := json.Marshal(familyKeys) + _, err = publish.Run(ctx, m.Index.RDB, []string{jobindex.ActiveKey, jobindex.GenerationKey}, previous, version, jobindex.Prefix(version), string(raw)).Result() + return err +} diff --git a/scraper-go/internal/catalogops/maintenance_integration_test.go b/scraper-go/internal/catalogops/maintenance_integration_test.go new file mode 100644 index 00000000..76ca91d0 --- /dev/null +++ b/scraper-go/internal/catalogops/maintenance_integration_test.go @@ -0,0 +1,296 @@ +package catalogops + +import ( + "context" + "database/sql" + "encoding/json" + "fmt" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/catalog" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobindex" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" + "github.com/alicebob/miniredis/v2" + "github.com/redis/go-redis/v9" + "github.com/stretchr/testify/require" + "os" + "strings" + "testing" + "time" +) + +func integration(t *testing.T) (*Maintenance, *redis.Client) { + t.Helper() + dsn := os.Getenv("PAV125_TEST_DATABASE_URL") + if dsn == "" { + t.Skip("PAV125_TEST_DATABASE_URL not set (isolated PostgreSQL required)") + } + root, e := sql.Open("postgres", dsn) + require.NoError(t, e) + schema := fmt.Sprintf("pav125_%d", time.Now().UnixNano()) + _, e = root.Exec(`CREATE SCHEMA ` + schema) + require.NoError(t, e) + db, e := sql.Open("postgres", dsn+"&search_path="+schema) + require.NoError(t, e) + migration, e := os.ReadFile("../../../backend/drizzle/0015_job_catalog.sql") + require.NoError(t, e) + _, e = db.Exec(string(migration)) + require.NoError(t, e) + t.Cleanup(func() { db.Close(); root.Exec(`DROP SCHEMA ` + schema + ` CASCADE`); root.Close() }) + mr := miniredis.RunT(t) + r := redis.NewClient(&redis.Options{Addr: mr.Addr()}) + t.Cleanup(func() { r.Close() }) + store := catalog.New(db, catalog.DefaultLifetime) + now := time.Now().UTC().Truncate(time.Second) + store.Now = func() time.Time { return now } + index := jobindex.New(r) + index.Now = func() time.Time { return store.Now() } + return &Maintenance{Store: store, Index: index, BatchSize: 2}, r +} +func identity(id string) string { j := sample(id, "backend"); return jobstore.StableID(&j) } +func sample(id, family string) domain.Job { + return domain.Job{ID: id, Title: "Job " + id, URL: "https://example.test/" + id, Source: "test", Keyword: "sql", Classification: &domain.Classification{PrimaryFamily: family, RelatedFamilies: []string{"platform"}, InScope: true}} +} +func TestCatalogInsertUpsertCycleAndCommit(t *testing.T) { + m, r := integration(t) + ctx := context.Background() + first := m.Store.Now() + result, e := m.Store.SaveBatch(ctx, []domain.Job{sample("a", "product"), sample("b", "product_design"), sample("a", "product")}) + require.NoError(t, e) + require.Len(t, result.Persisted, 2) + require.Equal(t, 2, result.Inserted) + require.Equal(t, int64(0), r.Exists(ctx, "scraper:jobs:index").Val()) + _, e = m.Index.Apply(ctx, result.Persisted, m.keys) + require.NoError(t, e) + require.NoError(t, m.Store.MarkIndexed(ctx, result.Persisted)) + m.Store.Now = func() time.Time { return first.Add(24 * time.Hour) } + updated := sample("a", "product_design") + updated.Source = "second" + result, e = m.Store.SaveBatch(ctx, []domain.Job{updated}) + require.NoError(t, e) + require.Equal(t, 1, result.Updated) + require.WithinDuration(t, first.Add(24*time.Hour).Add(catalog.DefaultLifetime), result.Persisted[0].CatalogExpiresAt, time.Microsecond) + _, e = m.Index.Apply(ctx, result.Persisted, m.keys) + require.NoError(t, e) + require.False(t, r.SIsMember(ctx, "scraper:jobs:family:primary:product", identity("a")).Val()) + require.True(t, r.SIsMember(ctx, "scraper:jobs:family:primary:product_design", identity("a")).Val()) + var count int + require.NoError(t, m.Store.DB.QueryRow("SELECT count(*) FROM job_catalog").Scan(&count)) + require.Equal(t, 2, count) + var seen, last time.Time + require.NoError(t, m.Store.DB.QueryRow("SELECT first_seen_at,last_seen_at FROM job_catalog WHERE id=$1", identity("a")).Scan(&seen, &last)) + require.WithinDuration(t, first, seen, time.Microsecond) + require.WithinDuration(t, m.Store.Now(), last, time.Microsecond) + m.Store.Now = func() time.Time { return first.Add(10 * 24 * time.Hour) } + jobs, e := m.Store.GetByIDs(ctx, []string{identity("a"), identity("b")}) + require.NoError(t, e) + require.Empty(t, jobs) + require.NoError(t, m.Store.DB.QueryRow("SELECT count(*) FROM job_catalog").Scan(&count)) + require.Equal(t, 2, count) + n, e := m.Expire(ctx) + require.NoError(t, e) + require.Equal(t, 2, n) + require.Zero(t, r.SCard(ctx, "scraper:jobs:index").Val()) + again, e := m.Expire(ctx) + require.NoError(t, e) + require.Zero(t, again) + v, e := m.Rebuild(ctx) + require.NoError(t, e) + require.Zero(t, r.SCard(ctx, jobindex.Prefix(v)+"index").Val()) +} +func TestCatalogRollbackAndCommitFailure(t *testing.T) { + m, r := integration(t) + ctx := context.Background() + _, e := m.Store.DB.Exec(`ALTER TABLE job_catalog ADD CONSTRAINT fail_batch CHECK (payload->>'title' <> 'fail')`) + require.NoError(t, e) + bad := sample("bad", "product") + bad.Title = "fail" + result, e := m.Store.SaveBatch(ctx, []domain.Job{sample("ok", "backend"), bad}) + require.Error(t, e) + require.Empty(t, result.Persisted) + n, e := m.Store.Count(ctx) + require.NoError(t, e) + require.Zero(t, n) + require.Zero(t, r.SCard(ctx, "scraper:jobs:index").Val()) + _, e = m.Store.DB.Exec(`CREATE FUNCTION fail_commit() RETURNS trigger AS $$ BEGIN IF NEW.payload->>'title'='commitfail' THEN RAISE EXCEPTION 'forced deferred commit failure'; END IF; RETURN NEW; END; $$ LANGUAGE plpgsql; CREATE CONSTRAINT TRIGGER fail_commit AFTER INSERT ON job_catalog DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION fail_commit();`) + require.NoError(t, e) + bad.Title = "commitfail" + result, e = m.Store.SaveBatch(ctx, []domain.Job{sample("ok", "backend"), bad}) + require.ErrorContains(t, e, "catalog commit") + require.Empty(t, result.Persisted) + n, e = m.Store.Count(ctx) + require.NoError(t, e) + require.Zero(t, n) + require.Zero(t, r.SCard(ctx, "scraper:jobs:index").Val()) + canceled, cancel := context.WithCancel(ctx) + cancel() + _, e = m.Store.SaveBatch(canceled, []domain.Job{sample("cancel", "backend")}) + require.Error(t, e) +} +func TestRebuildReconcileAndControlledCleanup(t *testing.T) { + m, r := integration(t) + ctx := context.Background() + _, e := m.Store.SaveBatch(ctx, []domain.Job{sample("a", "backend"), sample("b", "product"), sample("c", "product_design")}) + require.NoError(t, e) + r.Set(ctx, "session:external", "keep", 0) + v, e := m.Rebuild(ctx) + require.NoError(t, e) + require.True(t, strings.HasPrefix(v, "v2-")) + require.Equal(t, v, r.Get(ctx, jobindex.ActiveKey).Val()) + before := r.Get(ctx, jobindex.GenerationKey).Val() + report, e := m.Reconcile(ctx, false) + require.NoError(t, e) + require.False(t, report.Divergent(), fmt.Sprint(report)) + require.Equal(t, 3, report.Active) + var pending int + require.NoError(t, m.Store.DB.QueryRow("SELECT count(*) FROM job_catalog WHERE indexed_revision 2500 { + return 0, fmt.Errorf("index batch exceeds 2500") + } + if len(jobs) == 0 { + return 0, nil + } + now := m.Now().Unix() + plans := make([]Plan, 0, len(jobs)) + seen := map[string]bool{} + for _, j := range jobs { + if seen[j.ID] { + return 0, fmt.Errorf("duplicate ID in index batch") + } + seen[j.ID] = true + p, e := Build(j, keys(j)) + if e != nil { + return 0, e + } + if p.Expires <= now { + p.Keys = []string{} + p.Related = []string{} + } + meta := p + meta.Document = "" + encoded, e := json.Marshal(meta) + if e != nil { + return 0, e + } + p.Metadata = string(encoded) + plans = append(plans, p) + } + raw, err := json.Marshal(plans) + if err != nil { + return 0, err + } + result, err := applyScript.Run(ctx, m.RDB, []string{ActiveKey, GenerationKey}, version, Prefix(version), string(raw), now, draft).Int() + if err != nil { + return 0, fmt.Errorf("atomic index batch: %w", err) + } + return result, nil +} + +// Preflight validates every key type before any mutation: Redis Lua errors do +// not roll back previous writes. All removals are SREM of this stable job ID. +// Revision fencing makes retries and out-of-order reclassification idempotent. +var applyScript = redis.NewScript(` +local version,prefix,plans,now,draft=ARGV[1],ARGV[2],cjson.decode(ARGV[3]),tonumber(ARGV[4]),ARGV[5]=='1' +local function typed(k,want) + local t=redis.call('TYPE',k).ok + if t~='none' and t~=want then error('WRONGTYPE controlled index preflight') end +end +typed(KEYS[1],'string');typed(KEYS[2],'string') +local g=redis.call('GET',KEYS[2]);if g and (not string.match(g,'^%d+$') or tonumber(g)>=9007199254740991) then error('invalid generation') end +local active=redis.call('GET',KEYS[1]) or 'bootstrap' +if not draft and active~=version then return redis.error_reply('index version changed') end +typed(prefix..'index','set');typed(prefix..'expires','zset');typed(prefix..'keys','set') +if not draft then typed('scraper:jobs:index','set');typed('scraper:jobs:expires','zset') end +local old={} +for i,p in ipairs(plans) do + local meta=prefix..'index-membership:'..p.id + typed(meta,'string');typed(prefix..'job:'..p.id,'string') + local raw=redis.call('GET',meta) + old[i]=raw and cjson.decode(raw) or {revision=0,keys={}} + if type(old[i])~='table' or not tonumber(old[i].revision) or tonumber(old[i].revision)<0 or type(old[i].keys)~='table' then error('invalid index membership') end + for k,v in pairs(old[i].keys) do if type(k)~='number' or type(v)~='string' or k<1 or k>#old[i].keys then error('invalid index membership keys') end end + if not draft and not raw then + typed('scraper:job:'..p.id..':idx','set') + local legacy=redis.call('SMEMBERS','scraper:job:'..p.id..':idx') + for _,key in ipairs(legacy) do + if string.sub(key,1,13)=='scraper:jobs:' then table.insert(old[i].keys,string.sub(key,14)) end + end + end + for _,k in ipairs(old[i].keys) do typed(prefix..k,'set');if not draft then typed('scraper:jobs:'..k,'set') end end + for _,k in ipairs(p.keys) do typed(prefix..k,'set');if not draft then typed('scraper:jobs:'..k,'set') end end + if not draft then typed('scraper:job:'..p.id,'string');typed('scraper:jobs:index-membership:'..p.id,'string');typed('scraper:job:'..p.id..':idx','set') end +end +local changed=0 +for i,p in ipairs(plans) do + if tonumber(old[i].revision)0 or redis.call('SISMEMBER',prefix..'index',p.id)==1)) then + for _,k in ipairs(old[i].keys) do + redis.call('SREM',prefix..k,p.id) + if not draft then redis.call('SREM','scraper:jobs:'..k,p.id) end + end + local alive=p.expiresAt>now + if alive then + for _,k in ipairs(p.keys) do + redis.call('SADD',prefix..k,p.id);redis.call('SADD',prefix..'keys',prefix..k) + if not draft then redis.call('SADD','scraper:jobs:'..k,p.id) end + end + redis.call('SADD',prefix..'index',p.id);redis.call('ZADD',prefix..'expires',p.expiresAt,p.id) + redis.call('SET',prefix..'job:'..p.id,p.document,'EX',math.max(1,p.expiresAt-now)) + else + p.keys={};p.relatedFamilies={} + redis.call('SREM',prefix..'index',p.id);redis.call('ZREM',prefix..'expires',p.id);redis.call('DEL',prefix..'job:'..p.id) + end + local metadata=p.metadata + redis.call('SET',prefix..'index-membership:'..p.id,metadata) + redis.call('SADD',prefix..'keys',prefix..'index-membership:'..p.id,prefix..'job:'..p.id,prefix..'index',prefix..'expires') + if not draft then + if alive then + redis.call('SADD','scraper:jobs:index',p.id);redis.call('ZADD','scraper:jobs:expires',p.expiresAt,p.id) + redis.call('SET','scraper:job:'..p.id,redis.call('GET',prefix..'job:'..p.id),'EX',math.max(1,p.expiresAt-now)) + else redis.call('SREM','scraper:jobs:index',p.id);redis.call('ZREM','scraper:jobs:expires',p.id);redis.call('DEL','scraper:job:'..p.id) end + redis.call('SET','scraper:jobs:index-membership:'..p.id,metadata) + redis.call('DEL','scraper:job:'..p.id..':idx') + for _,k in ipairs(p.keys) do redis.call('SADD','scraper:job:'..p.id..':idx','scraper:jobs:'..k) end + end + if draft then + local ttl=math.max(1,p.expiresAt-now) + local function extend(k) local oldttl=redis.call('TTL',k);if oldttl0 then redis.call('SET',KEYS[1],version);redis.call('INCR',KEYS[2]) end +return changed +`) diff --git a/scraper-go/internal/jobindex/index_test.go b/scraper-go/internal/jobindex/index_test.go new file mode 100644 index 00000000..69a15107 --- /dev/null +++ b/scraper-go/internal/jobindex/index_test.go @@ -0,0 +1,207 @@ +package jobindex + +import ( + "context" + "encoding/json" + "fmt" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/taxonomy" + "github.com/alicebob/miniredis/v2" + "github.com/redis/go-redis/v9" + "github.com/stretchr/testify/require" + "os" + "testing" + "time" +) + +func setup(t *testing.T) (*Manager, *redis.Client) { + t.Helper() + r := miniredis.RunT(t) + c := redis.NewClient(&redis.Options{Addr: r.Addr()}) + t.Cleanup(func() { c.Close() }) + m := New(c) + m.Now = func() time.Time { return time.Unix(2000000000, 0) } + return m, c +} +func committed(id, family string, related []string, revision int64) domain.Job { + return domain.Job{ID: id, Title: "A job", Sources: []string{}, Keywords: []string{}, CatalogRevision: revision, CatalogExpiresAt: time.Unix(2000003600, 0), Classification: &domain.Classification{PrimaryFamily: family, RelatedFamilies: related}} +} +func noKeys(domain.Job) []string { return nil } +func TestCanonicalPlans(t *testing.T) { + for _, f := range taxonomy.Families() { + t.Run(f.ID, func(t *testing.T) { + p, e := Build(committed("id", f.ID, []string{f.ID, "platform", "platform"}, 1), nil) + require.NoError(t, e) + require.Contains(t, p.Keys, "family:"+f.ID) + require.Contains(t, p.Keys, "family:primary:"+f.ID) + require.NotContains(t, p.Keys, "family:related:"+f.ID) + require.NotContains(t, p.Keys, "family:other") + }) + } + for _, f := range []string{"finance", "Backend", ""} { + _, e := Build(committed("id", f, nil, 1), nil) + require.Error(t, e) + } + p, e := Build(committed("id", "other", nil, 1), nil) + require.NoError(t, e) + require.Empty(t, p.Keys) + _, e = Build(committed("id", "backend", []string{"finance"}, 1), nil) + require.Error(t, e) + _, e = Build(domain.Job{ID: "uncommitted"}, nil) + require.Error(t, e) +} +func TestReclassificationAtomicIdempotent(t *testing.T) { + m, c := setup(t) + ctx := context.Background() + j := committed("a", "leadership", []string{"backend"}, 1) + other := committed("b", "leadership", nil, 2) + n, e := m.Apply(ctx, []domain.Job{j, other}, noKeys) + require.NoError(t, e) + require.Equal(t, 2, n) + j = committed("a", "backend", []string{"platform", "backend", "platform"}, 3) + _, e = m.Apply(ctx, []domain.Job{j}, noKeys) + require.NoError(t, e) + p := Prefix(Bootstrap) + require.False(t, c.SIsMember(ctx, p+"family:primary:leadership", "a").Val()) + require.True(t, c.SIsMember(ctx, p+"family:primary:leadership", "b").Val()) + require.False(t, c.SIsMember(ctx, p+"family:related:backend", "a").Val()) + require.True(t, c.SIsMember(ctx, p+"family:backend", "a").Val()) + require.True(t, c.SIsMember(ctx, p+"family:related:platform", "a").Val()) + var meta Plan + require.NoError(t, json.Unmarshal([]byte(c.Get(ctx, p+"index-membership:a").Val()), &meta)) + require.Equal(t, []string{"platform"}, meta.Related) + require.Equal(t, taxonomy.Version(), meta.Taxonomy) + before := c.Get(ctx, GenerationKey).Val() + n, e = m.Apply(ctx, []domain.Job{j}, noKeys) + require.NoError(t, e) + require.Zero(t, n) + require.Equal(t, before, c.Get(ctx, GenerationKey).Val()) + _, e = m.Apply(ctx, []domain.Job{committed("a", "frontend", nil, 1)}, noKeys) + require.NoError(t, e) + require.True(t, c.SIsMember(ctx, p+"family:primary:backend", "a").Val()) + var doc domain.Job + require.NoError(t, json.Unmarshal([]byte(c.Get(ctx, p+"job:a").Val()), &doc)) + require.Equal(t, j.ID, doc.ID) +} +func TestProductTransitionsAndIndependence(t *testing.T) { + m, c := setup(t) + ctx := context.Background() + p := Prefix(Bootstrap) + for i, f := range []string{"other", "product", "product_design"} { + _, e := m.Apply(ctx, []domain.Job{committed("a", f, nil, int64(i+1))}, noKeys) + require.NoError(t, e) + } + require.False(t, c.SIsMember(ctx, p+"family:product", "a").Val()) + require.True(t, c.SIsMember(ctx, p+"family:primary:product_design", "a").Val()) + require.Equal(t, int64(0), c.Exists(ctx, p+"family:other").Val()) + for i, f := range []string{"fullstack", "devops", "platform"} { + _, e := m.Apply(ctx, []domain.Job{committed(f, f, nil, int64(i+10))}, noKeys) + require.NoError(t, e) + } + require.False(t, c.SIsMember(ctx, p+"family:backend", "fullstack").Val()) + require.False(t, c.SIsMember(ctx, p+"family:frontend", "fullstack").Val()) + require.False(t, c.SIsMember(ctx, p+"family:devops", "platform").Val()) +} +func TestPreflightFailureDoesNotPartiallyMutate(t *testing.T) { + m, c := setup(t) + ctx := context.Background() + p := Prefix(Bootstrap) + c.Set(ctx, p+"family:primary:product", "wrong type", 0) + _, e := m.Apply(ctx, []domain.Job{committed("valid", "backend", nil, 1), committed("bad", "product", nil, 2)}, noKeys) + require.Error(t, e) + require.Equal(t, int64(0), c.Exists(ctx, p+"family:backend", GenerationKey).Val()) + _, e = m.Apply(ctx, []domain.Job{committed("valid", "backend", nil, 1), committed("bad", "finance", nil, 2)}, noKeys) + require.Error(t, e) + require.Equal(t, int64(0), c.Exists(ctx, p+"family:backend").Val()) + _, e = m.Apply(ctx, []domain.Job{committed("same", "backend", nil, 1), committed("same", "frontend", nil, 2)}, noKeys) + require.Error(t, e) + _, e = m.Apply(ctx, make([]domain.Job, 2501), noKeys) + require.Error(t, e) +} +func TestExpirationAndDraftGeneration(t *testing.T) { + m, c := setup(t) + ctx := context.Background() + j := committed("a", "backend", nil, 1) + _, e := m.ApplyVersion(ctx, "v2-test", []domain.Job{j}, noKeys, true) + require.NoError(t, e) + require.Equal(t, int64(0), c.Exists(ctx, GenerationKey, "scraper:jobs:family:backend").Val()) + require.Positive(t, c.TTL(ctx, Prefix("v2-test")+"family:backend").Val()) + _, e = m.Apply(ctx, []domain.Job{j}, noKeys) + require.NoError(t, e) + m.Now = func() time.Time { return j.CatalogExpiresAt } + _, e = m.Apply(ctx, []domain.Job{j}, noKeys) + require.NoError(t, e) + require.False(t, c.SIsMember(ctx, Prefix(Bootstrap)+"family:backend", "a").Val()) + require.False(t, c.SIsMember(ctx, "scraper:jobs:index", "a").Val()) + before := c.Get(ctx, GenerationKey).Val() + n, e := m.Apply(ctx, []domain.Job{j}, noKeys) + require.NoError(t, e) + require.Zero(t, n) + require.Equal(t, before, c.Get(ctx, GenerationKey).Val()) + ctx, cancel := context.WithCancel(ctx) + cancel() + _, e = m.Apply(ctx, []domain.Job{j}, noKeys) + require.Error(t, e) +} + +func TestMalformedMembershipAndGenerationCannotCausePartialWrites(t *testing.T) { + for _, corruption := range []string{"membership", "generation"} { + t.Run(corruption, func(t *testing.T) { + m, c := setup(t) + ctx := context.Background() + if corruption == "membership" { + c.Set(ctx, Prefix(Bootstrap)+"index-membership:bad", `{"revision":"invalid","keys":[]}`, 0) + } else { + c.Set(ctx, GenerationKey, "1.5", 0) + } + _, e := m.Apply(ctx, []domain.Job{committed("first", "backend", nil, 1), committed("bad", "product", nil, 2)}, noKeys) + require.Error(t, e) + require.Zero(t, c.Exists(ctx, Prefix(Bootstrap)+"family:backend").Val()) + }) + } +} + +func TestRealValkeyProjection(t *testing.T) { + url := os.Getenv("PAV125_TEST_VALKEY_URL") + if url == "" { + t.Skip("isolated real Valkey URL not set") + } + opts, e := redis.ParseURL(url) + require.NoError(t, e) + client := redis.NewClient(opts) + defer client.Close() + ctx := context.Background() + m := New(client) + version := fmt.Sprintf("v2-real-%d", time.Now().UnixNano()) + prefix := Prefix(version) + defer func() { + var cursor uint64 + for { + names, next, err := client.Scan(ctx, cursor, prefix+"*", 100).Result() + require.NoError(t, err) + if len(names) > 0 { + client.Del(ctx, names...) + } + cursor = next + if cursor == 0 { + break + } + } + }() + j := committed("a", "product", nil, 1) + j.CatalogExpiresAt = time.Now().Add(time.Hour) + j.Sources = []string{} + j.Keywords = []string{} + _, e = m.ApplyVersion(ctx, version, []domain.Job{j}, noKeys, true) + require.NoError(t, e) + var actual domain.Job + require.NoError(t, json.Unmarshal([]byte(client.Get(ctx, prefix+"job:a").Val()), &actual)) + require.Equal(t, []string{}, actual.Sources) + require.Equal(t, []string{}, actual.Keywords) + j.CatalogRevision = 2 + j.Classification.PrimaryFamily = "product_design" + _, e = m.ApplyVersion(ctx, version, []domain.Job{j}, noKeys, true) + require.NoError(t, e) + require.False(t, client.SIsMember(ctx, prefix+"family:product", "a").Val()) + require.True(t, client.SIsMember(ctx, prefix+"family:primary:product_design", "a").Val()) +} diff --git a/scraper-go/internal/jobstore/jobstore.go b/scraper-go/internal/jobstore/jobstore.go index 534e3df0..ee507ec5 100644 --- a/scraper-go/internal/jobstore/jobstore.go +++ b/scraper-go/internal/jobstore/jobstore.go @@ -13,21 +13,24 @@ import ( "unicode" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/config" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/lib/pq" "github.com/redis/go-redis/v9" "golang.org/x/text/transform" "golang.org/x/text/unicode/norm" ) const ( - jobTTL = 9 * 24 * time.Hour + jobTTL = config.DefaultCatalogLifetime indexKey = "scraper:jobs:index" jobKeyPrefix = "scraper:job:" maxTransientAttempts = 3 ) type Store struct { - rdb *redis.Client + rdb *redis.Client + catalog Catalog } func New(rdb *redis.Client) *Store { @@ -45,8 +48,13 @@ type SaveResult struct { Persisted []domain.Job } -// SaveBatch persists a limited job batch with one Valkey transaction. +// SaveBatch returns only committed durable rows; the legacy adapter remains for migration/tests. func (s *Store) SaveBatch(ctx context.Context, jobs []domain.Job) (SaveResult, error) { + if s.catalog != nil { + var result SaveResult + err := retryTransient(ctx, func() error { var err error; result, err = s.catalog.SaveBatch(ctx, jobs); return err }) + return result, err + } var result SaveResult if len(jobs) == 0 { return result, nil @@ -255,6 +263,10 @@ func isTransient(err error) bool { if errors.Is(err, redis.Nil) { return false } + var pgError *pq.Error + if errors.As(err, &pgError) { + return pgError.Code == "40001" || pgError.Code == "40P01" || strings.HasPrefix(string(pgError.Code), "08") + } message := strings.ToLower(err.Error()) for _, token := range []string{ "invalid", @@ -273,6 +285,9 @@ func isTransient(err error) bool { } func (s *Store) GetByIDs(ctx context.Context, ids []string) ([]domain.Job, error) { + if s.catalog != nil { + return s.catalog.GetByIDs(ctx, ids) + } if cause := context.Cause(ctx); cause != nil { return nil, cause } @@ -308,6 +323,9 @@ func (s *Store) GetByIDs(ctx context.Context, ids []string) ([]domain.Job, error } func (s *Store) GetAll(ctx context.Context) ([]domain.Job, error) { + if s.catalog != nil { + return nil, fmt.Errorf("durable catalog requires streaming; use StreamActive") + } ids, err := s.rdb.SMembers(ctx, indexKey).Result() if err != nil { return nil, fmt.Errorf("jobstore.GetAll: SMembers: %w", err) @@ -342,6 +360,9 @@ func (s *Store) GetAll(ctx context.Context) ([]domain.Job, error) { } func (s *Store) GetSample(ctx context.Context, limit int) ([]domain.Job, error) { + if s.catalog != nil { + return s.catalog.Sample(ctx, limit) + } if limit <= 0 { return s.GetAll(ctx) } @@ -397,6 +418,9 @@ func (s *Store) GetSample(ctx context.Context, limit int) ([]domain.Job, error) } func (s *Store) Count(ctx context.Context) (int64, error) { + if s.catalog != nil { + return s.catalog.Count(ctx) + } n, err := s.rdb.SCard(ctx, indexKey).Result() if err != nil { return 0, fmt.Errorf("jobstore.Count: %w", err) @@ -459,3 +483,38 @@ func normalizeURL(raw string) string { u.Fragment = "" return strings.TrimRight(u.String(), "/") } + +// Catalog is the durable persistence port consumed by the Processor. +type Catalog interface { + SaveBatch(context.Context, []domain.Job) (SaveResult, error) + GetByIDs(context.Context, []string) ([]domain.Job, error) + Sample(context.Context, int) ([]domain.Job, error) + Count(context.Context) (int64, error) + MarkIndexed(context.Context, []domain.Job) error + ProcessingLease(context.Context) (func(), error) + StreamActive(context.Context, int, func(int64) error, func(domain.Job) error) error +} + +func NewDurable(rdb *redis.Client, catalog Catalog) *Store { + if catalog == nil { + panic("durable job catalog is required") + } + return &Store{rdb: rdb, catalog: catalog} +} +func (s *Store) Durable() bool { return s != nil && s.catalog != nil } +func (s *Store) MarkIndexed(ctx context.Context, jobs []domain.Job) error { + return s.catalog.MarkIndexed(ctx, jobs) +} +func (s *Store) ProcessingLease(ctx context.Context) (func(), error) { + return s.catalog.ProcessingLease(ctx) +} + +// MergeStored preserves the existing merge and stable-ID contract for SQL upserts. +func MergeStored(existing, incoming domain.Job) domain.Job { return mergeStored(existing, incoming) } + +func (s *Store) StreamActive(ctx context.Context, limit int, begin func(int64) error, emit func(domain.Job) error) error { + if !s.Durable() { + return fmt.Errorf("streaming requires durable catalog") + } + return s.catalog.StreamActive(ctx, limit, begin, emit) +} diff --git a/scraper-go/internal/pipeline/catalog_integration_test.go b/scraper-go/internal/pipeline/catalog_integration_test.go new file mode 100644 index 00000000..6ffd5c2a --- /dev/null +++ b/scraper-go/internal/pipeline/catalog_integration_test.go @@ -0,0 +1,63 @@ +package pipeline + +import ( + "context" + "database/sql" + "fmt" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/catalog" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobindex" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" + "github.com/alicebob/miniredis/v2" + "github.com/redis/go-redis/v9" + "github.com/stretchr/testify/require" + "os" + "testing" + "time" +) + +func TestDurablePipelineOnlyIndexesAfterConfirmedCommit(t *testing.T) { + dsn := os.Getenv("PAV125_TEST_DATABASE_URL") + if dsn == "" { + t.Skip("isolated PostgreSQL required") + } + ctx := context.Background() + root, e := sql.Open("postgres", dsn) + require.NoError(t, e) + schema := fmt.Sprintf("pav125_pipeline_%d", time.Now().UnixNano()) + _, e = root.Exec("CREATE SCHEMA " + schema) + require.NoError(t, e) + db, e := sql.Open("postgres", dsn+"&search_path="+schema) + require.NoError(t, e) + t.Cleanup(func() { db.Close(); root.Exec("DROP SCHEMA " + schema + " CASCADE"); root.Close() }) + raw, e := os.ReadFile("../../../backend/drizzle/0015_job_catalog.sql") + require.NoError(t, e) + _, e = db.Exec(string(raw)) + require.NoError(t, e) + _, e = db.Exec(`CREATE FUNCTION fail_commit() RETURNS trigger AS $$ BEGIN RAISE EXCEPTION 'forced deferred failure'; END; $$ LANGUAGE plpgsql; CREATE CONSTRAINT TRIGGER fail_commit AFTER INSERT ON job_catalog DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION fail_commit();`) + require.NoError(t, e) + mr := miniredis.RunT(t) + r := redis.NewClient(&redis.Options{Addr: mr.Addr()}) + defer r.Close() + source := catalog.New(db, catalog.DefaultLifetime) + cfg := processConfig{Store: jobstore.NewDurable(r, source), RDB: r, ClassificationBatchSize: 2, PersistBatchSize: 2, IndexBatchSize: 2} + _, stats, e := processIncomingJobs(ctx, feedJobs(inScopeJob(1), inScopeJob(2)), cfg) + require.Error(t, e) + require.Zero(t, stats.Indexed) + require.Equal(t, int64(0), r.Exists(ctx, jobindex.GenerationKey, jobindex.ActiveKey, "scraper:jobs:index").Val()) + n, e := source.Count(ctx) + require.NoError(t, e) + require.Zero(t, n) + _, e = db.Exec("DROP TRIGGER fail_commit ON job_catalog") + require.NoError(t, e) + jobs, stats, e := processIncomingJobs(ctx, feedJobs(inScopeJob(1), inScopeJob(2)), cfg) + require.NoError(t, e) + require.Len(t, jobs, 2) + require.Equal(t, 2, stats.Indexed) + require.Equal(t, 2, stats.Saved()) + for _, j := range jobs { + var revision, indexed int64 + require.NoError(t, db.QueryRow("SELECT revision,indexed_revision FROM job_catalog WHERE id=$1", j.ID).Scan(&revision, &indexed)) + require.Equal(t, revision, indexed) + require.True(t, r.SIsMember(ctx, "scraper:jobs:family:primary:backend", j.ID).Val()) + } +} diff --git a/scraper-go/internal/pipeline/index.go b/scraper-go/internal/pipeline/index.go index f5f3da04..618e4498 100644 --- a/scraper-go/internal/pipeline/index.go +++ b/scraper-go/internal/pipeline/index.go @@ -4,7 +4,6 @@ import ( "context" "fmt" "strings" - "time" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/config" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" @@ -14,7 +13,7 @@ import ( const ( globalIndexKey = "scraper:jobs:index" - indexTTL = 9 * 24 * time.Hour + indexTTL = config.DefaultCatalogLifetime indexNextSuffix = ":next" publishChunk = 100 ) @@ -490,3 +489,6 @@ func reindexPersistedJobs( _, err = indexJobsInValkeyBatched(ctx, rdb, jobs, keywords, batchSize, session) return err } + +// IndexKeys is the existing search-index projection reused by maintenance. +func IndexKeys(job domain.Job, keywords []string) []string { return invertedIndexKeys(job, keywords) } diff --git a/scraper-go/internal/pipeline/pipeline.go b/scraper-go/internal/pipeline/pipeline.go index 73d1a797..a7e31bc2 100644 --- a/scraper-go/internal/pipeline/pipeline.go +++ b/scraper-go/internal/pipeline/pipeline.go @@ -410,26 +410,12 @@ func classificationIndexKeys(job domain.Job) []string { values := make([]string, 0, 1+len(classification.RelatedFamilies)+len(classification.Technologies)) - if indexedClassificationFamily(classification.PrimaryFamily) { - normalized := normalizeIndexValue(classification.PrimaryFamily) - if normalized != "" { - values = append(values, - fmt.Sprintf("scraper:jobs:family:%s", normalized), - fmt.Sprintf("scraper:jobs:keyword:%s", normalized), - ) - } + primary := classification.PrimaryFamily + if taxonomy.IsPublic(primary) { + values = append(values, "scraper:jobs:family:"+primary, "scraper:jobs:family:primary:"+primary, "scraper:jobs:keyword:"+normalizeIndexValue(primary)) } - for _, family := range classification.RelatedFamilies { - if !indexedClassificationFamily(family) { - continue - } - normalized := normalizeIndexValue(family) - if normalized != "" { - values = append(values, - fmt.Sprintf("scraper:jobs:family:%s", normalized), - fmt.Sprintf("scraper:jobs:keyword:%s", normalized), - ) - } + for _, family := range taxonomy.Related(primary, classification.RelatedFamilies) { + values = append(values, "scraper:jobs:family:"+family, "scraper:jobs:family:related:"+family, "scraper:jobs:keyword:"+normalizeIndexValue(family)) } for _, technology := range classification.Technologies { normalized := normalizeIndexValue(technology) diff --git a/scraper-go/internal/pipeline/process.go b/scraper-go/internal/pipeline/process.go index 86f70c04..0f15d7a6 100644 --- a/scraper-go/internal/pipeline/process.go +++ b/scraper-go/internal/pipeline/process.go @@ -9,6 +9,7 @@ import ( "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/config" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/dedup" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobindex" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" "github.com/redis/go-redis/v9" ) @@ -69,7 +70,7 @@ func processIncomingJobs( if cfg.IndexBatchSize <= 0 { cfg.IndexBatchSize = config.DefaultIndexBatchSize } - if cfg.Index == nil && cfg.RDB != nil && cfg.IndexSession == nil { + if cfg.Index == nil && cfg.RDB != nil && cfg.IndexSession == nil && !cfg.Store.Durable() { cfg.IndexSession = newIndexSession(cfg.RunID) } @@ -507,6 +508,14 @@ func indexPersistedChunk( commands := 0 if cfg.Index != nil { err = cfg.Index(ctx, jobs) + } else if cfg.Store.Durable() { + err = jobstore.RetryTransient(ctx, func() error { + _, e := jobindex.New(cfg.RDB).Apply(ctx, jobs, func(job domain.Job) []string { return invertedIndexKeys(job, cfg.Keywords) }) + return e + }) + if err == nil { + err = cfg.Store.MarkIndexed(ctx, jobs) + } } else { commands, err = indexJobsInValkeyBatched(ctx, cfg.RDB, jobs, cfg.Keywords, cfg.IndexBatchSize, cfg.IndexSession) } @@ -519,7 +528,7 @@ func indexPersistedChunk( "size", len(jobs), "error", err, ) - if cfg.Store == nil || cfg.RDB == nil || len(ids) == 0 { + if cfg.Store == nil || cfg.RDB == nil || len(ids) == 0 || cfg.Store.Durable() { return err } if reconErr := reindexPersistedJobs(ctx, cfg.Store, cfg.RDB, ids, cfg.Keywords, cfg.IndexBatchSize, cfg.IndexSession); reconErr != nil { diff --git a/scraper-go/internal/pipeline/product_test.go b/scraper-go/internal/pipeline/product_test.go index 6397cd41..083df76f 100644 --- a/scraper-go/internal/pipeline/product_test.go +++ b/scraper-go/internal/pipeline/product_test.go @@ -51,22 +51,20 @@ func TestProductKeywordsRespectDiscoveryModes(t *testing.T) { assert.Equal(t, seed, batch.keywords) } -func TestProductClassificationDoesNotCreateFamilyIndexes(t *testing.T) { +func TestProductClassificationCreatesCanonicalFamilyIndexes(t *testing.T) { for _, title := range []string{"Product Manager", "Product Designer"} { job := domain.Job{Title: title, Description: "SQL"} classification := classifier.Classify(job) require.True(t, classification.InScope) job.Classification = &classification keys := classificationIndexKeys(job) - assert.NotContains(t, keys, "scraper:jobs:family:product") - assert.NotContains(t, keys, "scraper:jobs:family:product_design") - assert.NotContains(t, keys, "scraper:jobs:keyword:product") - assert.NotContains(t, keys, "scraper:jobs:keyword:product design") + assert.Contains(t, keys, "scraper:jobs:family:"+classification.PrimaryFamily) + assert.Contains(t, keys, "scraper:jobs:family:primary:"+classification.PrimaryFamily) assert.Contains(t, invertedIndexKeys(job, []string{title}), "scraper:jobs:keyword:"+normalizeIndexValue(title)) } job := domain.Job{Classification: &domain.Classification{PrimaryFamily: "backend", RelatedFamilies: []string{"product", "product_design"}, InScope: true}} assert.Contains(t, classificationIndexKeys(job), "scraper:jobs:family:backend") - assert.NotContains(t, classificationIndexKeys(job), "scraper:jobs:family:product") + assert.Contains(t, classificationIndexKeys(job), "scraper:jobs:family:related:product") } func TestDiagnosticAndUnknownFamiliesNeverIndexed(t *testing.T) { diff --git a/scraper-go/internal/pipeline/scrape.go b/scraper-go/internal/pipeline/scrape.go index 7237b21d..16322c66 100644 --- a/scraper-go/internal/pipeline/scrape.go +++ b/scraper-go/internal/pipeline/scrape.go @@ -16,6 +16,7 @@ import ( ) type SearchConfig struct { + Store *jobstore.Store `json:"-"` Keywords []string `json:"keywords"` SearchLocation string `json:"searchLocation"` SearchGeoID string `json:"searchGeoId"` @@ -52,6 +53,13 @@ func ScrapeAllSources( rdb *redis.Client, ) ([]domain.Job, ProcessStats, error) { config = normalizeSearchConfig(config) + if config.Store != nil && config.Store.Durable() { + release, err := config.Store.ProcessingLease(ctx) + if err != nil { + return nil, ProcessStats{}, fmt.Errorf("catalog processing lease: %w", err) + } + defer release() + } slog.Info("starting scrape", "keywords", config.Keywords) slog.Info("scraper concurrency budget", "global_limit", config.MaxConcurrency, @@ -102,7 +110,10 @@ func ScrapeAllSources( } if rdb != nil { processCfg.RDB = rdb - processCfg.Store = jobstore.New(rdb) + processCfg.Store = config.Store + if processCfg.Store == nil { + processCfg.Store = jobstore.New(rdb) + } } jobs, stats, err := runWithConcurrency(