diff --git a/.cspell/custom-dictionary-workspace.txt b/.cspell/custom-dictionary-workspace.txt index a2b8baa..4573e2f 100644 --- a/.cspell/custom-dictionary-workspace.txt +++ b/.cspell/custom-dictionary-workspace.txt @@ -1,71 +1,205 @@ +aacpnw +aarin +Aarin +abacusfinancegroup +abacusinsights Abaixo +abap +abbyy +ABBYY +abcellera aberta abertas +abilitypath +abinbev abordagem aborta abra abre abreviações Abrir +Absher +absherconstruction absoluta Absolutas +absolutesecurityintl absoluto abstração abstraído +absurdventures acabar ação +accela +Accela +accelerationpartners +Acceleron +acceleronfusion +accelschools +accenturefederalservices +accesshealthca +accessholdingsmanagementfirm +accesso +accreditedlabs +Accu +accuweather +aceable +Aceable aceita +aceitar aceitas aceitável Aceite acentos +acentria +Acentria Acessibilidade +acessível acesso acessos achar acidental +acilearning aciona +acluinternships +aclunc +acog +acommerce acompanhamento acompanhar acompanhe acontece acontecia +acoplamento acoplar +acornhealth +acquia +Acquia +Acrisure +acrisureinnovation +acryldata +actpowerservices +acuitymd acumulado +acutechgroupinc +adamsclinical adaptado adaptador Adaptadores +adapterutil +addepar +Addepar +Adelphi +adelphigrouplimited +adelphiresearch adequa +adfinternational +adias adicionada adicionais Adicionamos +Aditum +aditumbio +adjustjobs Administrativas administrativo administrativos +admios +Admios +Adswerve +adswerveinc +advancedtechnologyservices +advbox +ADVBOX +advertisingspecialtyinstitute +advocateconstruction +advocatesforchildrenofnewyork +Advogados +adyen +Adyen adzuna Adzuna +Aechelon +aechelontechnology +aegisventures +aegworldwide +aerospike +Aerospike +aestudio +AEVEX +aevexaerospace afetada +affinidi +Affinidi +affinitiv áfrica África +agebold +agecareers +agencywithin agendar +ageoflearninginc +agibank +Agibank +agilesix agoda agregadas agregado agregador agregando +AGRO +Agroindustrial +agrosearch +agrotools +Agrotools +Água Aguarda aguardando aguentar +Ahrefs +ahrefsjobs +aift +AIFT +aiko +Aiko ainda +aircapture +Aircapture +AIRCO +aircompany +airnorth +airsculpt +airtable +Airtable +Airtame +airtamejobs +airtrunk +aisolutions +Aizer +aizerhealth +ajboggs ajuda ajudar Ajusta ajustado ajustar Ajustes +Akido +akidolabs +akko +AKKO +akuity +Akuity AKVGSK +alabamatitleloansinc +alabs +Alamar +alamarbiosciences +alarmcom +albertmackenziellp aleatório alegre +alelo +Alelo além Alemanha alemania @@ -73,54 +207,175 @@ alerta alertas Alertas alertmanager +alertmedia alfanuméricos +algar +Algar alguma algumas Alguns alicebob +alicerce +Alicerce ALIMENTAÇÃO alimentar +Alimentos +alinemainericonsultoria Alinhado alinhados alinhar +ALKU +alkujobs +allbooked +allcareers +allencontrolsystems +alliancedefendingfreedom +alliedmaker +allinc +allintra +Allintra +alloyal +Alloyal +alltrna +Alltrna +allwebleads +aloyoga +alphaalternatives +alphafmcroles +alphagrepsecurities +alphapublicschools +alphasense +alphasensehelsinki +alphasenseindia +alpineinternships Altamente +Altana +altanaai +ALTEN +altentechnologyusa +altera alteração alterar alternativa alternativo +altium +Altium +altoqi +altoslabs +altscore +Aluga +alume +Alume +alumis +Alumis +alumniventures +alun +alura +Alura +alveole +Alveole alvo +alxafrica +Amae +amaehealth +Amanco +amaro +AMARO ambas +ambercharterschools ambiente ambientes +Ambiq +ambiqmicroinc +amcom +ameelio +Ameelio +amenitiz +Amenitiz América +americanfloodcoalition +americaninstitutesforresearch +amicustherapeutics amigáveis amigável +Amira +amiralearning amostra +amperesand +Amperesand +amperon +Amperon amplas +amplemarket +Amplemarket +ampm +ampsortation +Amtech +amtechsoftware +amwell +Amwell +amylyx +Amylyx +Amyris +amyrisinc Análise +analista analítico +analyticservicesinc +anaplan +Anaplan +anchanto +Anchanto andamento +Anduril +andurilindustries andybalholm anexar Ângela angelacao +Anhembi +animalmedicalcenter +ANINE +aninebing +Annexon +annexonbioscience anos Anos Anotações +anteriad +Anteriad anteriores +Anteris +anteristech antiga antigas antigo antigos +Antunes Anunciar +aoti +AOTI apaga apareça aparece aparecem aparecer aparecerem +apartmentlife +Apera +aperaaiinc +Aperia +aperiasolutions +aperiatechnologies apertado +apexcompanies +apexcompaniescsw +apexit apic +apiiro +Apiiro +apiphani +APIURL aplica aplicada aplicadas @@ -128,211 +383,941 @@ aplicado aplicam aplicar aplicável +apogeetherapeutics +apolloio apontando aponte +Apothe +apothecom +appdirect +appfacilita +appfire +Appfire +appian +Appian +appier +Appier +applovin +applytobambi +applytogreenspark +applytoslabstack +applytosuno +applytowhoosh +appmax +Appmax +appnovation +Appnovation +appodeal +Appodeal +appomni +appspace +Appspace +apptronik +Apptronik aprendiz Apresentação apresentável apresentou +aprix +Aprix +aprovação APROVADA aproveitar aproximação +Aptos +aptoslabs +aputah +aquaticcapitalmanagement aquela +aquia +Aquia +arborenergy +arcadiacareers +arcaea +Arcaea +arcanaanalytics +arcboatcompany +arcee +arceeai +Arcesium +arcesiumllc +archera +Archera +arcinstitute +arcoeducacao +ardentmc áreas +arenaclub +arenaim +Arevon +arevonenergyimpltest argumento +arielinvestments +arizeai +Arkestro +arkestroinc +arkoselabs +arlosolutionsllc +armamentsresearchcompany armazenadas +Armis +armissecurity +armorcode Arquitetura +arquivei +Arquivei +arrayeducation +arspharmaceuticalsoperationsinc +artefactlinkedin artefatos +articlegroup +Arva +arvaintelligence +arvinas +Arvinas +arxroboticsgmbh asar +ASCAS +ascasconsultoria +ascsolutions +asgjobs +Ashfield +ashfieldadvisory +ashfieldmedcomms Ásia +asigovernment +aspectbiosystems +aspirehealthalliance +assemblyai +Assessoria +assetliving +assetwatch +assim Assinaturas assíncrona assíncrono assistente associados +Assy +ASSYST +assystinc +Astera +asteralabs +astoundcommercesandbox +astranis +Astranis +astspacemobile +atalantatherapeutics ataque ataques +atariinc +atbayjobs +atek +ateliware +Ateliware atenção atender +athleticsbaseballops +athleticsbusinessops atingido +atip ativa ativadas ativado ativas ativos +atlantico +Atlântico +atlassand +atlastechnol +atlasxhm +atomai atômica atomicamente atômicas +atomiccartoons +atomicmachines atômico ator atrás +atriumcampus +attainpartners +attainsports +attaintalent +attentionarc +Attivo +attivopartners +Atto +attotrading atuais atualização atualizações atualizada atualizadas atualmente +Atwell +atwellgroup +auctane +Auctane +Audax +audaxgroup +audibene +audibenehearcom +audioeye +augmentcomputing +auldwhitetalentcommunity aumentamos Aumentei +Auros +aurosglobal austrália Austrália autentica autenticada autenticados +authenticbrandsgroup +authenticinsurance +autods +autogenai +autoglass +Autoglass +autom +Autom +automacao +automação +automações automática automaticamente automático automatizadas +automatticcareers +automind +Automind +automox +Automox +autopass +Autopass +autoproff autorização autorizadas +autoscout +autotradercanada +Auvo +auvotecnologia auxiliares auxlib avaliação avaliado +Avanath Avança Avançadas avançado +avantium +Avantium +avantus +Avantus +avebykormancommunities +avel +Ável +aviationinstituteofmaintenance +avidhealth +avidxchangeinc avisa +avisar +avivatec +Avivatec +avomdincdbaavo +avride +Avride +awin +Awin +axiad +Axiad +axicorpfinancialservicesptyltd +axonag +axontalentcommunity +Axsome +axsometherapeutics +axuall +Axuall +Aypa +aypapower +azos +Azos +Azra +azragames +Azul +Azurity +azuritypharmaceuticals +azuritypharmaceuticalsindia +babylist +Babylist +backblaze +Backblaze baixa baixos Baixos +bamboohr +bancopan +bancopine +bancotoyota +bankme +Bankme +banni +Banni +banyancanopygroup +banyaninfrastructure +banyansoftware +barbaricum +Barbaricum +baringa +Baringa +barkleyokrp barrar barras +barrfoundation +Barrieu +barrosadvogados baseado +basejobs baselib básica Básicas +bastante bater +bathworksmichigan +baublebar +bauerhockeycascademaveriklacrosse +Baya +bayada +BAYADA +bayasystems +bbyo +BBYO +bdainc +beaconbiosignals +beamtherapeutics +beamup +Beamup +Beatbox +beatboxbeveragesllc +beautifulai +beelinemedicines Beene +behavox +Behavox +bellcabinetry +bellroy +Bellroy belo +bemobi +Bemobi +benchprep Benefícios benetesla Benevanio +benup beorn +bequestdigital +berkadiatalentpool +berkshiregroupllc +berlinrosen +bernhoeft +Bernhoeft +bertramcapitalmanagement +berylls +Berylls +bestpass +besxar +Besxar +betaonline +betha +Betha +bethesdahealthgroup +betmgm +betterhelp +betterhelpcom +bettinghero +bettygamingca +bettyjobboard +Bevi +bevicareers +Bevlab +bevlabvet +beyondfinance +beyondtrust +Bezos +bgbx +bgbxconsulting +bgeinc +bgeinccampus Biblioteca +bidease +Bidease +billiontoone binário +biohub +Biohub +biolumina +Biolumina +biomechanicsconsultingandresearchllc +biomedrealty +bionexo +Bionexo +Biosignals +Biosystems +Biotherapeutics +Bioworks +bipa +Bipa +birgo +Birgo +bitcoindepot +bitgo +bitmex +bitpanda +Bitpanda +bixtecnologia +blackbirdhealth +blackcanyonconsulting +blackduck +blackedgecapital +blackforestlabs +blacklane +BLACKLANE +blackshark +Blackshark +blankstreet +blastpoint +blenheimchalcot +blenheimchalcotindia +blinkhealth blockdaemon +blockrenovation bloco blocos +bloombergorg +bloomerang +Bloomerang bloqueado Bloqueados bloquear Bloqueia bloqueio +bluecherry +blueconic +bluecrestcapitalmanagement +bluecubeservices +bluefishai +bluehole +Bluehole +bluelabsanalyticsinc +bluemoonmetals +blueprintmedicines +blueroseresearch +bluestarfamilies +Bluevine +bluevineindia +bluevineus +bluewaterthinking +Blume +blumeohaagua +blumira +Blumira +blytheco +Blytheco +bmnt +BMNT +Boggs +boingo +Boingo +boku +Boku +boldmetrics +boloai +bombas +Bombas +bondora +Bondora +bondvet +boomentertainment +Boomi +boomilp +boostedai +boostlingo +Boostlingo +Bosa +bosapropertiesinc bostonconsultinggroup +Boticário +Bottomline +bottomlinetechnologies +bouldercare +boxinc +bpcs +Bracebridge +bracebridgecapital +braeburn +Braeburn +brainlabs +Brainlabs +brainpop +brainstation +Braintrust +braintrusttutors branco +Brandtech +brandtechplus +Brange +brangemedia +brasilparalelo +braskem +Braskem +Braveheart +braveheartbio +breakwatercorp +breezeairways +breezecash +breezyseguros +brevium +Brevium brex +bridgebio +bridgewaterassociatescampusrecruitingreferral +brightai +Brightcore +brightcoreenergy +brightsign +Brightstone +brightstonetherapy +Brilla +brillapubliccharterschools +bringg +Bringg +britishasiantrust +britive +Britive +brivia +Brivia +brkz +BRKZ +brmediagroup +brms +broadsign +Broadsign +broadvoice +Broadvoice +broadwayventures +brookecharterschools +brooklinen +Brooklinen +brph +BRPH BRPOP +brqdigitalsolutions Bruna brunasilva +brunswickgroup +bruntworkwear bruta bruto brutos +bswift +btgpactualchile +btig +BTIG +bubbleskincare +bugcrowd +Bugcrowd +Bugre +buildingdecarbonizationcoalition +buncha +Buncha +bungie +Bungie +buscadas buscam Buscando buscas +businessoffashion +butcherbox +butlr +Butlr +butternutbox +buyersedgeplatformrecruiting +buynomics +Buynomics +buzzsolutions +bvnk +BVNK +bwreferrals +Byborg +byborgenterprises bypasssafe cabeçalhos cabem +cabify +Cabify cacheado cadastra cadastrado Cadastrando cadastrar Cadastro +cadencehealth +cadência +cadrehospice cadvisor +cafortune +cais +CAIS caixa +cakeai +Cala +calahealth calcula calculável +callrail +calyxinstitute +calyxo +Calyxo Camada camadas +Camara +campanhas +campuscompact canadá Canadá Cancela cancelado cancelamento +Cancos +cancostileandstone Cand canddate CANDIDATADAS Candidatar +candidato +candidatos candidatura Candidaturas +candidaturasdirecionadasxpinc +cannabisandglass +cannondale +Cannondale +canopyconnect +canopytax +canopyworks canva capacidade +capco +Capco +capef +CAPEF +capintel +capitalbank +capitalfarmcredit +capitalgymnasticscedarpark +capitalgymnasticsroundrock +capitalização +capitalontap +capitaltg +capstoneinvestmentadvisors captura capturados capturar caracteres +carbigdata +Carbigdata +carbonchain +carbondirect +carbonfuture +Carbonfuture +carbonrobotics +Cardápio +cardapioweb +cardata +Cardata +cardinalidade +cardinalpoint +careaccess +careerteam +cargomatic +Cargomatic +Cariad +cariadinc +cariboubiosciencesinc +carmichaellynch +carolinatitleloansinc carrega carregadas carregados carregamos carreira +carreiras +Carro +carrotfertility +carta +Carta +casabugre +cascadeloans cascadia +casechek +Casechek +caseguard +casestatus +cashcowlouisiana +cashme casos catálogo +catamountconstructors +catapultsports catarina +catarse +Catarse +catawiki +Catawiki +catchcreationllc +catdaddy CAUSA causar causava +cayena +Cayena +caylent +Caylent +CBEM +cbemllc +cbinsights +ccah +ccahremote +cclfg +cclim +ccompany +Ccompany +cdbabyjobs ceara +celcoin +Celcoin +Celero +celerocommunicationsinc +celigo +Celigo +cellanome +Cellanome +Cellera +cellsignalingtechnology +celonis +Celonis +censys +Censys +centerforemploymentopportunities +Centessa +centessapharmaceuticalsinc +centrais +centraldevagas Centraliza centralizado Centralizamos +centralreach +Centria +centriaautism +centriahealthcare +centrumhealth +Centura +centuracollege +cerc +CERC +ceribell +Ceribell Certificados +certifiedgroup certo +certta +Cescon +cesconbarrieu cespare chainalysis +chainguard +Chainguard chainlink +Chalcot chamada chamadas chamados Chamamos Chamar +championhq +championsgroupholdings channellib +chanzuckerberginitiative +chaosindustries +chaparralmedicalgroup +chariotdefense +charlesriverassociates +Chartbeat +chartbeatinc +charterup +chatguru chave Chaves checamos checar +checkalt +checkr +Checkr +chefman +Chefman chega chegam chegar +chenmoore +chicagotradingcampus +childrenstreehouse +chorusinnovations +chowbus +Chowbus +christfellowship ciclo ciclos Ciclos +Cidadania +cidadaniaja +cielo +Cielo +cientista +cinga +Cinga +Ciranda +cirandacultural +circleso Círculo cirúrgica +cision +Cision +cityoffortworth +citytherapeutics +cityvetinc +civicactions +Civis +civisanalytics clampeia +Claris +clarishealth +Clariti +clariticloudinc +clarityinnovates classifica +classificacao +classificação +classificar +claudecorps +clavis +Clavis +cleancroptech +Clearlink +clearlinktechnologiesllc +clearscoretechnologylimited +clearstreet +clearviewhealthcarepartners +clearwayjobs +clenera +Clēnera +cleoindia +clevelandguardiansbops clicar +clickbus +clickhouse +clicktherapeutics clientes +climateai +climatecabinet +clockworksystems +cloudbeds +Cloudbeds +cloudbedsthirdpartyboard +cloudchamberen +cloudsek +cloverhealth +cloverly +Cloverly +clubcolors +clubmonaco cmdmeta +coaktion +cobaltio +cobaltservicepartners +cobblestoneenergy +cobli +Cobli cobre +cobrem cobrir +coconutsoftware +codazen +Codazen +codeorg +coderoad +cofertility +Cofertility +COFRA +cofraholding +cogentbiosciences +cogna +Cogna +cognite +Cognite +cognitiv +Cognitiv +Cogstate +cogstateinc +coherehealth +Coherus +coherusbiosciences +coinspaid coisa colaboradores +colabsoftware colapsa colaterais coleções +colehourcoheninc +colemanresearch coleta coletadas coletando coletar coletores colidem +collegetrack coloca colômbia Colômbia +colovore +Colovore colunas comandos combinações combinar começa comentados +Comercial +commerceiq +commercetools commitar +commonthreadcollective +communitymanager +commvault +Commvault +Comolatti compactação +companysearch +Compara +comparaja compartilhado +compartilhados compatíveis +compeerfinancial competências compilador complementar @@ -341,6 +1326,7 @@ completamente Completar complexa complexas +compliancygroupllc Componentes Componentização Compor @@ -348,15 +1334,23 @@ comportamentos composta compostas composto +comstock +Comstock comum Comunicação Comunidade comuns +concatenar +concentra +concerthealth concluída concluído concorrência concorrente concorrentes +concretas +concretos +Condomínios conecta conectada conectadas @@ -364,6 +1358,9 @@ conectado conexao conexão confere +conferidos +confiabilidade +Confidenciais configuradas configurado configurados @@ -372,51 +1369,102 @@ confirmada conflita conflitam conflito +conheça conhece conhecer conhecidas conhecido +conifersaicareers conjunto +connectder +connectedcannabis +connectedcannabisco +connectwise +Connor Conseguir +consensys +Consensys considere consistência consolidadas consomem +constantcontact +constellationsoftwareinc constrói +constructionresources construída construtor +Construtora +construtoraelevacao +construtorapatriani consulta +consultados +Consultas +Consultoria +consumeredge +consumerreports consumidas consumidos Consumo +contaazul +contabilizei +Contabilizei contador contagem contagens contas +contasimples contém contêm contenham contentful conter conteúdo +contínua continuam continuar contratação +contratos +Contribuição +contribuições +contribuidores controla +controlai Controle Controles +Convenção +convenia +Convênia +convenientmd conversão conversar Converte convites +Conx +conxconstrutora +cooksys cooperado cooperativa +copperco +cordance +Cordance +Cordell +cordellcordell coreia +corelight +coreone +coretechsecurity +coretelligent +Coretelligent +coretrustpurchasinggroupllc +coreview coroutinelib corpo corporativo CORPORATIVO CORREÇÃO +correlação +correlationone corrente corretamente corretas @@ -424,49 +1472,176 @@ correto corretos corrigindo corrompidos +Cortica +corticamelmed +cosmoslabs cota +Cottingham +cottinghambutlerinsuranceservicesinc +couchbaseinc +Coupang +coupanginternal +courierhealth +covar +Covera +coverahealth +cpisecurity +cplusnow +craftsmansocials +cranialtechnologies +creatio +Creatio +creativefabrica +credaluga Credencial +Credi +credipronto +creditas +Creditas +Cresco +crescolabs +cresta +Cresta +Crestwood +crestwoodcareers +crexi +Crexi +crfamilyofcompanies criação criada criadas criamos Criei Criem +criminaljusticeagency criou Criptografado criptografia +crisprecruit Critérios crítica +criticalmass +criticalmassgroup críticas críticos +crmbonus +crocodilecloth +Croí +croihealth cronômetro +crunchyroll +Crunchyroll cruzada +crystaldynamics +CSCI +csciconsulting +csgconsultants +csmcy +csptecnologia +cssmerge +ctctech +cubos +Cubos +cubosacademy cuidado cujas cujo +culthealth +cultureamp +curaleaf +Curaleaf +Curi +curicapital currículo curta +customcomputerspecialists +customerio customuser +cuyana +Cuyana +cybersheath +cymulate +Cymulate +Dadich +dafiti +Dafiti +Dagster +dagsterlabs +Daiya +daiyafoodsinc daqui +Darkhorse +darkhorseemergency +darkwolfsolutions +dartz +Dartz +dashlane +Dashlane +databento +Databento +datacamp +datagrail +dataiku +Dataiku +dataikujobs +datakindinc +datarails +Datarails datas +datasocietyresearchinstitute +datasystemsanalystsinc +datavant +Datavant datname +datsolutions davecgh +davidzwirnergallery +dayblindscorporate +daybreakhealth +Daymark +daymarkhealth +dbathenextthingltd +dbeaver +dbservices +dcard +Dcard +ddbhealth +ddome +dealersites +debtbook debuglib +debutbiotech +Decarbonization +decibelsinc decidem decidir +Decima +decimainternational decisão declarado Dedicada +dedicado +dedicatedit dedupe deduplica deduplicação deduplicadas deduplicador deduplicar +deepintent +deeplocal +Deeplocal +defcon +DEFCON +defenseunicorns definida Definidas definidos Definimos +definitivehc +definitivehcindia +Definium +definiumtherapeutics defval deixa deixe @@ -475,10 +1650,17 @@ deleção delega deleta deliberada +deliveryassociates deliveryoo demais +democracyforward +democracypreppublicschools +densityai +denverbroncosteamllc departamento +Departamentos depende +dependência dependendo depender depois @@ -491,14 +1673,25 @@ desbloquear descarta descartada descoberta +descobertos +Descobrir +desconhecida +descope descreve descrevendo +descrições +descript +Descript desde desempenho +Desenho +desenvolvedora Desenvolvedores Desenvolver deserializar desfaz +designbridge +designedconveyorsystems desligando dessa destino @@ -511,33 +1704,68 @@ Detalhe detecção detecta detectadas +detectado determinístico determinísticos +detroitlions +developmentseed devem +devfuturetalent devolver devolvido +devrev +devries +devtechnology +dfinity +DFINITY +dhigroupinc +dhpace +diabolocom +Diabolocom diacríticos +dialpad +Dialpad +dianahealth +dianthustherapeutics diária Dica +dicefm diferencial diferenciar diferentes difflib dificuldade +Digi +digimarc +Digimarc +digisecuritysystems +digitalbiology +digitalbridge +digitalcurrencygroup +digitalextremes digitar digitou +digix +Digix +diligentcorporation +diligentrobotics +dimagi +Dimagi dimensões dinamicamente dinâmicas dinâmico direção +Direcionadas direitos direto diretoria diretorios diretos +discmedicine dispara disparado +disparam disparar disparem dispersas @@ -545,19 +1773,70 @@ disponibilidade Disponibilizar disponível disso +distantjob distinguir distribui +Distribuição +distribuído +distribuidos +districtofcolumbiainternationalschool distrito +distrokid +Divco +divcowest diversas +dkatalislabs +dkbcodefactory +dlhcorporation +dlpbank +dlrgroup +dmgevents +dnsfilter +doctolib +Doctolib +docugami +Docugami documentação documentadas documentado Documentar +Dodgshun +dodgshunmedlin dois +doitintl +dollarshaveclub +doma +Doma +donorbox +Donorbox +donorschoose +donorschoosefellowship doordash +doordashaustralia +doordashcanada +doordashinternational +doordashmexico +doordashusa +dotgroup dotlottie +doublegood +doubleverify +downtownmusic +doximity +Doximity +dqrtech +dragos +Dragos +drdansanimalhospital drena +drsquatch +drweng +dtidigital +dtlabs duas +Duetto +duettoresearch +Dunorte duplica duplicada duplicadas @@ -565,25 +1844,105 @@ duplicar duplicata duplicatas duração +Durin +durinmining +duxcompany +dvtrading +dxacirca +DXTRA +dynetherapeutics +dyopath +DYOPATH +Eames +eamesinstitute +earlycareerprograms +earnin +eastharlemtutorialprogram +Eastside +eastsidedermatology +easygo +Easygo +ebanx +EBANX +ebury +Ebury +Echodyne +echodynecorp +eclinicalsolutions +eclipsetrading +eclipseworks +ecoatmgazelle +ecore ecossistema +Edelman +edelmanfinancialenginesllc +edenred +Edenred +edgewoodpartnersinsurancecenter +edição edita editadas editados +edmentum +Edmentum +edobestsandbox +Educação efeito efetivamente efetivo +efficientcomputer +Efficio +efficioconsulting eficiência egito +eikontherapeutics +eiti +Eiti elas +eldersburgveterinaryhospital +electrasteel +elementalimpact +elementbiosciences Elementos +elend +Eleos +eleoshealth eles +Elevação elevada +eleventhhourgames +eliotcommunityhumanservices +elitedentalpartnersllc +elitetechnology +Elligint +elliginthealth +elogroup +elwoodtechnologies +emarketer +EMARKETER +embroker +Embroker embutida +emea +emergentlabsinc +emergeventures emite emitterc +Emota +emotainizioengage +empacota +Emplifi +emplifimonster +employerdirecthealthcare +empowerbrands empresas +emslinqinc +Enavate +enavatecareers Encerra encerrando +Encerrar +enchargeai encontra encontrada encontradas @@ -593,21 +1952,62 @@ Encontramos encontrar encontre encontrou +encora +Encora +encoura +Encoura +endeavourinspiredinfrastructure +Endor +endorlabs +eneba +Eneba +energage +Energage +Energia +energyexemplarllc +energyhub +energyimpactpartnerslp +energysolutions +energysolutionsinternships +energytec enfileira enfileirada enfileiradas enfileirado enfileirar +engageseniortherapy engajar +engenharia +engenheira Engenheiro engenheiros +engineersgate +englishcanada +enigmaio +enjoei +Enjoei +enlacehealth +ennoblecare +enova +Enova enriquece enriquecer enriquecimento enriquecimentos +ensco +ENSCO +enscoskillbridge então +entera +Entera +enterpret +Enterpret +entersekt +Entersekt entrada +entradatherapeutics entrar +entreehealth Entregar Entregas Entrevista @@ -617,22 +2017,60 @@ envia enviado envie enviou +envisionconsulting +enviva +Enviva envolver +envoyglobalinc +envoymortgage +enzrossi +EOLA +eolapower +eonio +epcminc +epickids +episodesix +episodesixlinkedin +eplusinc +eqtcorporation +equalexperts +equatic +Equatic +equilibriumenergy equipe +equipmentsharecom equivale equivalente +Eqvilent +eqvilentjobs +erasca +Erasca +ergeon +Ergeon +ernesta +Ernesta +ernestpackagingsolutions ernstandyoung ernstyouth errado +erural +esbuild Escalabilidade escalada escaláveis +Escola +escolaconquer +escoladnc escolhe Esconder escopo escreve +escrever +escribers escritorio +Escritórios Escuro +escuta esgotada esgotadas espaço @@ -655,13 +2093,21 @@ espelhar espera esperada esperado +esperados esperar +espirita +Espirita esqueci essa +essenceit +essencialnutricao +estabilidade estados Estados estagiario estágio +estapar +Estapar estar Estas estático @@ -681,20 +2127,71 @@ Estranha estranho Estratégia estratégica +Estrela +estrelabet Estrita estritamente Estrutura estruturados Estudantes Etapas +etchedai +etchinc +etec +etelligentgroup +ethernovia +Ethernovia +ethoslife +eudia +Eudia +eulerity +Eulerity +Eventbrite +eventbriteinc evento eventos +eventsandinterns +eventualmente +eveo +EVEO +everagtester +evergreennephrology +evergreenservicesgroup +everlane +Everlane +everlaw +Everlaw +everway +Everway +evgspecialtynetwork +evio +Evio +evismart +Evismart +evitam +evitando evitar evolução +evolui +evoluir +evoluiu +evolutionaryscale +evolutioniq +evolvevacationrental +exabeam +Exabeam +Exadel +exadelinc +exame +Exame +Exames exatamente +exati +Exati excecao exceções Excelente +excelsportsmanagement Excluir exclusão exclusivo @@ -710,15 +2207,22 @@ Exibindo exige exigem exigências +Eximio existam existem existente existentes existia existirem +exoduspoint +Expana +expandidas Expandir expectativa +Experi experiência +experigreen +expertnetwork Expira expiração expiradas @@ -731,7 +2235,9 @@ expirou explicitamente explícito explodir +Explora Explorar +explorasolutions exporta exportacao exportadas @@ -739,15 +2245,39 @@ Exportado Exportando Exposição expostos +Extenteam +extenteamcareers externa +externaljobboards externamente externas externo externos +extrahopnetworks extrai +extremegroup +eyeo +eyxo +Eyxo +ezcaterinc +ezcatertalentcommunity +Fabrica +Facilita +facilitapay +factorialenergy +Faeth +faeththerapeutics +fairlife +fairmarkit +Fairmarkit +Fairstead +fairsteadescllc fala falam Falar +falconi +Falconi +falconx falhar falhas falhe @@ -755,59 +2285,291 @@ falhou falso falsos falta +fambrands +familyoffice +familywell +fanaticscollectibles +fanaticsfbg +fanaticsinc +fará +faradayfuture +farmsanctuary +farmtech +Farmtech +faropoint +Faropoint +fartherfinance +fashionnova +fastautoloansinc +fastpaydayloansfloridainc fazem +fbiz +Fbiz +fcamara +fccincinnati +federato +Federato feitas +Feldman +felixandfido +FERMÀT +ferocia +Ferocia ferramentas +feverup +fgsglobal +fiap +FIAP +fiberhome +ficam ficar +fictiv +fieldstonebio +figureai +filescom +filson +Filson filtra filtrada +filtragem Filtre filtro filtros +finais Finaliza FINALIZADO +finallevel +Financeiras +financialtimes +findup +finepointconsulting +finitestate +Finom +fintalk +Fintalk fintech +fiori +firaxis +Firaxis +fireworksai +firmpilotailawfirmmarketing +firmus +Firmus +firstconnectinsurance +firstnationalbankofamerica +firstprinciples +firststepsforkids +fitenergia +fiveringsllc +fixify +Fixify +flagshippioneeringinc +flatironenergy +fleetio +Fleetio +Fleetworthy +flemingeducacao +fletcherjonesautomotivegroup flexibilidade flexíveis +flexport +Flexport +flighthub +flipdish +Flipdish +flockhomes +flodesk +Flodesk +flohealth +floraconsultoria florianopolis +flowtraders fluído +flutterbrazil flutuação fluxo +fluxon +Fluxon Fluxos +fluxx +Fluxx flyio +flyr +FLYR +flywheeldigital focado +focalsystems +foco +focusfinancialpartners +focuspartnerswealth +foliahealth +Follett +follettsoftware +FOLX +folxhealth fonte +forafinancial foram +forcetherapeutics +Foresite +foresitelabs +forgebiologics +forgeglobal +forgehealth +formaaiinc formata formatada Formatando +formationbio formato +formhealth forminfo +fornecer fornecidas fornecido fornecidos +forta +Forta +forter +Forter +forthea +Forthea +fortisfiresafety +fortra +Fortra +fortrobotics +forumone +forwardnetworks +Fospha +fosphamarketing +fossainc +foundationriskpartners +foundrydigital +fourhands +fourkites +fourthline +Fourthline +foxbit +Foxbit +foxen +Foxen +foxholetechnology fpconv +fractile +Fractile +Fractyl +fractylhealthinc +frameworkdigital frança França +freedassociates +freedomtogether +freeformfuturecorp freela +freenome +Freenome +freenow +Freenow +freestonecapitalmanagement frequentemente frequentes +freshprints +fretadao +Fretadão +frete +fricon +FRICON +fronteira +Frontera +fronterahealth +frontierdermatologyprovidercareers +fsastorecom +fsbholding +fueledcareers fulano Fulano funciona funcional funcionalidades funcionando +funga +Funga +funil +fuseglobalpartners +fusionworldwide +Fussball futura futuras +futurhealth +Futurhealth futuro +gaintheory +gaithersburg +Gaithersburg +galaxydigitalservices +galaxyservicepartners +gallaghereveliusjonesllp +galvanizeclimatesolutions +Gametime +gametimeunited +gapinternational +garagedoormedics garantindo +gardacp +garnerhealth +gassouth gasto +gatesventures +gatherai +Gatik +gatikaiinc gatilho +gavindebeckerassociates +gavindebeckerassociatesindeed +gcmgrosvenor +gcservicos +gearboxquebec +Gelber +gelbergroup +gelberhandshake +Gelfand +gelfandrennertfeldman +gemechanicalasp +genea +Genea +generalassembly +generalatlantic +generalcatalyst +generalmatter +generalproximity genérico +Genetix +genetixbiotherapeutics +genevatrading +Genezen +genezenlabs +geniant +geniantllc +geniussports +geniussportssn +genomines +GENOMINES +genscript +gensyn +Gensyn +genyo +Genyo +geocgi geohash geolocalização +georgiaautopawninc +Geospatial +geotab +Geotab +geotekoperationslimited gerados +gerdau +Gerdau gerencia gerenciadas gerenciado @@ -815,53 +2577,345 @@ Gerenciando gerenciar Gerente Gestão +getbuilt +getwhys +gibsondunn +Giga +gigaenergy +gillig +GILLIG +gingerlabsinc +ginkgobioworks +girleffect +gitai +GITAI +givecampus +givedirectly +givewell glassdoor Glassdoor +gleanwork globais +globalenergyallianceforpeopleandplanetgeappllc +globalhealthcareexchangeinc +globalizationpartners +globalli +Globalli +globalsystem +globalwebindex Globex Globo +glossgenius +glyphicbiotechnologies +goalcast +Goalcast +goalhonduras +goalsyria +goatgroup goautoneg +gobravo +gocardless +godfreydadichpartners +goflow +gofundme +gogroup +Gogroup +goguardian +Goken +gokenamericallc golangci +goldenstate goldmark +gomotive +gongio +goodbysilversteinpartners +goodfire +Goodfire +goodhouse +Goodhouse +goodinside +goodjobgames +goodnotes +Goodnotes +goodr +goodsservices +Goodway +goodwaygroup gopkg +GOPWR goquery +goremutualinsurance +gorjana +gotion +Gotion +gotjunk +govtechbarbados +gradial +Gradial +gradientai +grafanalabs Gráfico +grahamcapitalmanagement +gramgamescareers grandes +granum +Granum +graphcore +Graphcore +grasshopperasia +grassi +Grassi +grassrootsanalytics Grava gravar +Grayscale +grayscaleinvestments +greatstate greehouse +Greenbrook +greenbrookmedical +greenirony +greenplaces +Greenplaces +Greenpoint +greenpointtechnologies +greenthumbindustries +Greenworks +greenworkssunriseglobalmarketing +greyus +Gridmatic +gridmaticinc +groma +Groma +Groome +groomecareers groot +Grosvenor +grovecollaborative +grovelane +growe +Growe +GROWE +growetalents +gruns +Grüns +Grupo +grupoagcapital +grupoboticario +grupocentral +grupocomolatti +grupoeximio +grupofoodnation +grupofour +grupogcb +grupogcbinvestimentos +grupogopwr +grupoguiainvest +grupogvc +grupolider +grupolos +grupomedeiros +grupopermaneo +grupoprotege +gruposaga +gruposrm +grupotaking +grupothink +grvty +GRVTY +gsgcareers +gsrmarkets +guaracamp +Guaracamp +guardianrestoration +guardsquare +Guardsquare guia +Guidelight +guidelighthealth +guidepoint +Guidepoint +guidepointsecurity +guidepostmontessori +guildgaragegroup +gulfwindtechnology +gumgum Gupy +gwkinvestmentmanagementllc +Gyde +gympass +gymshark +Gymshark Habilidades +habilitadas habilitado habilitar +habitathealth +hackerrank +haigroup +Haize +haizelabs +hala +HALA +hana +Hankook +hankooktireamericacorp +Hanwha +hanwhaenergyusa +hanwharenewables +harbingermotors +harborglobal +harnessinc +harpergroup +harpia +Harpia +harrowhealth +harrys hashedpassword +hatchcareers +havenenglish +hbstudios +headlandsresearch +headlandstechnologiesllc +Headout +headoutcareers +headoutlinkedin +headoutreferrals +Headspace +headspaceproviders +healthjoy +healthlink +hearcom +hearcomin +heartaerospace +Heartflow +heartflowinc +heartpaw +hebrewpublic +heliosx +hellobackpack +hellofresh +hellommc +helpinghandsfamily +hereio +herselfhealth +hexagonbio +hexium +Hexium +heygen híbrid híbridas Híbridas híbrido +hiddenlayer Hierarquia +hífen +highdive +Highdive +higherlogic +highmetric +highnote +Highnote +hightouch +Hightouch +highwire +Highwire +hillandknowlton +hillhousehome +hillpointe +Hillpointe +hiltonbrasil hintrc +hiper +Hiper +hipokee +hiro +Hiro Histórico +hivewatch hoje holanda +holderconstruction +holisticindustries +homechef +homeconstructionregulatoryauthorityvolunteer +homeinstead +homelight +homemarketfoods +homesolutions +hometap +Hometap +hometapjobs +honeathome +honehealth +hoodhp +hootsuite +Hootsuite +hopscotchprimarycare +hopskipdrive +horacemannagents +horacemannservicecorporation +Horizen +horizenlabs +horizenlabstalentpool +horizonindustrieslimited horizonte hotjar +Hotmart +hotmartcareersbr +hourglasscosmetics +housemarque +Housemarque +housinganywhere +hoyoverse +hpiq +hrpliving hscan HSTS +huddleup +hudl +hugeinc huggingface +humanagency +humaninterest +humanrightswatch +humansignal +humeai +hummingbirdregtech +hungryroot +hydritechemicalco +hyliion hyperloglog +hyphenconnect +ians +ibanfirst içado +icapitalnetwork +icatu +Icatu +icatuseguros +icherry +iconcareers ícone +iconiq +iconit +idahotitleloansinc ideais idênticas idênticos identidade Identificação +identificado Identificador identificadores identificar +idme +idmeuniversityrecruiting +idnow +idwall +ifood +ifoodcarreiras +iftother ignora ignorada ignoradas @@ -870,11 +2924,23 @@ ignorados ignorando ignorar iguais +iherb +ihunters +ília ilike ilustram ilustrativos +imaflora +Imaflora +imaginepediatrics +imagineworldwide imediata imediatamente +Immunome +immunomeinc +Impinj +impinjexternal +impiricus implementa implementação implementações @@ -887,10 +2953,21 @@ Importando importante importantes Importar +Impulso +impulsogov imutáveis imutável +IMVT +imvtcorporation +Inalab +inalabconsulting +incadigitalinc +inchargeenergy incluem inclui +incluído +incode +incognia incompleto incrementa indefinidamente @@ -904,27 +2981,43 @@ indexam indexar Índia indica +indicacoesifoodinterno índice índices +indigenouspactpbcinc indiretamente indisponíveis indisponivel indisponível individuais individualmente +industrialelectricmanufacturing +industriasanhembi +industriouslabs inesperado inexistente inferência inferidos infinito +infinitumelectric inflar +inflectionai +infleet +Infleet infojobs +informa +informacao +informação informada informado informativa Informe +Infotec +infotecbrasil +infotrust Infraestrutura ingestão +inhometherapy Inicia iniciado Iniciados @@ -934,13 +3027,31 @@ inicializador Iniciando inicie início +Inizio +iniziomedical Injeta injetado injetar +inkind inmemory +inmobi +Inno +innodatainc +innogames +innoveahub insere +insiderstore inspecionam +inspiraeducation +inspiremedicalsystemsinc +inspiren +Insta +instabase +instacarro instacart +instalação +Instalação +instalador Instancia instância instanciado @@ -948,167 +3059,607 @@ instâncias instantâneo instantes instável +instawork +instiglio Institucional +Instituto +instride +instridehealth instruções Instrumentação +insurify +insurityindia intacto +intecrowd integra integração Integração integrações integrada integrado +integralmolecular integrar inteira inteligência Inteligência inteligente +intelluminc intencionais intentando +intera +Intera interativa +interbrand Intercepta interessa internacional +internaljobsatlush internamente internas +internos interrompa interrompe +interrompida interrupção Interseção +interstellarlab +intersystems +intertechne +Intertechne Intervalo +interviewengineering +interwellhealth +interworks +inthepocket +intradiem +intrinsicrobotics intuitiva Inválida invalidação invalidar inventário +inversionspace invertido invertidos invés +Investimentos +invivyd +inyova iolib +ionq +ionqcontractors +iorq +IORQ +iovancebiotherapeutics +iovino irlanda +irradianttechnologies +isaraerospace +isccareers isola +isolada isoladamente isoladas isolados isolar +isomorphiclabs +ispottv isso +isystems iterações +iterativehealth +iteris +Iteris +itriad +ITRIAD +itslogisticsllc +Ivanti +ivcevidensia +iventis +IVENTIS +ivxhealth +ixllearning +jacquemus +JACQUEMUS +jadebiosciences +jamfilled janela +janestreet +jeitto +Jeitto +jennikayne +jensenhughes +jjsnackfoods +jlkfsk joana joaosilva +jobsatphamily +jobsforthefuture jobsglobal jobsglobalscraper jobstore +johnmcorcorancompany +joinaffect +joinforage +joinparadigm +jomboymedia jooble Jooble +jordanparkgroup +joskoasp +joya +jrmconstructionmanagementllc +jshiddenevents +jukeboxhealth julho +julieproducts +jumio +jumpcrypto +jumptrading juridica +Jurídico +justanswer +justfoodcompany +justfund +justos +Justos +justprotein +justworks +juullabs +kaiko +Kaiko +kailera +kairospower +kalcon +kalepa +kalshi +kamino +Kamino +kanastra +Kanastra +karbon +kardfinancialinc +kardigan +karllagerfeld +karya +kasa +Katalis +katalyst +kateschwartzphysicaltherapyjobs +kayzen +kearlycareers +keeleyconstruction +keelinfrastructure +keepersecurity +kellerpostman +Kempner +kenjyatrusantgroup +kensingtoncorporate +kensingtontours +keplergroup +kernalbio +ketryx +keyfactorinc +khaerospace +khaite +khanacademy +khealthcareers KHTML +kickstarter +kidscountry +kife +Kife +kikoff kimi Kimi +kincellbio +kindbridgecorporation +kindsnacks +kinexus +kivaorg +kiwicoinc +kiwify +Kiwify klarna +klavi +kmadrid +knak +knightdivisiontactical +knithealth +knowbe +knowde +knowledgecity +Knowlton +koboldmetals +koboldmetalsdrc +kodiaksolutions +koleyjessen +kolmacintegratedbehavioralhealth +komodohealth +konovo +konux +Koopere +kooperecooperativa +Korman +kostelanetzllp +krafton +kraftonamericas +kraftonindia +kreativekids +krollbondratingagency +kronosresearch Krsj +kstack +Kstack +kunai +kuraoncology +kurosbiosciencesinc kvstore +kwalee +Kwalee kwsync +kyowakirinusa +labelbox +Labi +labiexames +labisaude +laboclin +Laboclin +labviva +labwarebrasil +lacarguy +Lagerfeld +lakeasbury lakefield lakefieldvet lakefieldveterinarygroup +lakefrontbiotherapeuticsinc lança Lançado lançam +landdesign +langanengineeringandenvironmentalservicesllc +LAPORTE +laportefr +laporteusa +lasenza +lastlink +Lastlink +lastpass Latência +launchpadtechnologiesinc +lawmatics +lawzero +layerhealth +layerzerolabs +lazarusenterprises +leaderetalent +leaflink +leagueinc +learneo +learningnetwork +learnlux +learnupon legado +legalservicesnyc +legatosecurity +legendcareers legível leiam +leit leitura leituras +lello +Lello +lemonenergia +lemurianlabs +lendingtree +lendo +leolabsinc lerá letras +letrus +Letrus +levanta +levelaccess +leveltenenergy +levelworks +levio +lexingtonmedical +lfaadvogados +lgairesearch +lgelectronics +Líber libera LICENCE +licor +lider +líder +lifelinkiii +lifeskillsautismacademy +Ligga +liggatelecom +Lightfarm +lightfarmstudios +lightfeatheriollc +lightforceorthodontics +lighthousebehavioralhealthsolutions +lightningai +lightspeeddms +lightspeedsystems +LIGN +lilasciences +limaconsulting limita limpeza limpos linha linhas linit +LINQ +linx +Linx +lionsheadprecisionmetals +liontree +lisc listagem listas +listo +Listo +litify +litmos +littlebutterflies +littlewordsproject +livecareer +livefront +livelo +Livelo +livemode +liveparalleljobs +liveperson +liveviewtechnologiesinc +livup loadlib +localcoin localidade +localitymediallcdbafirstdue localiza Localização localizado localmente +locusrobotics +loenbro +loftfederal loga logar +loggi +Loggi +logicalintelligence +logicgate +lokainc longa longo +looneyrickskiss +loonshotgames lote +Lótus +lotussolution +Lovehoney +lovehoneygroup +Lovin LPUSH +luby +Luby +lucidbots +lucidmotors +lucidsoftware +luckybeverageco +ludorobotics lugar +Luiza +lumahealth +lumbermens +Lume +lumimeds +lumine +Lumine +lunarenergy +Lunn +luno +lusternational +Luzandreya +luzandreyaconsultoria +lwsa +LWSA +lyellimmunopharma +lyncas +Lyncas +lynxanalytics +mabl +machinifyinc +mackaysposito +madano +maddoxindustrialtransformer +madisonlogicinc +maev +magazineluiza +magazord +Magazord mágica +magnasearch +magrathea +Maineri +mainstreethealth maintnotifications maior maiores maioria maiúsculas +makeawishamerica +malbon +mammothbrands +mandm +manifoldai +manifoldbio +manscaped mantém Mantemos mantendo +Mantenha manter mantido +mantl +mantrahealth manualmente +manychat +Mapa +mapahds Mapeando mapear Mapear mapeia +maravailifesciences marca marcado marcar margem mariaclara +mariadbplc +markforged +markmorrisdancegroup +markspainrealestate +marqeta +marqvision marrocos +marscousg +marshallwace +martellgrowthsolutions +maslanskycareers mastigada +matchpointtx +matera +Matera +materialbank +matherheadquarters +matherintysons +matherplace +mathersplendido +mathgroup mathlib +matik matriz +matteprojects +matx +mavenclinic +mavenrobotics +mavenscareers +mavensecuritiesholdingltd +Maverik +maxcessinternational máximas +maxmanufacturingcareers maxtimeout +maymobility +mazzatech +mbooth +mboothhealth +mcadams +mcclureoilcorporation +mcghealth +mcneeswallacenurickllc Mczn mecanismo +mechanicallicensingcollective +medelitellc +medeloop média +mediabrands +mediahero +mediasmart médio +medispend +medistrava +meditelecare +Medlin +medrio +medsien +medway +Medway meia +mejuri melhorar melhores MELHORIA +melio +meliuz +Méliuz +Melmed +melnick +Melnick membro membros +memic memlock memoria memória +mena +MENA menos mensagem mensagens mensais mensuráveis +mentalhealthcenterofdenver +mentimeter mentores Mentorias +merceradvisors +mercyforanimals +mergeworld +Meridial +meridianpartners +meruhealth mescladas +meshopticaltechnologies mesma mesmas Mesmos +messari metadados +metalab +metapack Metas +methodco +metisinc método métodos +metoxinternationalinc MÉTRICA métricas +metron méxico México +mgproperties +mgroofing +michaelanthonycontractingcorp +microblink microsserviços +midhaus +Midhaus +Mídia +midihealth +midpenhousing +midpointmarkets +mightynetworks Migrações +milhares milissegundos +millerremick +mindgrub +mindgym +mindtheproduct +mineralystherapeutics +minertecnologia Minhas mínima mínimas @@ -1116,139 +3667,493 @@ mínimo mínimos miniredis Miniredis +minitab +minlab +minno mintimeout minúsculas minuto +miopartners +miqdigital +mirakl +mirakllabs +miris +mirum +Mirum +misfitsmarket +mishimoto +missionhealthcare +missionlane +missionlanellc +mississippititleloansinc +missourititleloansinc +mithril +mitratech +mitsogoinc +mitsubishimotorsna +mixbook +mksolutions +mlbnetwork +mlops +mntn +mobentertainment +mobiik +mobileanesthesiologists +mobileye +Mobileye +mobilityware +mochihealth Mockada mockado mockados Mockar modelo +modelos +moderawealthmanagement +modernhealth modernos +modulrfinance +mogli +mojangab +moloco +momentenergy momento momentos +momentumcompany +momentumfinancialservicesgroup +momentus mondelezinternational +monetserviceinc +moneyherogroup +moneymart +moneysmart monitora Monitoradas monitorado Monitorar +monocl +monsterenergy +monstro montá +montada montar +monumentalsports +monzo +moodhealth +moonlite moonshotai +morganmorganjobsapplynow +morrisonmaierle +mossnewyorkllc Mostrando Mostrar +motifneurotech +motim +MOTIM motivo +mottu +Mottu +movecta +Movecta +movementstrategy +moveonorg movimente +mqreferrals +mrapplecareers +mrbeastyoutube +msfcareers +mspestudios +mthreerecruitingportal +muckrack muda mudam mude mudou +muitos múltiplas múltiplos mundiais +mundiale +Mundiale +muonspace +murj Mvgo +mwinternshipprogram +mxtechnologiesinc +myfitnesspal +myfundedfutures +mythicalgames myuser +nabis +nadiacare +nakedfarmer +nametag +nanit +nanonets +nanopathinc +narvar +nasce +nascompany +nashvillezoo +natera +nationalbusinesscapital +nationallutheraninc +nationalpublicradioinc +naturais +naturesbakery +naughtydog +navapbc navegável +navierboat +navtechnologies +navvis nayarakarinesilva +nearform +nearspacelabs negócio Nenhum +neocybernetica +neogrid +Neogrid +neonaerospace +neoris +neoway +Neoway +Nephrology +neptunebio +neptunemedical nerdwallet +nerostechnologies nessa +nessas nesse nesta +netbrain +netdocuments +neteasegames +neuehealth +neuraflash +neurahealth +Neurodevelopmental +nevadatitleandpaydayloansinc +newarkacademy +newengeninc +neweratech +neweratechnology +newglobesandbox +newlabcareers +newleafenergy +newlimit +newsela +newsrevenuehub +nexaas +Nexaas +nexightgroup +nextinsurance +nextjs +nextroll +nextrollinc +nflcareers NGJH +ngmbiopharmaceuticals +nibo +Nibo +nift +nightdivestudios +nimblerobotics +ninecon +Ninecon +ninedotholdingsinc ninguém +ninjatrader +ninjatradercontractors +nisc +nitricity Nível +nlcventures +nmcareers +NNOVEA +noahmedical +nobuhoteltoronto nocmp +noctrixhealth +noctuatechnology nodetype noemail noite nolint +nomadglobal +nomadhealth nomeado nomes +nomina +nonprofitfinancefund normaliza normalizada normalizado normalizados normalizar +northbeam +northeastanimalclinic +northmarq +northpointrecoveryholdingsllc +northpointtechnology +northspyre +northwestadministratorsinc Notas +notexternal notificação Notificações notificou +novacredit novamente +novoed novos novousuario +noyocareers +nozominetworks Npqgf nsis +ntconcepts +nuancelabs +nubank +nuclea +Núclea nulos +numa numbackends numérica números nunca +nurix +nutrafol +Nutrição +nutscom +nuvem +nuvemshop +Nuvemshop +nuvidio +Nuvidio +nycedc +nysonian +oafkenya +oasishealthpartners +oasissecurity +obexp objetivo +obrienveterinarygroup obrigatória obrigatório observabilidade Observação Observações observáveis +obsidiansecurity +obsidiantherapeutics obtém obter ocasionais +oceandrop +oceanx ocorre +ocrolusinc +octafy +Octafy +octaura +octus +odeko +odlesalescareers +offerup +offerzen +officehours +officespacesoftware +offshorelaunch +oficiais +Ogaleão +ogilvyhealthuk +ogilvyhealthusa +ogilvymena +ogroup +ohalogenetics +oklo +OKRP +oldcastlebuildingenvelope +olema +olgari +olipop +olist +Olist +Oliveira +oliveiraeantunes +oliverseapac +oliverusa +olsson +omadahealth +omenai +omidyarnetwork omitido +omnicomhealth +onapsis +onbe +oncoverycare onde +oneacrefundethiopia +oneacrefundglobal +oneacrefundmalawi +oneacrefundnigeria +oneacrefundrwanda +oneacrefundtanzania +oneacrefunduganda +oneacrefundzambia +oneearthfuture +oneenergyrenewables +oneimaging +onemodel +onenergy +oneoncology +onepath +onesixsolutions +onetrust +onexgeneral +onnitlabs +onxmaps +ooma Opção opcionais Opcional +opcionalmente opções +openap +opencoreventures +openeye +openfx +openlabs +opentable operação operacionais operacional operacionalmente operações Operador +operantai +operationscareers +oportun oportunidade oportunidades +opploans +opportunites optar +optimadermatologycareers +optimalcare +optimaldynamics +optimecare +optimove +optiverus +optotune +Optotune +oralsurgerypartners +orangetwist +orasio +Orasio +orbia +Orbia ordem ordenação +orennia Orfaos órfãos organicamente organização +orientada +orientado Origem origens originais +oriongroup +orium +orizon +Orizon +orkes orquestração +orthosportsmedphysicaltherapyjobs +oruka +osano osbuilders +oshihealth oslib osscluster otimização otimizado +otrcapital +ouihelp +oulahealth +ounceofcare +oura +outerspace +outpostspace outra +outschool +outsetmedical +overstory +owenscompaniesasp +ownwell +oxosmedical +pacaso +pacificfusion +pacificlegalfoundation +packardculliganwater +pacnyc pacote +Pactual +pacvue +padronizar +Padronizar pagas +pagaya +pagayais paginação páginas +paidyinc paineis +painéis painel +paipe +Paipe +pairteam países +palaktakidsacademy palavra palavras Paletas +palmettocleantech paloaltonetworks +palosverdes +pandadoc +pandvil +panthalassa +pantheonpublic +pantherlabs papéis +paperlessparts papo +parachutehealth +parachutehome paralelas paralelo +parallellearning parar parceiras +parciais Parcial Parece +paretocaptiveservicesllc +parloa +parrishdevaughn parseado +parsecautomationcorp parseia parserc +parsleyhealth partes Participação partir @@ -1256,28 +4161,59 @@ passa passada passadas passado +Passagem passar passem passos passou +passportbranddesign +pathrobotics +pathstream +pathward +pathwaysforlife +patientpoint +Patriani +patterndata Pausado Pausar +paveakatroveinformationtechnologies +pawapay +paxlabs +paynearmeinc +paypay +paypaycard +paytient +paytrack +Paytrack +pdtpartners +peachpilot +pearceservices +Pedra +pedraagroindustrial pegar pela +pelago pelas pelos peloton pendentes +pendo +penninteractive +pentest percentunit perdeu +peregrinetechnologies +perfectserve perfeitamente perfis performático pergunta perguntas periodo +perionnetworkltd permanecem permanente +Permaneo permissivo permite permitida @@ -1285,41 +4221,110 @@ permitidas permitidos permitindo permitir +perpay +perscholashires +persefoniaiinc persiste persistência persistente persistentes persistidas persistir +personalisinc +personalizedbeautydiscoveryincdbaipsy pesquisa pesquisadas +pesquisando Pesquisem pessoa Pessoas +petag +petitrh +phaidra +phamily +phantomai +Pharma +phasev +philadelphiaeagles +philliesbaseballoperations +phoenixcontact +phonepe +phynetdermatology +physicsx +picarroinc +picoquantitativetrading piechart +pieinsurance +piermontbank +pilothq +pineadvisorsolutions +pingidentity +pinn +pinwheelapi pior +pipetechnologies +Pipo +piposaude +pirateship +pistontechnologies +Pitaco +pitang +Pitang +pitchbookdata +pivotbio +pixability +placementsio planejar +planetlabs planetscale +planningcenter +platacard plataformas +platformbuilders +platformscience +platinumdermphysicians +playsports +plootocareers +plos +plscareers plugar +pluspower pmezard +pmfo +pmguk +pnlfin pôde Podemos poderem poderia poderosa +pointc +pointwild pois +pokemoncareers +Poliedro +política políticas +polyai +polychaincapital +polygonus +pomelocare ponteiro +pontera ponto Pontos +poppulo populada populado porém porque +porschesouthbay portanto portas +porternovelli portfólio +Portofino +poshmark posicionamento posições positivo @@ -1329,16 +4334,37 @@ possivelmente Possuam possui possuir +postmanlaw +postpartumsupportinternational +pottencial +Pottencial poucas pouco +powerdigitalmarketing +powerfinance +powerhousearts +practicebetter +practisinglawinstitute +praso +Praso +prathaminternational prática práticas prático práticos +pravaler +Pravaler +praxent +praxisprecisionmedicines precisa precisam +precisamos precisar +precisionaq +precisionmedicinegroup +precisionvehicleholdings preciso +predictiveindex preenche Preenchendo preenchida @@ -1347,27 +4373,52 @@ preenchidos Prefere Preferências preferido +preferir +Preferir +prefira prefixo Prefixo prefs +premiersoft +Premiersoft preparar +presencelearning presenciais presencial presente presentes preserva preservada +preservados +preservam preservando +presidentssummit +prezzee Pricebook +pricefox Primário primeira primeiro +primemedicine +primerai +primexbt principais +principia +Principia +priner +Priner printou Prioridade Priorização +pris +Pris +privateequityinsights +privatehealthmanagement privilegia +Prizma +prizmamidia PROBLEMA +procaresolutions processa Processadas processamento @@ -1375,65 +4426,180 @@ processando processar processo processos +processstreet +procfit +Procfit procura +productpeople +professionalstaff +Proff profissionais Profissionais profissional +programador +programadora Programathor progresso proibidos +projectaservicesgmbhcokg projetado Projetado projetos +prokidney +prolaio promauto prometheuscommunity promhttp promtail propaga propagar +prophecysimpledatalabs +prophero propriedades próprios +propublica +prosek +proskill +prosperhealth +prospertechtalents +prosus +Prosus proteção protegida +proteinqureinc +protillionbiosciences +protonai Protótipo provisionados próximas próximo Próximos +psibufet +pubgmadison Publicacao publicação publicada publicadas publicado +Publicar +publiclabel público públicos Publique +publishingglassdoor Puerkito +pulsebiosciences +pulumicorporation +pumpcareers pura +purestorage puro +purplestrategies +purposemed +pushpay +putnamassociatesllc +pyra +pyramidroofing +pzenainvestmentmanagement +qeevo +Qeevo +qgenda +qitech +qive +Qive +qohash +qphox +quadbridge +quadcode +Quadcode +quadraturecapital Quadro +quaise qualidade +qualifieddigital +qualio +quanata +quansight quantas +quanticdream Quantidade +quantifind +quantinuum +Quantinuum quanto +quantumsi +quantumspacellc +quartzbio +quberesearchandtechnologies Quebrada quebrado +quebrados +queracomputinginc +Quero +querodelivery +Querodelivery +queroeducacao +queropassagem +questbridge +quicksoft +quillbot +quintoandar quiser +rabinmartin +racapitalmanagementllc +rackner +radiantsecurity +radiclehealth +radixark +radixexperienced +radixuniversity +railsware +ramosmarbleandgranite +rangeviewinc +rankmyapp rápida +rapidfortinc rápido +rapidsos +rapp +Raro +rarolabs +rayeitconsulting +razorpaysoftwareprivatelimited +rclco +rcxsports +rdccareers +rdsourcing +rdstation +reactjs readerc readwriter +readysettechnologyinc +reag +REAG reais +realcapital +realchemistry Realiza realizadas Realizem +reallygreatreading realmente reaproveita +reaproveitados +reativar +rebag +rebelliondefense +rebtel +rebuildmanufacturing recebe recebendo receber recente recentemente +recidiviz +Reclame +reclameaqui recoloca Recomenda Recomendações @@ -1444,67 +4610,131 @@ recomendados reconcilia reconecta reconhece +reconhecida +recordedfuture recorre +recriar +recruitingprograms recrutador Recrutadores recrutamento +rectanglehealth recupera Recuperação recuperado recuperar +recursionpharmaceuticals Recurso recursos recusou +redcellpartners +redpartners +redpeak +reduz +reduzindo reduzir +redventures +redwoodmaterials +redwoodsoftware Reescrever refatorado referência +referralsuseonly +refletem +refugeerights +regiao registrado registro registros regra regras Regressão +regscale +reidopitaco reindexação reiniciada reiniciar Reiniciar reino Reino +reinstalar +Reinstalar rejeição rejeita rejeitar relação +Relacionada relança +relationalai +relativas relativo relativos RELATÓRIO Relatórios +relaygraduateschoolofeducation +relaypayments +relaypro +relaytherapeutics relevantes +relishworks relname +reltio +remedyhomehealthcare +Remessa +remessaonline +remixtherapeutics +remodelhealth +remoracarbon remotas +remotasks +remotecom +remotereferralboardinternaluseonly remova removendo removida removidas removido +removidos +renaissancelearning renderização +Renewables +renewedvision +Rennert +renpsg +rentbrella +Rentbrella +rentcars +Rentcars +renttherunway repetidas +repetindo repetitivos +repisodic repositório +repositórios Representação reprocessamento +reproductivefreedomforall +reprofreedomforallinternships repropaga requisições Requisitos +researchpartnership reservados +Resid +residclub +residenthome resiliência +resolvem +resolvetosavelives resolvido +resortpass respeita respeitando respeitar respiro responde +respondemos respondendo respondeu Responsável @@ -1514,10 +4744,12 @@ Responsivo respostas restantes Restaura +restaurantsupply restringir restritivo Resultado resultados +resultsforamerica RESUMO retenção retomar @@ -1535,207 +4767,853 @@ reutilizada reutilizaveis reutilizáveis reutilizável +revero revisão Revisar +revivn +revloncorporate +rewardsnetwork +rexfordindustrial +Rhitmo +rhitmotech +rhombuspower +rhythmx +rialtic +ribolifamilywines +ridgeline +rimestechnologies +ringtherapeutics +riogaleao riotgames +ripcpc risco +rithum +rithumliboard +ritterandbrogdenorthodontics +rivainternationalinc +rivaltechnologies +riverai +riversidenaturalfoodsltd +riviamind +robertrauschenbergfoundation robinhood +roboforce +robotsandpencils +robusta robustez +rockbot +rocketlab +rocketlawyer +rocketmiles +rockstargames rodada +rodadas rodam +roivantsciences +rondoenergy +roofr +roofstock rootfs +Rosen +Rossi +rotação rotacionado rotativo +rotativos Roteamento +rothesaygraduates +rothesaylife rslave +rubiconcarbon +ruelala +ruggable +ruggedrobotics +runwise +rushdownstudios +rushstreetinteractive +russett +rvohcontentfreelance +rvohealth +rxsense +rzero SADD +safebreach +safetyworxs +sagebionetworks +saída sairá saiu salario salário +salientmotion +saltxc Salvando salvas +samainc +samayaai +samsungresearchamerica +samsungresearchamericainternship +samsungsemiconductor +sanar +Sanar +Sancor +sancorsegurosbrasil +sandscapitalmanagementllc +sandstonecarecastlerock +saraworks saudáveis +saudável saúde +saxbys +saxllp +sayari +sbigrowth +sbusa scaleai scannerc +schonfeld +schrdinger +schulze +Schulze +Schwarzman +scileads +scopely +scorpionenterprisesllc +scoutai +scoutmotors +scoutspace +scowtt scrapa +scsfinancial +sdcpinternshipprogram +seafireresortltd +seaporttherapeutics +seatown +seattlesoundersfc +seazone +Seazone Seção +secondharvest secreta +secretariatadvisorsllc secreto Secundário +securitize +securityscorecard +segredos Seguem seguindo seguinte segunda segundos segura +Seguradora seguras seguro seguros +segurossura +seisandbox sejam SELECIONADA selecionar seletivos seletor +selffinancial +selinicapital Selo +semafor semáforos Semáforos semanal semanas +semantix +Semantix +sendabiosciences +sendcloudnew sendo Senhas +senhasegura Sênior +sensei +sensiblecare sensível sentido +seoulrobotics separa +separada separadas separados Separando +septerna +sequenciais serem +serhant seria serializa serializar série +sertis servem +servicechampions +serviceexpertsllc +servicewizard serviços +sesai +sesolabor sess sessões setar +setsales +seuestagio +sevenresearch +sezzle +sfox +shakepay +shapedigital +shapercapital +sharebite +sharepeoplehub +sharkninjaoperatingllc +sharpelectronics +shein +shennonbiotechnologies +shieldshealthsolutions +shifttechnology +shinola +shinolaretail +shipbobinc +Shipmanagement +shipmonk +shopltk +shopmy +showpad +shyftsolutionsllc +siboneinc +sicredi +Sicredi +sidecarhealth +sidia +Sidia sido +siei +siem +sierrallc +sightlinemediagroup +sigmacomputing +signerscareers significa +signifyd +silananotechnologies silencia silenciosamente silencioso +silverado +silverfin +Silverfin +Silverstein +silvr +silvus +simdigital +similarweb +simplesense +simpletechnologysolutions +simplextrading +simplifed simplificado simplificados +simplisafe +simpluris +simpplr +simtrabps simula simulada simulado simultaneamente simultâneos +sinais sinal +sincronia sincronização Sincronizado Sincronizar síncrono Singapura singleflight +singlestore +Sinonimos +sintaxe +sirenopt +sirum SISMEMBER +sixgeninc +sixspeed +sixthstreet +skedda +skeelo +Skeelo +skildai +Skillbridge +skilledwoundcare +skinlaundry +skyepointdecisionsinc +skylighthq +skyone +Skyone +skyryse +skysafe +Slabstack +sleepdoctor +slingshotaerospace +slingshotbiosciences +smaamerica +smallgirlspr +smartasset +smartbear +smarterdx +smarterdxprivate +smartling +smartlyio +smartrent +smartsheet +smartypantsvitamins +smavagmbh SMEMBERS +smithrx +snapmobileinc +snorkelai +snowcompanies +snsone +sobe sobjects sobrecarregar Sobrescreve +sobrescrevem sobrescrever sobrescrito sobrescritos Sobrevivente sobreviver sobrou +soci sociais +socialfinance +sociallabsa +socialscienceresearchcouncil +softvaro +Softvaro sohohouse +solarisbank +soldejaneiro +soldejaneirointernship +solfacil +Solfácil solicitado solicitar Solicitar +solidpower +sollishealth +solmentalhealth +soloioinc SOLUÇÃO +Soluções +solutis +Solutis somente +sonatus +sonderaustralia +sonicwall +sonobello +sonyinteractiveentertainmentglobal +sonymusicasiacareers +sonymusiccanada +sonymusiccareersafrica +sonymusiccareersfrance +sonymusicentertainment +sonypicturesanimation +sonypicturesimageworks soql +soraunion +sorcero +Soros +Sortation sortedset +sothebys +sotran +Sotran +soundagriculture +soundcloud +sourcegraph +sourcemeridian +southernpovertylawcenter +southwesttitleloans +sovrn sozinhas +spacecorporation +spacekinetic +spacexglobal +sparetech +sparkadvisors +sparkfund +spauldingridge +spcareers +specterops +spectrumvascular +spinnakersupport +splashfinancial +splitero +sportalliance +sportandspinephysicaltherapy +spothopper +spotme +springboardmentors +springfertility +springhealth +sprintersportses +spsnorthamerica +spycloud +spyretherapeutics +Squatch +squishable sslmode +stablekernel +stackadapt +stackav +stackblitz +stackcommerce +stackline +standardmetrics +stannesbelfieldschool +starcloud +starfaceworld +starfishneuroscience +starrez +startale +startcampus +stateroadah +steadfasthealth +stemhealthcare +sterlingtonpllc +stirlingpdf +stockx +stonekite +stonepatrocina +stoque +Stoque +storycannabis +strandtherapeutics +stratacareers +stratainformationgroup +strategichr +strategicprojectpartners +stratolaunch strconv +stressfree stretchr +stri +striiminc stringlib +strivehealth +strivepharmacy +striveworks +strm +Strm +strongpointpartners structmap +stubhubinc +studiokraftonboard +studsinc +studycontractors +stunion +stylusmedicine +Suba subconjunto subfunções subir +subpastas +subsplash substitui +substituídos substituir subtítulo +successacademycharterschool +successfactors +successkpiinc +sudstop suficiente sufixo +sugerido sugeridos sujar sumário sumir sumisse sumiu +summitonevanderbilt +summittherapeutics +sumofus +sumologic +sumup +sunne +Sunne +sunnyside +Suno +suntimes supabase +superfrete +supergoop +superlogica +Superlógica +supersod suportar Suporte +supplyhouse +supportingstrategies +surefirecyber +survata +surveymonkey +sustainabletalent +sustainablewestchester +svetness +swanloveland +swave +Swave +swellmedia +swiftsolar +sylvamo +Sylvamo +sympla +Sympla +synack +synacksrt +synaptrixlabs +sypnewsitetest +syskahennessy +systemstechnologyresearch tabela tabelas Tabelas tablelib +tactilemedical +takealotcom +takealotgroup +taketwo talentos +talentt +Talentt +talentx +talkdesk +talkspace +talkspacepsychiatry +talkspacetherapist talvez +tandemlaunch +tandemmoneylimited +tanium +tankww +tapestryenergy +tarefa tarefas +taskrabbit +tastylive +tastytrade +tatari +taxbit +taxvalet +tazewellpikeanimalclinic +teachinglab +teads +teague +teamlfg +teammobot +teampicnic +teamrubicon +tebra +tecer +Techfin +techholding +techlead +Techlead +techstars técnica Técnicas técnico Tecnologias +tecovas +tecsul +Tecsul +tefron +tegnainc +tegra +Tegra +tekion +tekmetric Tela telas +telavita +Telavita +Telefônica +Telematics teletrabalho +Telligent +telnyx tema +tembici +Tembici temos +temporaltechnologies temporária temporariamente temporárias Temporário +temus +tenableinc +teneolinkedin tenha +tennesseetitleloansinc +tenstorrent +tenstorrentuniversity +tenstreet tenta tentativa tentativas terá +teravision +termina terminar termo +terraclear +terranorbitalcorporation +terzo +tesseratherapeutics testall testamos +testar teste +testlio +testnisc testuser +texasairsystems +texascartitleandpaydayloanservicesinc +texaschillersystemsasp texto textuais +textus +thalamusgme +thanx +thatlot +thatsnomoonentertainment +thealleninstitute +thebaltimorebanner +thebrattlegroup +thechempetitivegroupllc +thedoulanetwork +thedutchie +theeconomistgroup +theeverycompany +thefarmersdog +thefloridapanthers +thefork +thegialliancemanagementllccompany +thehealthmanagementacademy +thehutgroup +theiconic +theknotworldwide +theloomisagency +themaritimeaquarium +thematherevanston +themichaeljfoxfoundation +themjcos +themotleyfool themuse +themuseumofscience +thena +thenewyorktimes +thenuclearcompany +theoncologyinstitute +theorchard +theplaceforchildrenwithautism +theplanningshopus +thesciongroupllc +thesiscareers +thesocialhub +thetradedesk +thevirtussolution +thevitacococompany +theweathercompany +thewolvescompany +thinkacademymy +thinkacademyus +thinkingmachines +thinkmarkets +thirdlove +thirdwaveautomation +thlightrebuild +thomasdoor +thomasvillechildcare +thoropass +thoughtworksreferral +threatlocker +thrivemarket +thtbc +thunderstecnologia +tiendanube +tigera +tigergraph tigra +tinnova +Tinnova +tintai tipado tipados tipagem +tippingpointcommunity titulo tiver +tlatechinc +toastmastersinternational tocar +togetherai tolerância +tollbit +tomofunfurbo +tomorrowhealth +toogoodtogo +toojaysdeli topo +topsort +topsteptrader +torcrobotics +tornar tornou +toroinvestimentos +torq +toshibaglobalcommercesolutions +totalrecon +totusmedicines +totvs +TOTVS +towerresearchcapital +townsq +tpcengineeringholdingsllc +tpgcareers +tpgroup +TPGROUP +tpreducationllc +trabalhar +tradelink +traegergrills tráfego Tráfego +trampay +Trampay transação +transactlyconnect +transcendinc +transcendtherapeutics +transfergo +transfero +Transfero transição +transitórios +transmarketgroup +trás +trase trata +tratados tratam tratamento +tratar travar +traveledgenetwork traz trazer +treasuryprime +trecho +Treehouse +treelinebiosciences três Triagem triar +tribalscale +trigild trilhas trimpath +trinca +Trinca +trinityparktalent +trinnus +Trinnus +Trinuscool +tripla +Tripla +tripledotstudios +triplewhale +triumvirateenvironmental +trivelta +triviumpoint trocas +trovohealth +trueanomalyinc +truebill +truecaller +truemedia +trufflesecurity trunca truncado +trustautomation +trustbank +trustwill +truveta +tsugu +Tsugu +ttcglobal +tubescience +tubitv +tucows TUDO +tudorgroup +turbi +Turbi +turbineone +turbotenant +TURMA +turnkeycareers +twinhealth +twistbioscience +twosixtechnologies typeform +tysonmendesllp +uareai +uasi +uberfreight +ubiquiti +ubiquitygp +ubots +Ubots +udacity +udemybedi +uello +Uello Uguwk +uhdfnvbkldfnbhrpkdfgbdvtyhro +uinlinepromotions uintptr +ujet ulimits última +ultimagenomics últimas últimos ultrapassa ultrapassar +umaeducationinc +umistone +unanet +unbounce +undercontrolroboticsinc +underdogfantasy +understoodcare +unframe única unicas únicas @@ -1745,81 +5623,400 @@ unido Unido unidos Unidos +unimar +Unimar +uninter +UNINTER +unionit +unisonhomeownershipinvestors +unispace +uniswapfoundation unitários +unitech +Unitech +unitedfirm +unitedmasterstranslation +unitedmedia +uniteus unitytechnologies +unknownworlds +unlockhealth unpic +unrealsnacks Unrk unstub +unybrands +upbound +upboundext +upda +UPDA +upflux +upriteconstruction +upshop +upstack +upstatementrecruiting +upstreamusa +upwork +urbansky +urbansportsclub +urlkey +urlscan +Urlscan +URLSCAN +ursamajor usada usadas usados usam usaremos +usarmos +usconec usem usemos +usenourish usernametest usertest +utahtitleloansinc úteis útil utilitárias utilitário +utilitários utilizando +vacasa +vacationinc +vaco vaga +vagasbyintera +vagasconfidenciais +vailhealthprivate +valaratomics válida +validações validada validado +validados Validamos +Validando validar validos válidos valkey Valkey +valohealth +valpro +valtech vamos +vanmetre +vannevarlabs +vantagescore +vardaspace variam variar várias variável +varicent varsitytutors +vaticlabs +vatto +Vatto +vaxcyte vazia VAZIAS vazio vazios +vectara vectorset +veeamsoftware veem +vegaamericas +veir +Veja +velocityelectronics +venncity +ventureglobal +venturus +Venturus +veocorporatecareers +veracyte +verainstituteofjustice +veranahealth +veratherapeuticsinc vercel verdade Verificações verificamos +veristainc +veritasvetpartners +verkada +verramobility +versaterm versionada versionado +versionados +versionar versões +versprite +verstela +verticalrh +vestmark +vestwell +veterinaryemergencygroupst +veterinaryemergencyservices +veterinarypracticepartners +vetevolve vetorial +vettoai vezes +vhsys viagem +viamrobotics +vianttechnology +vibesllc vierem +vikingglobalinvestors vinda vindas vindo +vinta +Vinta +vipvermontinformationprocessing vira +viralnation +virbiotechnologyinc +virginbetsa vírgula +virtru Visão +VISASQ +viseai +viseu +Viseu +visia +visiersolutionsinc visitantes +visitingmedia visuais +visualconcepts Visualização visualizadas visualizar Visualizem +vitalvoicesglobalpartnership +vitest +vitoriainternational +vitru +Vitru +vivcourtevents +vivopcd +vivvi +vixtra +Vixtra +vixxo +vmax +vmlenterprisesolutions +vobi +Vobi +vocedm +vockan +Vockan +vogliodigitalmarketing +volastratherapeutics +voltada Voltar +Voltta +volttaenergy +vonage +vooban +vorbiopharma +vortx +Vórtx +votorantim +Votorantim +voxmedia +voyagertechnologiesinc +voyagertherapeutics +vpawashington +vrental +vsapartners +vscfiresecurityinc +vsco Vueh +vuejs +vulcanelements +vulncheck +vwgds +vynamic +vynyl +wakam +waldensecurity +wallapop +walleyecapital +wallstreetprep +wargamingen +waterloocoop +waverlyadvisorsllc +wavin +Wavin +wayback +Wayback +WAYBACK +waymark +waymo wealthfront +webershandwick +webflow +wedgewoodpharmacy +weedmaps +weee +wehandle weightsbiases +weinsteinproperties +weissassetmanagement +welbehealth +Wellhub +wellist +wellsaidlabs +wellthy +wellz +Wellz +weploy +westcancercenter +westernacher +Westernacher +westmonroe +westwing +Westwing +wettermarkkeith +wevy +Wevy +wfclainc +whalarinc +wheely +whisperaero +whitewatermidstream +whogivesacrap +wildalaskancompany +wildcardcreativegroup +wildlifestudios +willbank +wilsonelser +wilsonelserattorneys +wimanagementllc +windcraft +WINDCRAFT +winhomeinspection +winnerscirclegroupoftexas +winnin +Winnin +winstaller +wisconsinautotitleloansinc +wisetack +withcoverage +withmeinc +wizardcommerce +wizixtechnologygroupinc +wolt +wonderschool +wongdoody +wooga +woolpert +workatbackbase +workato +workera +workhelix +workithealth +workleap +workleapfr +workoverseas +workrightnw +workstream +Workwear +workwize +worldlabs +worldquant +wovencare +wppmedia +wrike writerc +wsut +wundercapital +wwclinic +wypoon +Wypoon +xairatherapeutics +xantium +xapo +Xchange +xdinizioengage +xealth +xebiaapac +xebiacee +xebiausa +xendit +xenergyinc xerrors +xgenomics +xohealthinc +xometry +xometryeurope +xometryturkey xpack +xpengmotors +xpinc +xsolisinc +xtxmarketstechnologies xxhash +yalochatinc yamlh yamlprivateh +yandeh +Yandeh +yerbamadre +yhgbhfg +yipitdata +yipitdatajobs +ylopo +yoodliinc +youcom +yugabyte +yurtsai +zafinlabsamericasinc +zallpy +Zallpy +zambold +zappts +Zappts +zarminalihealth +zeffy zelândia Zelândia +zenbusiness +zencoder +zengrc +zennioptical +zenoti +zephyrhome zerados +zerezes +Zerezes +zerostudios +zerotothree +zetachain +zetaglobal +zinnov +zipcolimited +ziprecruiter +ziro ZLDS +zocalohealth +zocdoc +zonecompanysoftwareconsultingllc +zora +zubiad +Zuckerberg +zuora +zupinnovation +Zwirner +zyngacareers +zyngaearlycareers diff --git a/.env.example b/.env.example index c7cf023..cde3e25 100644 --- a/.env.example +++ b/.env.example @@ -24,7 +24,7 @@ CORS_ALLOWED_ORIGINS=http://localhost:5173,http://localhost:5174 SEARCH_LOCATION=Brasil SEARCH_GEO_ID=106057199 SEARCH_LANGUAGE=pt -REMOTE_ONLY=true +REMOTE_ONLY=false JOB_TYPES=C,F TIME_FILTER=r604800 SEARCH_KEYWORDS=UX Designer,UI Designer,Product Manager,Product Owner @@ -69,6 +69,24 @@ EMAIL_QUEUE_ATTEMPTS=3 ADZUNA_APP_ID= ADZUNA_APP_KEY= JOOBLE_API_KEY= +LINKEDIN_KEYWORD_SLOT_SIZE=30 +ADZUNA_KEYWORD_SLOT_SIZE=30 +GUPY_ENABLED=true +GUPY_RAW_DISCOVERY_ENABLED=true +GUPY_FULL_SWEEP_ENABLED=true +GUPY_FULL_REMOTE_SWEEP_ENABLED=true +GUPY_QUERY_LIMIT=60 +INHIRE_ENABLED=true +INHIRE_TENANTS_FILE=./internal/interfaces/inhireTenants.json +INHIRE_ENRICH_DETAILS=false +INHIRE_DETAILS_MODE=ambiguous +INHIRE_DETAILS_CONCURRENCY=8 +INHIRE_DETAILS_TIMEOUT_MS=10000 +GREENHOUSE_ENABLED=true +GREENHOUSE_COMPANIES_FILE=./internal/interfaces/greenhouseCompanies.json +LEVER_ENABLED=false +LEVER_COMPANIES_FILE=./internal/interfaces/leverCompanies.json +LEVER_INCLUDE_ALL_JOBS=true # OAuth credentials GOOGLE_CLIENT_ID= diff --git a/.gitignore b/.gitignore index 11c070b..eb27250 100644 --- a/.gitignore +++ b/.gitignore @@ -9,6 +9,8 @@ dist/ build/ output/ backend/output/ +data/ +tools/ npm-debug.log yarn-error.log yarn-debug.log diff --git a/BACKEND.md b/BACKEND.md index b2eb227..126b0de 100644 --- a/BACKEND.md +++ b/BACKEND.md @@ -63,7 +63,7 @@ Módulos principais: - `src/modules/users` — perfis e preferências do usuário (`UsersController`, `UsersService`). - `src/modules/savedJobs` — CRUD de vagas salvas (`SavedJobsController`, `SavedJobsService`). - `src/modules/notifications` — notificações do usuário autenticado. -- `src/modules/jobs` — regras de matching/score de vagas. +- `src/modules/jobs` — busca, parsing de filtros, fallback pós-filtro e regras de matching/score de vagas. - `src/modules/admin` — usuários admin, permissões, scrapers, auditoria, dashboard e observabilidade. Adaptadores externos: @@ -81,8 +81,9 @@ Database / Schemas (Drizzle): Cache & Indexes: -- `src/lib/cache.ts` — helpers para Redis/Valkey; usado por `jobs.routes` para obter ids e buscar vagas em memória. +- `src/lib/cache.ts` — helpers para Redis/Valkey; usado pelo módulo de jobs para obter ids e buscar vagas em memória. - Busca por palavras-chave usa índices invertidos e interseção para eficiência. +- Filtros estruturados podem usar índices por família (`family`), tecnologia (`technology`), senioridade (`seniority`), localização, modelo e contrato. ## Módulo de E-mail @@ -174,6 +175,7 @@ Base: `/` - Jobs - `GET /jobs/search?keywords=...` — busca vagas utilizando índices/Valkey/Redis. Retorna paginação e fonte (`source`). + - Filtros aceitos incluem `keywords`, `family`, `technology`, `seniority`, `level`, `location`, `country`, `state`, `city`, `type`/`model`, `contract`/`contractType`/`jobTypes` e `matchSort`. - Keywords - `GET /keywords` — lista keywords persistidas no banco. @@ -247,6 +249,7 @@ Definidas/consumidas em `src/config.ts` e outros módulos: - `goScraper.ts` faz POST em `${GO_SCRAPER_URL}/scrape` com `ScrapeParams` e valida `ScrapeResponse`. - `goKeywords.ts` consulta e publica keywords via endpoints do serviço Go (`/api/keywords`). +- O backend lê os índices criados pelo scraper no Valkey, incluindo `scraper:jobs:keyword:*`, `scraper:jobs:family:*`, `scraper:jobs:technology:*` e `scraper:jobs:seniority:*`. ## Banco de dados diff --git a/README.md b/README.md index 0fe7ce1..1b81131 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -# Painel de Vagas +# > Novo no projeto? Comece por aqui: [LOCAL_DEVELOPMENT.md](LOCAL_DEVELOPMENT.md) @@ -7,15 +7,26 @@ ![Monorepo](https://img.shields.io/badge/architecture-monorepo-0A66C2) ![License ISC](https://img.shields.io/badge/license-ISC-lightgrey) +**Documentação**: +[README](README.md) | +[SCRAPER](SCRAPER.md) | +[BACKEND](BACKEND.md) | +[TESTING](TESTING.md) | +[CONTRIBUTING](contribuition.md) | +[ESCOPO](ESCOPO.md) | +[Frontend](frontend/README.md) | +[Frontend Architecture](frontend/ARCHITECTURE.md) | +[Front Admin](front_admin/README.md) + Plataforma de captura, agregação e consulta de vagas com arquitetura monorepo, composta por frontend web, API Node.js, scraper Go, painel administrativo e aplicação desktop com Electron. O produto evoluiu para um modelo orientado a serviços (API + scraper Go + cache/índices), com autenticação, preferências de usuário e integração com banco de dados. ## Links oficiais -- Gestão de produto (Linear): https://linear.app/tatame/team/PAV/all +- Gestão de produto (Linear): - Guia do usuário — Linear e integração com GitHub: [GUIA-LINEAR-GITHUB.md](GUIA-LINEAR-GITHUB.md) -- Design system oficial (Figma): https://www.figma.com/design/gollJBtK8PGkffNN4zk9t9/Painel-Dev---releitura?node-id=0-1&p=f&t=zU8zrFzPsNPxZ3qU-0 +- Design system oficial (Figma): - Documentação backend detalhada: [BACKEND.md](BACKEND.md) - Documentação scraper Go: [SCRAPER.md](SCRAPER.md) - Guia de testes: [TESTING.md](TESTING.md) @@ -87,13 +98,13 @@ Objetivo de produto: fornecer uma base robusta para busca, filtragem e gestão d Use este fluxo para subir Postgres, Valkey, scraper Go, backend, frontend e front_admin com a mesma rede Docker. -1. Instale as dependências locais: +1 - Instale as dependências locais: ```bash npm install ``` -2. Crie o `.env` da raiz a partir do exemplo versionado: +2 - Crie o `.env` da raiz a partir do exemplo versionado: ```bash cp .env.example .env @@ -105,7 +116,7 @@ No Windows PowerShell: Copy-Item .env.example .env ``` -3. Crie a rede Docker compartilhada, se ela ainda não existir: +3 - Crie a rede Docker compartilhada, se ela ainda não existir: ```bash docker network create vagas-net @@ -113,19 +124,19 @@ docker network create vagas-net Se a rede já existir, o Docker vai avisar e você pode seguir para o próximo passo. -4. Suba Postgres, Valkey, scraper, backend, frontend e front_admin: +4 - Suba Postgres, Valkey, scraper, backend, frontend e front_admin: ```bash docker compose -f docker-compose.infra.yml -f docker-compose.yml -f docker-compose.migrate.yml up --build -d ``` -5. Acesse os serviços: +5 - Acesse os serviços: -- Frontend: http://localhost:5173 -- Front admin: http://localhost:5174 -- Backend health: http://localhost:3001/health -- Scraper health: http://localhost:8081/health -- Vagas salvas no scraper: http://localhost:8081/admin/jobs/count +- Frontend: +- Front admin: +- Backend health: +- Scraper health: +- Vagas salvas no scraper: ### Desenvolvimento com Node local @@ -272,6 +283,7 @@ Usuários: Jobs: - GET /jobs/search + - Busca vagas nos índices do Valkey e aceita filtros por `keywords`, `family`, `technology`, `seniority`, `level`, `location`, `country`, `type`/`model`, `contract` e ordenação por `matchSort`. Keywords: @@ -360,17 +372,17 @@ Por isso o `docker-compose.yml` e o `docker-compose.migrate.yml` sobrescrevem va Serviços padrão: -- Frontend: http://localhost:5173 -- Front admin: http://localhost:5174 -- Backend: http://localhost:3001 -- Scraper Go: http://localhost:8081 +- Frontend: +- Front admin: +- Backend: +- Scraper Go: Endpoints úteis do scraper: -- Health: http://localhost:8081/health -- Status da execução: http://localhost:8081/admin/scrape/status -- Contagem de vagas salvas: http://localhost:8081/admin/jobs/count -- Lista de vagas salvas: http://localhost:8081/admin/jobs +- Health: +- Status da execução: +- Contagem de vagas salvas: +- Lista de vagas salvas: ## Desktop com Electron @@ -431,6 +443,17 @@ Variáveis centrais de operação: - WAIT_BETWEEN_SEARCHES_MS - PAGE_TIMEOUT_MS - MAX_PAGES_PER_KEYWORD +- LINKEDIN_KEYWORD_SLOT_SIZE +- ADZUNA_KEYWORD_SLOT_SIZE +- GUPY_ENABLED +- GUPY_RAW_DISCOVERY_ENABLED +- GUPY_FULL_SWEEP_ENABLED +- GUPY_FULL_REMOTE_SWEEP_ENABLED +- GUPY_QUERY_LIMIT +- INHIRE_ENABLED +- INHIRE_ENRICH_DETAILS +- GREENHOUSE_ENABLED +- LEVER_ENABLED - CACHE_TTL_MS - VITE_API_BASE_URL - VITE_API_URL @@ -443,6 +466,11 @@ Segurança operacional: - Preferir acesso interno para banco/cache em VPS. - Em ambiente externo, usar TLS para conexões de dados sempre que possível. +Observação operacional do scraper: + +- O padrão atual privilegia resposta rápida e estabilidade. LinkedIn e Adzuna usam slots rotativos de keywords por execução, e Gupy limita a quantidade de queries expandidas por rodada. Isso reduz cobertura imediata por execução, mas evita milhares de requests em uma única rodada e distribui a coleta ao longo das próximas execuções. +- `tools/` e `data/` são ignorados pelo Git. Scripts versionados não devem depender de arquivos nessas pastas, a menos que sejam tratados como utilitários locais opcionais. + ## Testes e qualidade Estrutura: @@ -489,7 +517,7 @@ Observação: o painel admin já possui testes e build próprios, mas ainda deve Padrão oficial: -1. Abrir card no Linear: https://linear.app/tatame/team/PAV/all +1. Abrir card no Linear: 2. Criar branch de feature a partir de master 3. Desenvolver e testar localmente 4. Abrir PR da feature para develop @@ -499,7 +527,7 @@ Padrão oficial: Convenção recomendada de branch: -- feature/- +- feature/{id-do-card}-{descricao-curta} ## Git Hooks e qualidade local @@ -529,8 +557,6 @@ Observação importante: - CI continua obrigatório e independente de hooks locais. Mesmo com bypass local, o pipeline valida cobertura/lint/build antes de merge. - - ## Roadmap técnico sugerido ### DX e onboarding @@ -539,7 +565,6 @@ Observação importante: - Adicionar verificação automática de scripts quebrados no CI. - Padronizar comandos cross-platform (evitar dependência de sintaxe de variável de ambiente Unix em scripts críticos). - ### Segurança - Aplicar política de rotação de SESSION_SECRET e credenciais OAuth. diff --git a/SCRAPER.md b/SCRAPER.md index 26d9a29..75025cb 100644 --- a/SCRAPER.md +++ b/SCRAPER.md @@ -4,6 +4,8 @@ Este documento descreve o scraper implementado em Go localizado em `scraper-go/` ## Visão geral +O scraper Go concentra a coleta e normalização de vagas. Ele recebe configurações de busca do backend, executa fontes externas habilitadas, aplica deduplicação e classificação local, persiste vagas no Valkey e publica índices para a API consultar com baixa latência. + ## Exemplos por adaptador Abaixo há exemplos simplificados do payload esperado internamente e de como cada adaptador normalmente formata/retorna vagas. @@ -28,6 +30,7 @@ Abaixo há exemplos simplificados do payload esperado internamente e de como cad - Adzuna - Comportamento: usa API oficial (quando `ADZUNA_APP_ID`/`ADZUNA_APP_KEY` configurados) e retorna JSON com campos estruturados. + - Operação: suporta `SearchBatch` com slot rotativo de keywords e limite padrão de 5 páginas por keyword. - Exemplo (Job adaptado): ```json @@ -46,6 +49,7 @@ Abaixo há exemplos simplificados do payload esperado internamente e de como cad - Jooble - Comportamento: integra com Jooble API quando `JOOBLE_API_KEY` presente; pode usar Redis para controle de cota. + - Operação: usa slot rotativo, cota diária e cadência de 12h para proteger a integração. - Exemplo (Job adaptado): ```json @@ -80,7 +84,38 @@ Abaixo há exemplos simplificados do payload esperado internamente e de como cad } ``` -Esses exemplos ilustram o contrato interno entre adaptadores e pipeline: o pipeline espera `models.Job` com campos normalizados (URL limpa, título/empresa/local, `Source` e `StableID` calculável). O `jobstore.StableID` deriva o ID a partir de título+empresa+local ou URL. +### Descobrir tokens do Greenhouse + +O `board_token` é o trecho final da URL pública da Greenhouse. Por exemplo: + +- `https://job-boards.greenhouse.io/reddit` → token `reddit` +- `https://job-boards.greenhouse.io/gitlab` → token `gitlab` + +Para validar tokens ou testar nomes de empresas, use: + +```bash +cd scraper-go +go run ./cmd/greenhouse-discover -names Reddit,GitLab +``` + +Saída esperada: + +```text +OK gitlab 186 vagas https://job-boards.greenhouse.io/gitlab +OK reddit 195 vagas https://job-boards.greenhouse.io/reddit +``` + +Também é possível validar o JSON atual: + +```bash +cd scraper-go +go run ./cmd/greenhouse-discover -file internal/interfaces/greenhouseCompanies.json +``` + +Tokens com `OK` podem entrar em `internal/interfaces/greenhouseCompanies.json`. +Tokens com `MISS status=404` devem ser removidos ou substituídos. + +Esses exemplos ilustram o contrato interno entre adaptadores e pipeline: o pipeline espera `domain.Job` com campos normalizados (URL limpa, título/empresa/local, `Source` e `StableID` calculável). O `jobstore.StableID` deriva o ID a partir de título+empresa+local ou URL. ## Especificação OpenAPI (local) @@ -95,8 +130,12 @@ O scraper é um serviço HTTP em Go que consulta múltiplas fontes de vagas (Lin Componentes principais: - `cmd/server` — inicializador e ponto de entrada. -- `internal/adapters` — coleções de adaptadores por fonte que implementam a interface `Adapter`. -- `internal/pipeline` — core do pipeline de scraping, orquestra execução concorrente e indexação. +- `internal/domain` — modelos centrais do scraper (`Job`, `ScrapeRequest`, `ScrapeResponse`, classificação). +- `internal/ports` — contratos internos da aplicação, como fontes de vagas, repositórios, cache e métricas. +- `internal/adapters` — composição de adapters concretos e registry das fontes habilitadas. +- `internal/adapters/` — implementação isolada de cada fonte externa (`gupy`, `inhire`, `greenhouse`, `lever`, `jooble`, `linkedin`, `themuse`, `adzuna`). +- `internal/adapters/adapterutil` — helpers compartilhados entre adapters concretos. +- `internal/pipeline` — camada de aplicação do pipeline de scraping, orquestra execução concorrente e indexação. - `internal/jobstore` — persistência em Redis (valkey) para jobs, índices e IDs estáveis. - `internal/cache` — abstração de cache com implementação Redis e memória (fallback). - `internal/dedup` — regras para deduplicação/merge de vagas. @@ -104,6 +143,34 @@ Componentes principais: - `internal/inflight` — deduplicador de requisições concorrentes (singleflight). - `internal/cronjob` — scheduler de scraping em background e execução manual. - `internal/metrics` — métricas Prometheus por fonte/execução. +- `cmd/greenhouse-discover` — utilitário Go para validar tokens de empresas Greenhouse antes de atualizar `internal/interfaces/greenhouseCompanies.json`. + +## Arquitetura atual + +O scraper evolui para uma arquitetura hexagonal de forma incremental. O domínio fica em `internal/domain`, as portas em `internal/ports` e os adapters concretos ficam em subpastas de `internal/adapters`. + +A fronteira principal é a porta `ports.JobSource`: cada fonte externa implementa `SourceName`, `Search` e, opcionalmente, `SearchBatch`. O pipeline recebe uma lista de `ports.JobSource` já montada pelo servidor/registry, evitando que a camada de aplicação conheça diretamente as implementações concretas. + +Desenho atual: + +```text +cmd/server + monta cache, Valkey, stores, scheduler e adapters + +internal/domain + modelos centrais do scraper + +internal/ports + contratos que a aplicação consome + +internal/pipeline + orquestra scraping, dedupe, classificação e indexação + +internal/adapters + registry e implementações concretas por fonte +``` + +Ainda há pontos a evoluir: `pipeline` e `cronjob` continuam recebendo `*redis.Client` em fluxos de indexação/cadência, e métricas Prometheus ainda são chamadas diretamente. Os próximos passos naturais são criar adapters outbound para Valkey e Prometheus por trás de portas específicas, reduzindo ainda mais o acoplamento de infraestrutura. ## Como executar @@ -126,7 +193,7 @@ go run ./cmd/server Docker: há um `Dockerfile` em `scraper-go/`. No Docker Compose, configure `VALKEY_URL=redis://valkey:6379/0` no `.env` da raiz para que o scraper acesse o Valkey pelo nome do serviço na rede Docker. -No Compose da raiz, o serviço escuta em http://localhost:8081. +No Compose da raiz, o serviço escuta em . ## Endpoints HTTP @@ -193,10 +260,11 @@ Fluxo principal: 1. Recebe `ScrapeRequest` com keywords e configuração. 2. Verifica cache (`internal/cache`). Se encontrado, retorna resultado cacheado. 3. Caso contrário, executa `pipeline.ScrapeAllSources` que: - - Constrói lista de adaptadores (`adapters.GetAdapters`). - - Cria tarefas (adapter × keyword) e executa concorrente com limite (`MaxConcurrency`). - - Cada adaptador realiza requisições HTTP específicas, parseia HTML/JSON com `goquery` e retorna `models.Job`. + - Recebe a lista de fontes já montada pelo servidor/registry. + - Cria uma tarefa por fonte batch (`SearchBatch`) ou uma tarefa por keyword para fontes sem batch, sempre respeitando `MaxConcurrency`. + - Cada adaptador realiza requisições HTTP específicas, parseia HTML/JSON quando necessário e retorna `domain.Job`. - Agrega resultados e aplica deduplicação (`dedup.DedupeJobs`). + - Classifica vagas por família, tecnologias e senioridade antes da indexação. - Persiste vagas novas no `jobstore` (Redis) e atualiza índices invertidos (função `IndexJobsInValkey`). - Escreve resultado no cache para próximas requisições. @@ -205,16 +273,28 @@ Concorrência e resiliência: - Semáforos por adaptador (ex.: LinkedIn usa um semáforo de 5 simultâneos para proteção). - Tratamento de status 429 com backoff; aborta apenas a keyword afetada em caso de falhas persistentes. - Uso de `inflight` para evitar que múltiplas requisições idênticas disparem scrapes simultâneos. +- Slots rotativos reduzem o número de keywords/queries por rodada em fontes caras, preservando cobertura progressiva em execuções futuras. ## Adaptadores -Cada adaptador em `internal/adapters` implementa a interface `Adapter` com `SourceName()` e `Search(ctx, keyword, req)`. +Cada adaptador em `internal/adapters/` implementa a porta `ports.JobSource` com `SourceName()` e `Search(ctx, keyword, req)`. Quando a fonte consegue buscar várias keywords em uma chamada/lote, ela também pode implementar `ports.BatchJobSource`. Implementações incluem: -- `linkedin.go` — busca via endpoint público `jobs-guest` do LinkedIn; parsing com `goquery`. -- `adzuna.go` — integra via API Adzuna (se configurada com `ADZUNA_APP_ID`/`ADZUNA_APP_KEY`). -- `jooble.go` — integra com Jooble API (se `JOOBLE_API_KEY` configurada); pode usar Redis para quota. -- `greenhouse.go`, `lever.go`, `themuse.go` — adaptadores por empresa/plataforma com parsing/integração próprios. +- `internal/adapters/linkedin` — busca via endpoint público `jobs-guest` do LinkedIn; parsing com `goquery`. +- `internal/adapters/adzuna` — integra via API Adzuna (se configurada com `ADZUNA_APP_ID`/`ADZUNA_APP_KEY`). +- `internal/adapters/jooble` — integra com Jooble API (se `JOOBLE_API_KEY` configurada); pode usar Redis para quota. +- `internal/adapters/greenhouse`, `internal/adapters/lever`, `internal/adapters/themuse` — adaptadores por empresa/plataforma com parsing/integração próprios. +- `internal/adapters/gupy`, `internal/adapters/inhire` — adaptadores para fontes usadas no fluxo atual de vagas. + +### Estratégia por fonte + +- LinkedIn: sempre habilitado; `SearchBatch` seleciona um slot rotativo de keywords por execução. Defaults atuais: 5 páginas por keyword e 30 keywords por rodada. +- Adzuna: habilitado quando `ADZUNA_APP_ID` e `ADZUNA_APP_KEY` existem; `SearchBatch` usa slot rotativo. Defaults atuais: 5 páginas por keyword e 30 keywords por rodada. +- Gupy: habilitado com `GUPY_ENABLED=true`; usa queries expandidas, descoberta por termos amplos e sweep opcional. O default atual limita o sweep a offset 10000 e processa 60 queries por rodada. +- Jooble: habilitado com `JOOBLE_API_KEY`; usa cota diária, slot rotativo e cadência de 12h. +- Greenhouse: habilitado com `GREENHOUSE_ENABLED=true`; cria um adapter por empresa listada em `internal/interfaces/greenhouseCompanies.json`. +- Lever: habilitado com `LEVER_ENABLED=true`; cria adapters a partir de `internal/interfaces/leverCompanies.json`. +- InHire: habilitado com `INHIRE_ENABLED=true`; consulta tenants de `internal/interfaces/inhireTenants.json` e só enriquece detalhes quando `INHIRE_ENRICH_DETAILS=true`. Boas práticas nos adaptadores: @@ -226,6 +306,8 @@ Boas práticas nos adaptadores: - `jobstore.SaveBatch` persiste vagas no Redis com TTL e mantém um índice global (`scraper:jobs:index`). - `pipeline.IndexJobsInValkey` cria índices invertidos por keyword e sub-termos (`scraper:jobs:keyword:`), além de manter TTL para índices. +- A classificação local também gera índices estruturados: `scraper:jobs:family:`, `scraper:jobs:technology:` e `scraper:jobs:seniority:`. +- Filtros estruturados de localização, modelo, contrato e senioridade continuam em chaves como `scraper:jobs:country:`, `scraper:jobs:model:` e `scraper:jobs:contract:`. - `jobstore.StableID` garante IDs determinísticos para permitir identificação e deduplicação entre execuções. ## Cache e configuração @@ -243,14 +325,19 @@ Boas práticas nos adaptadores: - `VALKEY_URL` — conexão Redis/Valkey. Em Docker Compose, use `redis://valkey:6379/0`; em execução local fora do Docker, use uma URL acessível pelo host, por exemplo `redis://localhost:6379/0`. - `JOOBLE_API_KEY` — Jooble integration. - `ADZUNA_APP_ID` / `ADZUNA_APP_KEY` — Adzuna API. -- Configurações de logging e quota podem ser definidas via `.env`. +- `LINKEDIN_KEYWORD_SLOT_SIZE` — quantidade máxima de keywords do LinkedIn por execução quando a busca vier com uma lista grande. Padrão: `30`. +- `ADZUNA_KEYWORD_SLOT_SIZE` — quantidade máxima de keywords do Adzuna por execução. Padrão: `30`. +- `GUPY_ENABLED` — habilita o adapter Gupy. +- `GUPY_RAW_DISCOVERY_ENABLED` — adiciona queries amplas de tecnologia na Gupy. +- `GUPY_FULL_SWEEP_ENABLED` / `GUPY_FULL_REMOTE_SWEEP_ENABLED` — controla sweeps amplos na Gupy. +- `GUPY_QUERY_LIMIT` — limita quantas queries expandidas da Gupy rodam por execução. Padrão: `60`. +- `INHIRE_ENABLED`, `INHIRE_TENANTS_FILE`, `INHIRE_ENRICH_DETAILS`, `INHIRE_DETAILS_MODE`, `INHIRE_DETAILS_CONCURRENCY`, `INHIRE_DETAILS_TIMEOUT_MS` — controlam fonte e enriquecimento InHire. +- `GREENHOUSE_ENABLED`, `GREENHOUSE_COMPANIES_FILE` — controlam fonte Greenhouse. +- `LEVER_ENABLED`, `LEVER_COMPANIES_FILE`, `LEVER_INCLUDE_ALL_JOBS` — controlam fonte Lever. +- Configurações de logging, quota e performance podem ser definidas via `.env`. ## Observações operacionais - Projetado para rodar frequentemente; use caching e indexação para reduzir chamadas repetidas. - Monitorar erros 429 e ajustar `WaitBetweenSearchesMs` / semáforos por adaptador. - Verifique logs estruturados (slog JSON) e `/metrics` para métricas de sucesso/falhas por adaptador. - ---- - -Arquivo gerado: `scraper-go/SCRAPER.md`. diff --git a/backend/bruno/Auth/GitHubAuthUrl.bru b/backend/bruno/Auth/GitHubAuthUrl.bru index 6b22a0d..318b17f 100644 --- a/backend/bruno/Auth/GitHubAuthUrl.bru +++ b/backend/bruno/Auth/GitHubAuthUrl.bru @@ -1,7 +1,7 @@ meta { name: GitHubAuthUrl type: http - seq: 2 + seq: 3 } get { diff --git a/backend/bruno/Auth/GoogleAuthUrl.bru b/backend/bruno/Auth/GoogleAuthUrl.bru index 09193f1..1a1449b 100644 --- a/backend/bruno/Auth/GoogleAuthUrl.bru +++ b/backend/bruno/Auth/GoogleAuthUrl.bru @@ -1,7 +1,7 @@ meta { name: GoogleAuthUrl type: http - seq: 2 + seq: 4 } get { diff --git a/backend/bruno/Auth/Login.bru b/backend/bruno/Auth/LoginLocal.bru similarity index 96% rename from backend/bruno/Auth/Login.bru rename to backend/bruno/Auth/LoginLocal.bru index df12566..2920b26 100644 --- a/backend/bruno/Auth/Login.bru +++ b/backend/bruno/Auth/LoginLocal.bru @@ -1,7 +1,7 @@ meta { - name: Login + name: LoginLocal type: http - seq: 2 + seq: 6 } post { diff --git a/backend/bruno/Auth/LoginServer.bru b/backend/bruno/Auth/LoginServer.bru new file mode 100644 index 0000000..66734a8 --- /dev/null +++ b/backend/bruno/Auth/LoginServer.bru @@ -0,0 +1,47 @@ +meta { + name: LoginServer + type: http + seq: 5 +} + +post { + url: {{base_url}}/auth/login + body: json + auth: inherit +} + +body:json { + { + "email":"hudsonlimatavares@gmail.com", + "password":"Hrjm11202330@" + } +} + +tests { + console.log("HEADERS:", res.headers); + + const rawCookie = + res.headers["set-cookie"] || + res.headers["Set-Cookie"]; + + console.log("RAW COOKIE:", rawCookie); + + if (!rawCookie) { + throw new Error("Set-Cookie não encontrado"); + } + + const cookie = Array.isArray(rawCookie) + ? rawCookie[0] + : rawCookie; + + const sessionCookie = cookie.split(";")[0]; + + bru.setEnvVar("sessionCookie", sessionCookie); + + console.log("SESSION COOKIE:", sessionCookie); +} + +settings { + encodeUrl: true + timeout: 0 +} diff --git a/backend/bruno/Auth/Register.bru b/backend/bruno/Auth/Register.bru index 392d77f..ada8efa 100644 --- a/backend/bruno/Auth/Register.bru +++ b/backend/bruno/Auth/Register.bru @@ -1,7 +1,7 @@ meta { name: Register type: http - seq: 2 + seq: 7 } post { diff --git a/backend/bruno/Jobs/SearchJobs.bru b/backend/bruno/Jobs/SearchJobs.bru index 5e6440d..21d8f1a 100644 --- a/backend/bruno/Jobs/SearchJobs.bru +++ b/backend/bruno/Jobs/SearchJobs.bru @@ -5,15 +5,16 @@ meta { } get { - url: {{base_url}}/jobs/search?location=Brasil&level=Junior + url: {{base_url}}/jobs/search?location=Brasil&level=pleno&keywords=node&modality=remota body: none auth: inherit } params:query { location: Brasil - level: Junior - ~keywords: + level: pleno + keywords: node + modality: remota } headers { diff --git a/backend/bruno/environments/Local.bru b/backend/bruno/environments/Local.bru index e2cc283..3c724c8 100644 --- a/backend/bruno/environments/Local.bru +++ b/backend/bruno/environments/Local.bru @@ -1,4 +1,5 @@ vars { base_url: http://localhost:3001 - sessionCookie: vagas_session=Fe26.2*1*6b1ee4abcbe0ef8e42e8203e5e9110ead280c442ba3c9e1bf78d0c1f24fb5f5d*7c9X00fVX_sg3MhS9HZe6w*TPGS8-0VdhkyeqTAGGkuJB-4mBhf8ybA4wEB5RrO0S-tC9ykpwRVP8y1b1T8cQFYOPsm7Q-9dccgg92xnaMswi9GoU3Q55N1flcZHo_r7PmLsF8EsadRSraRo5TQOjG2isq6B6UavP_zf_Zr5jkpbA*1780502414077*3727896ffed5c4dd607f5d8460989c13de4218813e5d6263848dc882ae4a9027*Q7YEyyEcQbXbMzCdy4gyJU4RBccUkj6GKw5WW8JhzmE~2 + sessionCookie: vagas_session=Fe26.2*1*83cec28903717d08da08d9b4285ca53ca1e2667af4e9bbd48961f65f682567f0*uBiRKQHRIWWf4HjUzVYgLw*D-SdRa9Lwl4x53IFIZnGHE5k59Kr5cUK4sfydjpR9VgckYuyGzUEqgQhOn72rZM5iBZXyKb5DwwtwsZXvFH0ww*1786625038265*18af4e7b76c9620666f699dd4d852ff239aefa6a5f0b707ca927acbba66301c4*Se2BA8jdSjwdmpbcTGGQWrXUWdeUc2FDhrOLEQNxMEA~2 + ~base_url: https://api.candidate.app.br } diff --git a/backend/src/config.ts b/backend/src/config.ts index 7856822..59eb0ab 100644 --- a/backend/src/config.ts +++ b/backend/src/config.ts @@ -65,7 +65,7 @@ export function getConfig(): AppConfig { searchLocation: process.env.SEARCH_LOCATION ?? "Brasil", searchGeoId: process.env.SEARCH_GEO_ID ?? "106057199", searchLanguage: process.env.SEARCH_LANGUAGE ?? "pt", - remoteOnly: parseBoolean(process.env.REMOTE_ONLY, true), + remoteOnly: parseBoolean(process.env.REMOTE_ONLY, false), jobTypes: process.env.JOB_TYPES ?? "C,F", timeFilter: parseTimeFilter(process.env.TIME_FILTER, "r604800"), databaseUrl: process.env.DATABASE_URL?.trim() ?? "", diff --git a/backend/src/db/schema/userPreferences.ts b/backend/src/db/schema/userPreferences.ts index da8878c..56b12d5 100644 --- a/backend/src/db/schema/userPreferences.ts +++ b/backend/src/db/schema/userPreferences.ts @@ -23,7 +23,7 @@ export const userPreferences = pgTable("user_preferences", { searchLocation: text("search_location"), searchLanguage: varchar("search_language", { length: 10 }), - remoteOnly: boolean("remote_only").default(true), + remoteOnly: boolean("remote_only").default(false), jobTypes: text("job_types").array().default([]), emailNotifications: boolean("email_notifications").default(false), diff --git a/backend/src/lib/cache.ts b/backend/src/lib/cache.ts index 088f360..e3109dd 100644 --- a/backend/src/lib/cache.ts +++ b/backend/src/lib/cache.ts @@ -1,6 +1,7 @@ import { randomUUID } from "node:crypto"; import { createClient, type RedisClientType } from "redis"; import { logger } from "../logger"; +import { cacheOperationsTotal } from "../metrics/metrics"; export const TTL = { PROFILE: 60 * 60, // 1 hora @@ -46,15 +47,29 @@ function key(suffix: string): string { return `${NS}${suffix}`; } +function recordCacheOperation(operation: string, result: string): void { + cacheOperationsTotal.inc({ operation, result }); +} + export async function cacheGet(suffix: string): Promise { const client = await getCache(); - const raw = await client.get(key(suffix)); - if (!raw) return null; - try { - return JSON.parse(raw) as T; - } catch { - return raw as unknown as T; + const raw = await client.get(key(suffix)); + if (!raw) { + recordCacheOperation("get", "miss"); + return null; + } + + recordCacheOperation("get", "hit"); + + try { + return JSON.parse(raw) as T; + } catch { + return raw as unknown as T; + } + } catch (error) { + recordCacheOperation("get", "error"); + throw error; } } @@ -66,16 +81,28 @@ export async function cacheSet( const client = await getCache(); const serialized = typeof value === "string" ? value : JSON.stringify(value); - if (ttlSeconds > 0) { - await client.set(key(suffix), serialized, { EX: ttlSeconds }); - } else { - await client.set(key(suffix), serialized); + try { + if (ttlSeconds > 0) { + await client.set(key(suffix), serialized, { EX: ttlSeconds }); + } else { + await client.set(key(suffix), serialized); + } + recordCacheOperation("set", "ok"); + } catch (error) { + recordCacheOperation("set", "error"); + throw error; } } export async function cacheDel(suffix: string): Promise { const client = await getCache(); - await client.del(key(suffix)); + try { + await client.del(key(suffix)); + recordCacheOperation("delete", "ok"); + } catch (error) { + recordCacheOperation("delete", "error"); + throw error; + } } export async function invalidateUser(userId: string): Promise { @@ -92,12 +119,26 @@ export async function cacheAbsoluteSMembers( absoluteKey: string, ): Promise { const client = await getCache(); - return await client.sMembers(absoluteKey); + try { + const result = await client.sMembers(absoluteKey); + recordCacheOperation("smembers", result.length > 0 ? "hit" : "miss"); + return result; + } catch (error) { + recordCacheOperation("smembers", "error"); + throw error; + } } export async function cacheAbsoluteSCard(absoluteKey: string): Promise { const client = await getCache(); - return await client.sCard(absoluteKey); + try { + const result = await client.sCard(absoluteKey); + recordCacheOperation("scard", "ok"); + return result; + } catch (error) { + recordCacheOperation("scard", "error"); + throw error; + } } /** @@ -115,11 +156,20 @@ export async function cacheSearchKeywords( const keys = keywordSearchKeys(keywords); - if (keys.length === 0) return []; - if (keys.length === 1) return await client.sMembers(keys[0]); + if (keys.length === 0) { + recordCacheOperation("search_keywords", "miss"); + return []; + } - // SUNION → vagas que têm QUALQUER uma das keywords (OU) - return await client.sUnion(keys); + try { + const result = + keys.length === 1 ? await client.sMembers(keys[0]) : await client.sUnion(keys); + recordCacheOperation("search_keywords", result.length > 0 ? "hit" : "miss"); + return result; + } catch (error) { + recordCacheOperation("search_keywords", "error"); + throw error; + } } function normalizeIndexValue(value: string): string { @@ -180,7 +230,11 @@ function keywordIndexKeyVariants(keyword: string): string[] { if (term) variants.add(term); } - return [...variants].map((value) => `scraper:jobs:keyword:${value}`); + return [...variants].flatMap((value) => [ + `scraper:jobs:keyword:${value}`, + `scraper:jobs:technology:${value}`, + `scraper:jobs:family:${value}`, + ]); } function keywordSearchKeys(keywords: string[]): string[] { @@ -191,6 +245,9 @@ function keywordSearchKeys(keywords: string[]): string[] { export type CacheJobIndexFilters = { keywords?: string[]; + family?: string | string[]; + technology?: string | string[]; + seniority?: string; level?: string; location?: string; continent?: string; @@ -226,6 +283,9 @@ function cacheJobIndexKey(kind: string, value: string): string { function cacheJobIndexKeyGroups(filters: CacheJobIndexFilters): string[][] { const entries: Array<[string, string | string[] | undefined]> = [ + ["family", filters.family], + ["technology", filters.technology], + ["seniority", filters.seniority], ["level", filters.level], ["location", filters.location], ["continent", filters.continent], @@ -269,19 +329,30 @@ export async function cacheSearchJobIds( ); try { + let result: string[]; if (keywordKeys.length === 0 && filterKeys.length === 0) { - return await client.sMembers("scraper:jobs:index"); + result = await client.sMembers("scraper:jobs:index"); + recordCacheOperation("search_jobs", result.length > 0 ? "hit" : "miss"); + return result; } if (keywordKeys.length === 0) { - if (filterKeys.length === 1) return await client.sMembers(filterKeys[0]); - return (await client.sendCommand(["SINTER", ...filterKeys])) as string[]; + result = + filterKeys.length === 1 + ? await client.sMembers(filterKeys[0]) + : ((await client.sendCommand(["SINTER", ...filterKeys])) as string[]); + recordCacheOperation("search_jobs", result.length > 0 ? "hit" : "miss"); + return result; } if (keywordKeys.length === 1) { const keys = [keywordKeys[0], ...filterKeys]; - if (keys.length === 1) return await client.sMembers(keys[0]); - return (await client.sendCommand(["SINTER", ...keys])) as string[]; + result = + keys.length === 1 + ? await client.sMembers(keys[0]) + : ((await client.sendCommand(["SINTER", ...keys])) as string[]); + recordCacheOperation("search_jobs", result.length > 0 ? "hit" : "miss"); + return result; } const tempKey = `scraper:jobs:search:${randomUUID()}`; @@ -291,8 +362,15 @@ export async function cacheSearchJobIds( await client.expire(tempKey, 30); const keys = [tempKey, ...filterKeys]; - if (keys.length === 1) return await client.sMembers(keys[0]); - return (await client.sendCommand(["SINTER", ...keys])) as string[]; + result = + keys.length === 1 + ? await client.sMembers(keys[0]) + : ((await client.sendCommand(["SINTER", ...keys])) as string[]); + recordCacheOperation("search_jobs", result.length > 0 ? "hit" : "miss"); + return result; + } catch (error) { + recordCacheOperation("search_jobs", "error"); + throw error; } finally { await Promise.all(tempKeys.map((key) => client.del(key))); } @@ -304,9 +382,15 @@ export async function cacheGetJobsByIds(ids: string[]): Promise { if (ids.length === 0) return []; const keys = ids.map((id) => `scraper:job:${id}`); - const results = await client.mGet(keys); + let results: Array; + try { + results = await client.mGet(keys); + } catch (error) { + recordCacheOperation("mget_jobs", "error"); + throw error; + } - return results + const jobs = results .filter((raw): raw is string => raw !== null) .map((raw) => { try { @@ -316,6 +400,9 @@ export async function cacheGetJobsByIds(ids: string[]): Promise { } }) .filter(Boolean); + + recordCacheOperation("mget_jobs", jobs.length > 0 ? "hit" : "miss"); + return jobs; } async function cacheDeleteByPattern(pattern: string): Promise { @@ -359,5 +446,12 @@ export async function cacheClearJobs(): Promise<{ export async function cachePing(): Promise { const client = await getCache(); - return await client.ping(); + try { + const result = await client.ping(); + recordCacheOperation("ping", "ok"); + return result; + } catch (error) { + recordCacheOperation("ping", "error"); + throw error; + } } diff --git a/backend/src/modules/admin/dashboard/dashboard.service.ts b/backend/src/modules/admin/dashboard/dashboard.service.ts index 34948fa..ea169dd 100644 --- a/backend/src/modules/admin/dashboard/dashboard.service.ts +++ b/backend/src/modules/admin/dashboard/dashboard.service.ts @@ -91,7 +91,7 @@ export class DashboardService { status: status.running ? "running" : "idle", running: status.running, lastRunAt: status.lastRunAt ?? null, - jobsCollected: status.jobsCollected ?? count?.total ?? null, + jobsCollected: count?.total ?? status.jobsCollected ?? null, }; } catch { return { diff --git a/backend/src/modules/admin/observability/metrics.service.ts b/backend/src/modules/admin/observability/metrics.service.ts index f79d48b..a494d85 100644 --- a/backend/src/modules/admin/observability/metrics.service.ts +++ b/backend/src/modules/admin/observability/metrics.service.ts @@ -26,6 +26,7 @@ type PrometheusRangeResult = { type DashboardPanelDefinition = Omit & { expr: string; + instantExpr?: string; legend?: (metric: Record) => string; }; @@ -125,6 +126,21 @@ async function queryPrometheusRange({ } } +async function instantSeries( + expr: string, + legend?: DashboardPanelDefinition["legend"], +): Promise { + const value = await queryPrometheus(expr); + if (value === null) return []; + + return [ + { + label: legend?.({}) ?? "atual", + points: [{ timestamp: new Date().toISOString(), value }], + }, + ]; +} + function toSeries( result: PrometheusRangeResult["data"]["result"], legend?: DashboardPanelDefinition["legend"], @@ -184,7 +200,7 @@ const DASHBOARDS: Array< title: "Cache hit rate", unit: "percent", visualization: "line", - expr: 'sum(rate(cache_operations_total{result="hit"}[5m])) / sum(rate(cache_operations_total[5m])) * 100', + expr: 'sum(rate(cache_operations_total{result="hit"}[5m])) / clamp_min(sum(rate(cache_operations_total[5m])), 1) * 100', legend: () => "hit rate", }, ], @@ -252,6 +268,7 @@ const DASHBOARDS: Array< unit: "count", visualization: "stat", expr: "sum(increase(job_searches_total[24h]))", + instantExpr: "sum(increase(job_searches_total[24h])) or vector(0)", legend: () => "buscas", }, { @@ -267,7 +284,9 @@ const DASHBOARDS: Array< title: "Cache hit rate", unit: "percent", visualization: "stat", - expr: 'sum(rate(cache_operations_total{result="hit"}[24h])) / sum(rate(cache_operations_total[24h])) * 100', + expr: 'sum(rate(cache_operations_total{result="hit"}[24h])) / clamp_min(sum(rate(cache_operations_total[24h])), 1) * 100', + instantExpr: + 'sum(rate(cache_operations_total{result="hit"}[24h])) / clamp_min(sum(rate(cache_operations_total[24h])), 1) * 100 or vector(0)', legend: () => "hit rate", }, { @@ -398,13 +417,18 @@ export class MetricsService { DASHBOARDS.map(async (dashboard) => ({ ...dashboard, panels: await Promise.all( - dashboard.panels.map(async ({ expr, legend, ...panel }) => ({ - ...panel, - series: toSeries( - await queryPrometheusRange({ expr, range, step }), - legend, - ), - })), + dashboard.panels.map( + async ({ expr, instantExpr, legend, ...panel }) => ({ + ...panel, + series: + panel.visualization === "stat" && instantExpr + ? await instantSeries(instantExpr, legend) + : toSeries( + await queryPrometheusRange({ expr, range, step }), + legend, + ), + }), + ), ), })), ); diff --git a/backend/src/modules/admin/scrapers/scrapers.controller.ts b/backend/src/modules/admin/scrapers/scrapers.controller.ts index e78c655..a37af29 100644 --- a/backend/src/modules/admin/scrapers/scrapers.controller.ts +++ b/backend/src/modules/admin/scrapers/scrapers.controller.ts @@ -25,6 +25,35 @@ export class ScrapersController { } } + async triggerOne(req: Request, res: Response): Promise { + try { + const scraperName = String(req.params.id ?? ""); + const result = await this.scrapersService.triggerScraper(scraperName); + + this.auditService.fromRequest(req, "scrapers.trigger", { + type: "scrapers", + id: scraperName, + }); + + res.status(202).json({ ...result, scraper: scraperName }); + } catch (error) { + if (error instanceof ScraperAlreadyRunningError) { + res.status(409).json({ ok: false, message: error.message }); + return; + } + + const message = + error instanceof Error && error.message.startsWith("scraper desconhecido") + ? error.message + : "erro ao iniciar scraper"; + + res.status(message.startsWith("scraper desconhecido") ? 404 : 500).json({ + ok: false, + message, + }); + } + } + async status(req: Request, res: Response): Promise { try { const result = await this.scrapersService.getStatus(); diff --git a/backend/src/modules/admin/scrapers/scrapers.service.ts b/backend/src/modules/admin/scrapers/scrapers.service.ts index d37d35f..657eb02 100644 --- a/backend/src/modules/admin/scrapers/scrapers.service.ts +++ b/backend/src/modules/admin/scrapers/scrapers.service.ts @@ -21,6 +21,15 @@ export class ScrapersService { } } + async triggerScraper(name: string): Promise { + const normalized = name.trim().toLowerCase(); + if (normalized !== "go-scraper") { + throw new Error(`scraper desconhecido: ${name}`); + } + + return this.triggerScrape(); + } + async getStatus(): Promise { return scraperClient.getStatus(); } @@ -35,7 +44,7 @@ export class ScrapersService { status: status.running ? "running" : "idle", running: status.running, lastRunAt: status.lastRunAt ?? null, - jobsCollected: status.jobsCollected ?? count?.total ?? null, + jobsCollected: count?.total ?? status.jobsCollected ?? null, }, ]; } diff --git a/backend/src/modules/jobs/controllers/searchJobs.controller.ts b/backend/src/modules/jobs/controllers/searchJobs.controller.ts new file mode 100644 index 0000000..751f990 --- /dev/null +++ b/backend/src/modules/jobs/controllers/searchJobs.controller.ts @@ -0,0 +1,41 @@ +import type { NextFunction, Request, Response } from "express"; +import { logWarn } from "../../../logger"; +import { jobSearchesTotal } from "../../../metrics/metrics"; +import { searchJobsService } from "../services/searchJobs.service"; + +function hasKeywords(query: Request["query"]): boolean { + const value = query.keywords; + const values = Array.isArray(value) ? value : [value]; + + return values.some((item) => { + if (typeof item !== "string") return false; + return item + .split(",") + .some((keyword) => keyword.trim().length > 0); + }); +} + +export async function searchJobsController( + req: Request, + res: Response, + _next: NextFunction, +): Promise { + jobSearchesTotal.inc({ has_keywords: hasKeywords(req.query) ? "true" : "false" }); + + try { + const result = await searchJobsService.execute({ + query: req.query, + userId: req.session?.userId, + }); + + res.json(result); + } catch (error) { + logWarn("Erro ao buscar vagas no ecossistema Valkey", { + error: (error as Error).message, + }); + res.status(500).json({ + message: "Erro ao recuperar vagas em memória.", + error: (error as Error).message, + }); + } +} diff --git a/backend/src/modules/jobs/filters/jobSearch.filter.ts b/backend/src/modules/jobs/filters/jobSearch.filter.ts new file mode 100644 index 0000000..cc29d8d --- /dev/null +++ b/backend/src/modules/jobs/filters/jobSearch.filter.ts @@ -0,0 +1,257 @@ +import type { + ParsedJobSearchQuery, + SearchJob, +} from "../types/jobSearch.types"; + +export function normalizeComparable(value: string): string { + return value + .normalize("NFD") + .replace(/[\u0300-\u036f]/g, "") + .toLowerCase() + .replace(/[^\p{L}\p{N}]+/gu, " ") + .replace(/\s+/g, " ") + .trim(); +} + +function normalizeLevelFilter(value: string): string { + const normalized = normalizeComparable(value); + if ( + normalized === "estagio trainee" || + normalized === "estagio" || + normalized === "trainee" || + normalized === "intern" || + normalized === "internship" + ) { + return "estagio"; + } + return normalized; +} + +function containsTokenOrPhrase(text: string, needle: string): boolean { + if (needle.includes(" ")) return text.includes(needle); + return ` ${text} `.includes(` ${needle} `); +} + +function containsAny(text: string, needles: string[]): boolean { + return needles.some((needle) => containsTokenOrPhrase(text, needle)); +} + +function inferJobLevel(title: string): string { + const normalized = normalizeComparable(title); + if ( + containsAny(normalized, [ + "estagio", + "estagiario", + "intern", + "internship", + "trainee", + "aprendiz", + ]) + ) { + return "estagio"; + } + if ( + containsAny(normalized, [ + "senior", + "sr", + "especialista", + "lead", + "principal", + "staff", + ]) + ) { + return "senior"; + } + if (containsAny(normalized, ["junior", "jr", "entry level", "assistente"])) { + return "junior"; + } + return "pleno"; +} + +function inferJobType(job: SearchJob): string { + const normalized = normalizeComparable( + [job.title, job.location, job.modality, job.description] + .filter(Boolean) + .join(" "), + ); + + if (normalized.includes("hibrid") || normalized.includes("hybrid")) { + return "hibrido"; + } + if ( + normalized.includes("remot") || + normalized.includes("home office") || + normalized.includes("teletrabalho") || + normalized.includes("anywhere") || + normalized.includes("worldwide") + ) { + return "remoto"; + } + if ( + normalized.includes("presencial") || + normalized.includes("onsite") || + normalized.includes("on site") || + normalized.includes("on-site") || + normalized.includes("in office") || + normalized.includes("escritorio") + ) { + return "presencial"; + } + + return "presencial"; +} + +function inferLocationCountry(location: string): string { + const normalized = normalizeComparable(location); + if (!normalized) return ""; + + if ( + containsAny(normalized, [ + "estados unidos", + "united states", + "usa", + "eua", + "florida", + "miami", + "new york", + "california", + "texas", + "boston", + "seattle", + "chicago", + "atlanta", + "denver", + ]) + ) { + return "estados unidos"; + } + + if ( + containsAny(normalized, [ + "brasil", + "brazil", + "sao paulo", + "rio de janeiro", + "minas gerais", + "belo horizonte", + "parana", + "curitiba", + "santa catarina", + "joinville", + "rio grande do sul", + "porto alegre", + "pernambuco", + "recife", + "bahia", + "salvador", + "ceara", + "fortaleza", + "piaui", + "teresina", + ]) + ) { + return "brasil"; + } + + if (containsAny(normalized, ["portugal", "lisboa", "porto"])) { + return "portugal"; + } + + return ""; +} + +function matchesLocationFilter(jobLocation: string, location: string): boolean { + if (!location) return true; + + const normalizedLocation = normalizeComparable(jobLocation); + const inferredCountry = inferLocationCountry(jobLocation); + if (inferredCountry) return inferredCountry === location; + + return normalizedLocation.includes(location); +} + +export function filterJobs( + jobs: unknown[], + filters: ParsedJobSearchQuery, +): unknown[] { + const level = normalizeLevelFilter(filters.level); + const seniority = normalizeLevelFilter(filters.seniority); + const location = normalizeComparable(filters.country || filters.location); + const types = filters.type.map(normalizeComparable); + const families = filters.family.map(normalizeComparable); + const technologies = filters.technology.map(normalizeComparable); + + if ( + !level && + !seniority && + !location && + types.length === 0 && + families.length === 0 && + technologies.length === 0 + ) { + return jobs; + } + + return jobs.filter((job) => { + const candidate = job as SearchJob; + const title = candidate.title ?? ""; + const jobLocation = candidate.location ?? ""; + const classification = candidate.classification; + const classifiedFamilies = [ + classification?.primaryFamily, + ...(classification?.relatedFamilies ?? []), + ] + .filter(Boolean) + .map((value) => normalizeComparable(String(value))); + const classifiedTechnologies = (classification?.technologies ?? []) + .filter(Boolean) + .map((value) => normalizeComparable(String(value))); + const classifiedSeniority = normalizeLevelFilter( + classification?.seniority ?? "", + ); + + const matchesLevel = !level || inferJobLevel(title) === level; + const matchesSeniority = + !seniority || + classifiedSeniority === seniority || + inferJobLevel(title) === seniority; + const matchesLocation = matchesLocationFilter(jobLocation, location); + const matchesType = + types.length === 0 || types.includes(inferJobType(candidate)); + const matchesFamily = + families.length === 0 || + families.some((family) => classifiedFamilies.includes(family)); + const matchesTechnology = + technologies.length === 0 || + technologies.some((technology) => + classifiedTechnologies.includes(technology), + ); + + return ( + matchesLevel && + matchesSeniority && + matchesLocation && + matchesType && + matchesFamily && + matchesTechnology + ); + }); +} + +export function sortJobsByMatch( + jobs: T[], + direction: "asc" | "desc", +): T[] { + return [...jobs].sort((first, second) => { + const firstJob = first as { matchScore?: number | null }; + const secondJob = second as { matchScore?: number | null }; + const firstScore = + typeof firstJob.matchScore === "number" ? firstJob.matchScore : 0; + const secondScore = + typeof secondJob.matchScore === "number" ? secondJob.matchScore : 0; + + return direction === "desc" + ? secondScore - firstScore + : firstScore - secondScore; + }); +} diff --git a/backend/src/modules/jobs/jobMatch.service.ts b/backend/src/modules/jobs/jobMatch.service.ts index 0b2d760..be10321 100644 --- a/backend/src/modules/jobs/jobMatch.service.ts +++ b/backend/src/modules/jobs/jobMatch.service.ts @@ -1,158 +1 @@ -import { SavedJob, User } from "../../db/schema"; -import { toPublicUser } from "../users/users.mapper"; - -type TechnologyExperience = { - name: string; - years: number; -}; - -export type MatchableJob = { - id?: string | null; - title?: string | null; - jobTitle?: string | null; - company?: string | null; - location?: string | null; - modality?: string | null; - type?: string | null; - level?: string | null; - keyword?: string | null; - keywords?: string[] | null; - description?: string | null; - url?: string | null; - [key: string]: unknown; -}; - -export type MatchedJob = MatchableJob & { - matchScore?: number; - matchSource?: "backend_profile"; - matchedTechnologies?: string[]; -}; - -function normalizeMatchText(value: string) { - return value - .normalize("NFD") - .replace(/[\u0300-\u036f]/g, "") - .toLowerCase() - .replace(/[^\p{L}\p{N}]+/gu, " ") - .replace(/\s+/g, " ") - .trim(); -} - -function matchAliases(technology: string) { - const normalized = normalizeMatchText(technology); - const aliases = new Set([normalized]); - - if (normalized.endsWith(" js")) { - aliases.add(normalized.replace(/\s+js$/, "js")); - } - if (normalized.endsWith("js") && normalized.length > 2) { - aliases.add(normalized.replace(/js$/, " js")); - } - - return [...aliases].filter(Boolean); -} - -function textMatchesAlias(text: string, alias: string) { - if (!alias) return false; - if (alias.includes(" ")) return text.includes(alias); - return ` ${text} `.includes(` ${alias} `); -} - -function jobMatchText(job: MatchableJob) { - const rawValues = Object.values(job).flatMap((value) => - Array.isArray(value) ? value : [value], - ); - - return normalizeMatchText( - rawValues.map((value) => (typeof value === "string" ? value : "")).join(" "), - ); -} - -function parseTechnologiesFromUser(user: User): TechnologyExperience[] { - const publicUser = toPublicUser(user); - const experiences = publicUser.technologyExperiences; - - if (Array.isArray(experiences)) { - return experiences - .map((item) => { - if (!item || typeof item !== "object") return null; - const data = item as Record; - const name = typeof data.name === "string" ? data.name.trim() : ""; - const years = typeof data.years === "number" ? data.years : 1; - return name ? { name, years: Math.max(0, years) } : null; - }) - .filter((item): item is TechnologyExperience => Boolean(item)); - } - - return (publicUser.technologies ?? []) - .map((name) => name.trim()) - .filter(Boolean) - .map((name) => ({ name, years: 1 })); -} - -export function getUserMatchTechnologies(user: User | undefined | null) { - if (!user) return []; - return parseTechnologiesFromUser(user); -} - -export function scoreJobWithTechnologies( - job: MatchableJob, - technologies: TechnologyExperience[], -): MatchedJob { - const normalizedTechnologies = [ - ...new Map( - technologies - .filter((technology) => technology.name.trim()) - .map((technology) => [ - normalizeMatchText(technology.name), - { - name: technology.name.trim(), - years: Math.max(0, technology.years), - }, - ]), - ).values(), - ]; - - if (normalizedTechnologies.length === 0) return job; - - const text = jobMatchText(job); - const matchedTechnologies = normalizedTechnologies.filter((technology) => - matchAliases(technology.name).some((alias) => textMatchesAlias(text, alias)), - ); - - const totalWeight = normalizedTechnologies.reduce( - (total, technology) => total + Math.max(1, technology.years), - 0, - ); - const matchedWeight = matchedTechnologies.reduce( - (total, technology) => total + Math.max(1, technology.years), - 0, - ); - const coverage = matchedWeight / totalWeight; - const score = - matchedTechnologies.length === 0 - ? 45 - : Math.min( - 99, - 55 + - Math.round(coverage * 35) + - Math.min(matchedTechnologies.length * 4, 9), - ); - - return { - ...job, - matchScore: score, - matchSource: "backend_profile", - matchedTechnologies: matchedTechnologies.map((item) => item.name), - }; -} - -export function jobNotificationIdentity(job: MatchableJob | SavedJob) { - const url = - "url" in job && typeof job.url === "string" - ? job.url - : "jobLink" in job && typeof job.jobLink === "string" - ? job.jobLink - : ""; - return url.trim() || String(job.id ?? ""); -} +export * from "./services/jobMatch.service"; diff --git a/backend/src/modules/jobs/parsers/jobSearchQuery.parser.ts b/backend/src/modules/jobs/parsers/jobSearchQuery.parser.ts new file mode 100644 index 0000000..e5efa06 --- /dev/null +++ b/backend/src/modules/jobs/parsers/jobSearchQuery.parser.ts @@ -0,0 +1,65 @@ +import type { Request } from "express"; +import type { ParsedJobSearchQuery } from "../types/jobSearch.types"; + +export function firstQueryValue(value: unknown): string { + if (Array.isArray(value)) return firstQueryValue(value[0]); + return typeof value === "string" ? value.trim() : ""; +} + +export function queryValues(value: unknown): string[] { + const values = Array.isArray(value) ? value : [value]; + + return values + .flatMap((item) => (typeof item === "string" ? item.split(",") : [])) + .map((item) => item.trim()) + .filter(Boolean); +} + +export function parseJobSearchQuery( + query: Request["query"], +): ParsedJobSearchQuery { + const matchSortValue = + firstQueryValue(query.matchSort) || firstQueryValue(query.sort); + const type = + queryValues(query.model).length > 0 + ? queryValues(query.model) + : queryValues(query.type); + + return { + keywords: queryValues(query.keywords), + family: queryValues(query.family), + technology: queryValues(query.technology), + type, + level: firstQueryValue(query.level), + seniority: firstQueryValue(query.seniority), + location: firstQueryValue(query.location), + continent: firstQueryValue(query.continent), + country: firstQueryValue(query.country), + state: firstQueryValue(query.state), + city: firstQueryValue(query.city), + contract: + firstQueryValue(query.contract) || + firstQueryValue(query.contractType) || + firstQueryValue(query.jobTypes), + matchSort: + matchSortValue === "asc" || matchSortValue === "desc" + ? matchSortValue + : null, + }; +} + +export function hasStructuredFilters(filters: ParsedJobSearchQuery): boolean { + return Boolean( + filters.level || + filters.location || + filters.country || + filters.continent || + filters.state || + filters.city || + filters.family.length > 0 || + filters.technology.length > 0 || + filters.seniority || + filters.type.length > 0 || + filters.contract, + ); +} diff --git a/backend/src/modules/jobs/services/jobMatch.service.ts b/backend/src/modules/jobs/services/jobMatch.service.ts new file mode 100644 index 0000000..587a847 --- /dev/null +++ b/backend/src/modules/jobs/services/jobMatch.service.ts @@ -0,0 +1,162 @@ +import { SavedJob, User } from "../../../db/schema"; +import { toPublicUser } from "../../users/users.mapper"; + +export type TechnologyExperience = { + name: string; + years: number; +}; + +export type MatchableJob = { + id?: string | null; + title?: string | null; + jobTitle?: string | null; + company?: string | null; + location?: string | null; + modality?: string | null; + type?: string | null; + level?: string | null; + keyword?: string | null; + keywords?: string[] | null; + description?: string | null; + url?: string | null; + [key: string]: unknown; +}; + +export type MatchedJob = MatchableJob & { + matchScore?: number; + matchSource?: "backend_profile"; + matchedTechnologies?: string[]; +}; + +function normalizeMatchText(value: string) { + return value + .normalize("NFD") + .replace(/[\u0300-\u036f]/g, "") + .toLowerCase() + .replace(/[^\p{L}\p{N}]+/gu, " ") + .replace(/\s+/g, " ") + .trim(); +} + +function matchAliases(technology: string) { + const normalized = normalizeMatchText(technology); + const aliases = new Set([normalized]); + + if (normalized.endsWith(" js")) { + aliases.add(normalized.replace(/\s+js$/, "js")); + } + if (normalized.endsWith("js") && normalized.length > 2) { + aliases.add(normalized.replace(/js$/, " js")); + } + + return [...aliases].filter(Boolean); +} + +function textMatchesAlias(text: string, alias: string) { + if (!alias) return false; + if (alias.includes(" ")) return text.includes(alias); + return ` ${text} `.includes(` ${alias} `); +} + +function jobMatchText(job: MatchableJob) { + const rawValues = Object.values(job).flatMap((value) => + Array.isArray(value) ? value : [value], + ); + + return normalizeMatchText( + rawValues + .map((value) => (typeof value === "string" ? value : "")) + .join(" "), + ); +} + +function parseTechnologiesFromUser(user: User): TechnologyExperience[] { + const publicUser = toPublicUser(user); + const experiences = publicUser.technologyExperiences; + + if (Array.isArray(experiences)) { + return experiences + .map((item) => { + if (!item || typeof item !== "object") return null; + const data = item as Record; + const name = typeof data.name === "string" ? data.name.trim() : ""; + const years = typeof data.years === "number" ? data.years : 1; + return name ? { name, years: Math.max(0, years) } : null; + }) + .filter((item): item is TechnologyExperience => Boolean(item)); + } + + return (publicUser.technologies ?? []) + .map((name) => name.trim()) + .filter(Boolean) + .map((name) => ({ name, years: 1 })); +} + +export function getUserMatchTechnologies(user: User | undefined | null) { + if (!user) return []; + return parseTechnologiesFromUser(user); +} + +export function scoreJobWithTechnologies( + job: MatchableJob, + technologies: TechnologyExperience[], +): MatchedJob { + const normalizedTechnologies = [ + ...new Map( + technologies + .filter((technology) => technology.name.trim()) + .map((technology) => [ + normalizeMatchText(technology.name), + { + name: technology.name.trim(), + years: Math.max(0, technology.years), + }, + ]), + ).values(), + ]; + + if (normalizedTechnologies.length === 0) return job; + + const text = jobMatchText(job); + const matchedTechnologies = normalizedTechnologies.filter((technology) => + matchAliases(technology.name).some((alias) => + textMatchesAlias(text, alias), + ), + ); + + const totalWeight = normalizedTechnologies.reduce( + (total, technology) => total + Math.max(1, technology.years), + 0, + ); + const matchedWeight = matchedTechnologies.reduce( + (total, technology) => total + Math.max(1, technology.years), + 0, + ); + const coverage = matchedWeight / totalWeight; + const score = + matchedTechnologies.length === 0 + ? 45 + : Math.min( + 99, + 55 + + Math.round(coverage * 35) + + Math.min(matchedTechnologies.length * 4, 9), + ); + + return { + ...job, + matchScore: score, + matchSource: "backend_profile", + matchedTechnologies: matchedTechnologies.map((item) => item.name), + }; +} + +export function jobNotificationIdentity(job: MatchableJob | SavedJob) { + const url = + "url" in job && typeof job.url === "string" + ? job.url + : "jobLink" in job && typeof job.jobLink === "string" + ? job.jobLink + : ""; + return url.trim() || String(job.id ?? ""); +} diff --git a/backend/src/modules/jobs/services/jobProfileMatch.service.ts b/backend/src/modules/jobs/services/jobProfileMatch.service.ts new file mode 100644 index 0000000..a8c4793 --- /dev/null +++ b/backend/src/modules/jobs/services/jobProfileMatch.service.ts @@ -0,0 +1,72 @@ +import { logWarn } from "../../../logger"; +import { NotificationsService } from "../../notifications/notifications.service"; +import { UsersService } from "../../users/users.service"; +import { MatchTechnology } from "../types/jobSearch.types"; +import { + getUserMatchTechnologies, + MatchableJob, + MatchedJob, + scoreJobWithTechnologies, +} from "./jobMatch.service"; + +export class JobProfileMatchService { + async getUserTechnologies(userId?: string): Promise { + if (!userId) return []; + + try { + const user = await new UsersService().getUserById(userId); + return getUserMatchTechnologies(user); + } catch (error) { + logWarn("Não foi possível carregar perfil para cálculo de match", { + userId, + error: (error as Error).message, + }); + + return []; + } + } + + async enrich( + userId: string | undefined, + jobs: MatchableJob[], + technologies: MatchTechnology[], + options: { notifyHighMatches?: boolean } = {}, + ): Promise { + if (technologies.length === 0) { + return jobs as MatchedJob[]; + } + + const matchedJobs = jobs.map((job) => + scoreJobWithTechnologies(job, technologies), + ); + + if (options.notifyHighMatches !== false) { + await this.notifyHighMatches(userId, matchedJobs); + } + + return matchedJobs; + } + + private async notifyHighMatches( + userId: string | undefined, + jobs: MatchedJob[], + ): Promise { + if (!userId) return; + + const notifications = new NotificationsService(); + + await Promise.all( + jobs + .filter((job) => (job.matchScore ?? 0) >= 85) + .map((job) => + notifications.createHighMatchIfMissing(userId, job).catch((error) => { + logWarn("Não foi possível registrar notificação de alto match", { + error: (error as Error).message, + userId, + job: job.title ?? job.jobTitle ?? job.id, + }); + }), + ), + ); + } +} diff --git a/backend/src/modules/jobs/services/searchJobs.service.ts b/backend/src/modules/jobs/services/searchJobs.service.ts new file mode 100644 index 0000000..da910d2 --- /dev/null +++ b/backend/src/modules/jobs/services/searchJobs.service.ts @@ -0,0 +1,202 @@ +import { + cacheAbsoluteSMembers, + cacheGetJobsByIds, + cacheSearchJobIds, + cacheSearchKeywords, +} from "../../../lib/cache"; +import { paginate, parsePagination } from "../../../lib/pagination"; +import { filterJobs, sortJobsByMatch } from "../filters/jobSearch.filter"; +import { + hasStructuredFilters, + parseJobSearchQuery, +} from "../parsers/jobSearchQuery.parser"; +import type { SearchJobsInput, SearchJobsResult } from "../types/jobSearch.types"; +import type { MatchableJob } from "./jobMatch.service"; +import { JobProfileMatchService } from "./jobProfileMatch.service"; + +async function legacyResolveIds( + keywords: string[], +): Promise<{ ids: string[]; source: string }> { + if (keywords.length > 0) { + return { + ids: await cacheSearchKeywords(keywords), + source: `valkey_filtered_by_keywords:${keywords.join("+")}`, + }; + } + + return { + ids: await cacheAbsoluteSMembers("scraper:jobs:index"), + source: "valkey_global_index", + }; +} + +function toSearchResult( + jobs: unknown[], + pagination: ReturnType["pagination"], + source: string, +): SearchJobsResult { + return { + total: pagination.total, + page: pagination.page, + limit: pagination.limit, + totalPages: pagination.totalPages, + hasNext: pagination.hasNext, + hasPrev: pagination.hasPrev, + jobs, + source, + }; +} + +export class SearchJobsService { + constructor( + private readonly profileMatchService = new JobProfileMatchService(), + ) {} + + async execute(input: SearchJobsInput): Promise { + const filters = parseJobSearchQuery(input.query); + const pagination = parsePagination(input.query); + const hasFilters = hasStructuredFilters(filters); + const matchTechnologies = + await this.profileMatchService.getUserTechnologies(input.userId); + + let ids: string[] = []; + let source = + filters.keywords.length > 0 + ? `valkey_filtered_by_keywords:${filters.keywords.join("+")}` + : "valkey_global_index"; + + if (hasFilters) { + ids = await cacheSearchJobIds({ + keywords: filters.keywords, + family: filters.family, + technology: filters.technology, + seniority: filters.seniority, + level: filters.level, + location: filters.location, + continent: filters.continent, + country: filters.country, + state: filters.state, + city: filters.city, + type: filters.type, + model: filters.type, + contract: filters.contract, + }); + source = `${source}:structured_indexes`; + + if (ids.length === 0) { + return await this.searchWithPostFilterFallback( + filters, + pagination, + matchTechnologies, + input.userId, + `${source}:legacy_post_filter_fallback`, + ); + } + + const indexedJobs = await cacheGetJobsByIds(ids); + const filteredJobs = filterJobs(indexedJobs, filters); + return await this.paginateFilteredJobs( + filteredJobs, + filters.matchSort, + pagination, + matchTechnologies, + input.userId, + `${source}:verified`, + ); + } + + const legacy = await legacyResolveIds(filters.keywords); + ids = legacy.ids; + source = legacy.source; + + if (filters.matchSort) { + const allJobs = await cacheGetJobsByIds(ids); + const matchedJobs = await this.profileMatchService.enrich( + input.userId, + allJobs as MatchableJob[], + matchTechnologies, + { notifyHighMatches: false }, + ); + const sortedJobs = sortJobsByMatch(matchedJobs, filters.matchSort); + const { data: jobs, pagination: meta } = paginate(sortedJobs, pagination); + await this.profileMatchService.enrich( + input.userId, + jobs as MatchableJob[], + matchTechnologies, + ); + + return toSearchResult(jobs, meta, `${source}:match_sorted_${filters.matchSort}`); + } + + const { data: pageIds, pagination: meta } = paginate(ids, pagination); + const pageJobs = await cacheGetJobsByIds(pageIds); + const jobs = await this.profileMatchService.enrich( + input.userId, + pageJobs as MatchableJob[], + matchTechnologies, + ); + + return toSearchResult(jobs, meta, source); + } + + private async searchWithPostFilterFallback( + filters: ReturnType, + pagination: ReturnType, + matchTechnologies: Parameters[2], + userId: string | undefined, + source: string, + ): Promise { + const legacy = await legacyResolveIds(filters.keywords); + const legacyJobs = await cacheGetJobsByIds(legacy.ids); + const filteredJobs = filterJobs(legacyJobs, filters); + + return await this.paginateFilteredJobs( + filteredJobs, + filters.matchSort, + pagination, + matchTechnologies, + userId, + source, + ); + } + + private async paginateFilteredJobs( + jobs: unknown[], + matchSort: "asc" | "desc" | null, + pagination: ReturnType, + matchTechnologies: Parameters[2], + userId: string | undefined, + source: string, + ): Promise { + if (matchSort) { + const matchedJobs = await this.profileMatchService.enrich( + userId, + jobs as MatchableJob[], + matchTechnologies, + { notifyHighMatches: false }, + ); + const { data: pageJobs, pagination: meta } = paginate( + sortJobsByMatch(matchedJobs, matchSort), + pagination, + ); + await this.profileMatchService.enrich( + userId, + pageJobs as MatchableJob[], + matchTechnologies, + ); + + return toSearchResult(pageJobs, meta, source); + } + + const { data: pageJobs, pagination: meta } = paginate(jobs, pagination); + const enrichedJobs = await this.profileMatchService.enrich( + userId, + pageJobs as MatchableJob[], + matchTechnologies, + ); + + return toSearchResult(enrichedJobs, meta, source); + } +} + +export const searchJobsService = new SearchJobsService(); diff --git a/backend/src/modules/jobs/types/jobSearch.types.ts b/backend/src/modules/jobs/types/jobSearch.types.ts new file mode 100644 index 0000000..866ad0f --- /dev/null +++ b/backend/src/modules/jobs/types/jobSearch.types.ts @@ -0,0 +1,54 @@ +import type { Request } from "express"; +import type { TechnologyExperience } from "../services/jobMatch.service"; + +export type SearchJob = { + id?: string; + title?: string | null; + jobTitle?: string | null; + location?: string | null; + modality?: string | null; + description?: string | null; + matchScore?: number | null; + classification?: { + primaryFamily?: string | null; + relatedFamilies?: string[] | null; + technologies?: string[] | null; + seniority?: string | null; + } | null; +}; + +export type MatchTechnology = TechnologyExperience; + +export type MatchSort = "asc" | "desc" | null; + +export type ParsedJobSearchQuery = { + keywords: string[]; + family: string[]; + technology: string[]; + type: string[]; + level: string; + seniority: string; + location: string; + continent: string; + country: string; + state: string; + city: string; + contract: string; + matchSort: MatchSort; +}; + +export type SearchJobsInput = { + query: Request["query"]; + userId?: string; +}; + +export type SearchJobsResult = { + total: number; + page: number; + limit: number; + totalPages: number; + hasNext: boolean; + hasPrev: boolean; + jobs: unknown[]; + source: string; +}; diff --git a/backend/src/modules/notifications/notifications.service.ts b/backend/src/modules/notifications/notifications.service.ts index ea713a3..3fbf2d8 100644 --- a/backend/src/modules/notifications/notifications.service.ts +++ b/backend/src/modules/notifications/notifications.service.ts @@ -1,13 +1,17 @@ import { and, count, desc, eq, isNull } from "drizzle-orm"; import { db } from "../../db/client"; -import { NewUserNotification, SavedJob, userNotifications } from "../../db/schema"; +import { + NewUserNotification, + SavedJob, + userNotifications, +} from "../../db/schema"; import { DB } from "../../db/types/types"; import { ownedBy } from "../../lib/authorization/ownership"; import { AppError } from "../../lib/errors"; import { jobNotificationIdentity, MatchedJob, -} from "../jobs/jobMatch.service"; +} from "../jobs/services/jobMatch.service"; import { ListNotificationsQuery } from "./schemas/notifications.schemas"; const statusLabels: Record = { @@ -67,7 +71,10 @@ export class NotificationsService { } async create(data: NewUserNotification) { - const result = await this.tx.insert(userNotifications).values(data).returning(); + const result = await this.tx + .insert(userNotifications) + .values(data) + .returning(); return result[0]; } diff --git a/backend/src/modules/savedJobs/savedJobs.service.ts b/backend/src/modules/savedJobs/savedJobs.service.ts index eb08a1f..c93849a 100644 --- a/backend/src/modules/savedJobs/savedJobs.service.ts +++ b/backend/src/modules/savedJobs/savedJobs.service.ts @@ -71,8 +71,13 @@ export class SavedJobsService { } async delete(userId: string, jobId: string): Promise { - await this.tx + const result = await this.tx .delete(savedJobs) - .where(and(eq(savedJobs.id, jobId), ownedBy(userId, savedJobs.userId))); + .where(and(eq(savedJobs.id, jobId), ownedBy(userId, savedJobs.userId))) + .returning({ id: savedJobs.id }); + + if (!result[0]) { + throw AppError.notFound("Vaga não encontrada"); + } } } diff --git a/backend/src/modules/savedJobs/schemas/savedJobs.schemas.ts b/backend/src/modules/savedJobs/schemas/savedJobs.schemas.ts index 835e0a7..31c290f 100644 --- a/backend/src/modules/savedJobs/schemas/savedJobs.schemas.ts +++ b/backend/src/modules/savedJobs/schemas/savedJobs.schemas.ts @@ -16,5 +16,9 @@ export const createSavedJobSchema = z.object({ export const updateSavedJobSchema = createSavedJobSchema.partial(); +export const savedJobParamsSchema = z.object({ + id: z.string().uuid("ID da vaga salva inválido."), +}); + export type CreateSavedJobInput = z.infer; export type UpdateSavedJobInput = z.infer; diff --git a/backend/src/routes/admin.routes.ts b/backend/src/routes/admin.routes.ts index cda50d4..8b20065 100644 --- a/backend/src/routes/admin.routes.ts +++ b/backend/src/routes/admin.routes.ts @@ -47,6 +47,11 @@ router.post( requirePermission("scrapers", "trigger"), scrapersCtrl.trigger.bind(scrapersCtrl), ); +router.post( + "/scrapers/:id/run", + requirePermission("scrapers", "trigger"), + scrapersCtrl.triggerOne.bind(scrapersCtrl), +); router.get( "/observability/metrics", diff --git a/backend/src/routes/jobs.routes.ts b/backend/src/routes/jobs.routes.ts index 335cf14..0f8283e 100644 --- a/backend/src/routes/jobs.routes.ts +++ b/backend/src/routes/jobs.routes.ts @@ -1,372 +1,8 @@ -import { Request, Response, Router } from "express"; -import { - cacheAbsoluteSMembers, - cacheGetJobsByIds, - cacheSearchJobIds, - cacheSearchKeywords, -} from "../lib/cache"; -import { paginate, parsePagination } from "../lib/pagination"; -import { logWarn } from "../logger"; -import { - getUserMatchTechnologies, - MatchableJob, - MatchedJob, - scoreJobWithTechnologies, -} from "../modules/jobs/jobMatch.service"; -import { NotificationsService } from "../modules/notifications/notifications.service"; -import { UsersService } from "../modules/users/users.service"; +import { Router } from "express"; +import { searchJobsController } from "../modules/jobs/controllers/searchJobs.controller"; export const jobsRoutes = Router(); -type SearchJob = { - title?: string | null; - location?: string | null; - modality?: string | null; - description?: string | null; -}; - -type MatchTechnology = { - name: string; - years: number; -}; - -function firstQueryValue(value: unknown): string { - if (Array.isArray(value)) return firstQueryValue(value[0]); - return typeof value === "string" ? value.trim() : ""; -} - -function queryValues(value: unknown): string[] { - const values = Array.isArray(value) ? value : [value]; - - return values - .flatMap((item) => (typeof item === "string" ? item.split(",") : [])) - .map((item) => item.trim()) - .filter(Boolean); -} - -function normalizeComparable(value: string): string { - return value - .normalize("NFD") - .replace(/[\u0300-\u036f]/g, "") - .toLowerCase() - .replace(/[^\p{L}\p{N}]+/gu, " ") - .replace(/\s+/g, " ") - .trim(); -} - -function normalizeLevelFilter(value: string): string { - const normalized = normalizeComparable(value); - if ( - normalized === "estagio trainee" || - normalized === "estagio" || - normalized === "trainee" || - normalized === "intern" || - normalized === "internship" - ) { - return "estagio"; - } - return normalized; -} - -function containsTokenOrPhrase(text: string, needle: string): boolean { - if (needle.includes(" ")) return text.includes(needle); - return ` ${text} `.includes(` ${needle} `); -} - -function containsAny(text: string, needles: string[]): boolean { - return needles.some((needle) => containsTokenOrPhrase(text, needle)); -} - -function inferJobLevel(title: string): string { - const normalized = normalizeComparable(title); - if ( - containsAny(normalized, [ - "estagio", - "estagiario", - "intern", - "internship", - "trainee", - "aprendiz", - ]) - ) { - return "estagio"; - } - if ( - containsAny(normalized, [ - "senior", - "sr", - "especialista", - "lead", - "principal", - "staff", - ]) - ) { - return "senior"; - } - if ( - containsAny(normalized, ["junior", "jr", "entry level", "assistente"]) - ) { - return "junior"; - } - return "pleno"; -} - -function inferJobType(job: SearchJob): string { - const normalized = normalizeComparable( - [ - job.title, - job.location, - job.modality, - job.description, - ] - .filter(Boolean) - .join(" "), - ); - - if (normalized.includes("hibrid") || normalized.includes("hybrid")) { - return "hibrido"; - } - if ( - normalized.includes("remot") || - normalized.includes("home office") || - normalized.includes("teletrabalho") || - normalized.includes("anywhere") || - normalized.includes("worldwide") - ) { - return "remoto"; - } - if ( - normalized.includes("presencial") || - normalized.includes("onsite") || - normalized.includes("on site") || - normalized.includes("on-site") || - normalized.includes("in office") || - normalized.includes("escritorio") - ) { - return "presencial"; - } - - return "presencial"; -} - -function inferLocationCountry(location: string): string { - const normalized = normalizeComparable(location); - if (!normalized) return ""; - - if ( - containsAny(normalized, [ - "estados unidos", - "united states", - "usa", - "eua", - "florida", - "miami", - "new york", - "california", - "texas", - "boston", - "seattle", - "chicago", - "atlanta", - "denver", - ]) - ) { - return "estados unidos"; - } - - if ( - containsAny(normalized, [ - "brasil", - "brazil", - "sao paulo", - "rio de janeiro", - "minas gerais", - "belo horizonte", - "parana", - "curitiba", - "santa catarina", - "joinville", - "rio grande do sul", - "porto alegre", - "pernambuco", - "recife", - "bahia", - "salvador", - "ceara", - "fortaleza", - "piaui", - "teresina", - ]) - ) { - return "brasil"; - } - - if (containsAny(normalized, ["portugal", "lisboa", "porto"])) { - return "portugal"; - } - - return ""; -} - -function matchesLocationFilter(jobLocation: string, location: string): boolean { - if (!location) return true; - - const normalizedLocation = normalizeComparable(jobLocation); - const inferredCountry = inferLocationCountry(jobLocation); - if (inferredCountry) return inferredCountry === location; - - return normalizedLocation.includes(location); -} - -function getTypeFilters(query: Request["query"]): string[] { - const rawTypes = queryValues(query.model).length > 0 - ? queryValues(query.model) - : queryValues(query.type); - - return [...new Set(rawTypes.map(normalizeComparable).filter(Boolean))]; -} - -function filterJobs(jobs: unknown[], query: Request["query"]): unknown[] { - const level = normalizeLevelFilter(firstQueryValue(query.level)); - const location = normalizeComparable( - firstQueryValue(query.country) || firstQueryValue(query.location), - ); - const types = getTypeFilters(query); - - if (!level && !location && types.length === 0) return jobs; - - return jobs.filter((job) => { - const candidate = job as SearchJob; - const title = candidate.title ?? ""; - const jobLocation = candidate.location ?? ""; - - const matchesLevel = !level || inferJobLevel(title) === level; - const matchesLocation = matchesLocationFilter(jobLocation, location); - const matchesType = - types.length === 0 || types.includes(inferJobType(candidate)); - - return matchesLevel && matchesLocation && matchesType; - }); -} - -function hasStructuredFilters(query: Request["query"]): boolean { - return Boolean( - firstQueryValue(query.level) || - firstQueryValue(query.location) || - firstQueryValue(query.country) || - firstQueryValue(query.continent) || - firstQueryValue(query.state) || - firstQueryValue(query.city) || - firstQueryValue(query.type) || - firstQueryValue(query.model) || - firstQueryValue(query.contract) || - firstQueryValue(query.contractType) || - firstQueryValue(query.jobTypes), - ); -} - -function getKeywordsArray(query: Request["query"]): string[] { - const keywords = firstQueryValue(query.keywords); - if (!keywords) return []; - - return keywords - .split(",") - .map((k) => k.trim()) - .filter(Boolean); -} - -function getMatchSort(query: Request["query"]): "asc" | "desc" | null { - const value = firstQueryValue(query.matchSort) || firstQueryValue(query.sort); - return value === "asc" || value === "desc" ? value : null; -} - -function sortJobsByMatch( - jobs: T[], - direction: "asc" | "desc", -) { - return [...jobs].sort((first, second) => { - const firstJob = first as { matchScore?: number | null }; - const secondJob = second as { matchScore?: number | null }; - const firstScore = - typeof firstJob.matchScore === "number" ? firstJob.matchScore : 0; - const secondScore = - typeof secondJob.matchScore === "number" ? secondJob.matchScore : 0; - - return direction === "desc" - ? secondScore - firstScore - : firstScore - secondScore; - }); -} - -async function legacyResolveIds( - keywordsArray: string[], -): Promise<{ ids: string[]; source: string }> { - if (keywordsArray.length > 0) { - return { - ids: await cacheSearchKeywords(keywordsArray), - source: `valkey_filtered_by_keywords:${keywordsArray.join("+")}`, - }; - } - - return { - ids: await cacheAbsoluteSMembers("scraper:jobs:index"), - source: "valkey_global_index", - }; -} - -async function getCurrentUserMatchTechnologies(req: Request) { - const userId = req.session?.userId; - if (!userId) return []; - - try { - const user = await new UsersService().getUserById(userId); - return getUserMatchTechnologies(user); - } catch (error) { - logWarn("Não foi possível carregar perfil para cálculo de match", { - error: (error as Error).message, - userId, - }); - return []; - } -} - -async function enrichJobsWithProfileMatch( - req: Request, - jobs: unknown[], - technologies: MatchTechnology[], - options: { notifyHighMatches?: boolean } = {}, -) { - if (technologies.length === 0) return jobs; - - const matchedJobs = jobs.map((job) => - scoreJobWithTechnologies(job as MatchableJob, technologies), - ); - if (options.notifyHighMatches === false) return matchedJobs; - - await notifyHighMatchJobs(req, matchedJobs); - return matchedJobs; -} - -async function notifyHighMatchJobs(req: Request, matchedJobs: MatchedJob[]) { - const userId = req.session?.userId; - if (!userId) return; - - const notifications = new NotificationsService(); - await Promise.all( - matchedJobs - .filter((job) => (job.matchScore ?? 0) >= 85) - .map((job) => - notifications.createHighMatchIfMissing(userId, job).catch((error) => { - logWarn("Não foi possível registrar notificação de alto match", { - error: (error as Error).message, - userId, - job: job.title ?? job.jobTitle ?? job.id, - }); - }), - ), - ); -} - /** * @swagger * /api/jobs/search: @@ -380,174 +16,4 @@ async function notifyHighMatchJobs(req: Request, matchedJobs: MatchedJob[]) { * type: string * description: 'Termos para filtrar (ex: "react,node") separados por vírgula' */ -jobsRoutes.get("/search", async (req: Request, res: Response) => { - try { - const keywordsArray = getKeywordsArray(req.query); - const pagination = parsePagination(req.query); - const hasFilters = hasStructuredFilters(req.query); - const matchSort = getMatchSort(req.query); - const matchTechnologies = await getCurrentUserMatchTechnologies(req); - - let ids: string[] = []; - let source = - keywordsArray.length > 0 - ? `valkey_filtered_by_keywords:${keywordsArray.join("+")}` - : "valkey_global_index"; - - if (hasFilters) { - ids = await cacheSearchJobIds({ - keywords: keywordsArray, - level: firstQueryValue(req.query.level), - location: firstQueryValue(req.query.location), - continent: firstQueryValue(req.query.continent), - country: firstQueryValue(req.query.country), - state: firstQueryValue(req.query.state), - city: firstQueryValue(req.query.city), - type: queryValues(req.query.type), - model: queryValues(req.query.model), - contract: - firstQueryValue(req.query.contract) || - firstQueryValue(req.query.contractType) || - firstQueryValue(req.query.jobTypes), - }); - source = `${source}:structured_indexes`; - - if (ids.length === 0) { - const legacy = await legacyResolveIds(keywordsArray); - const legacyJobs = await cacheGetJobsByIds(legacy.ids); - const filteredJobs = filterJobs(legacyJobs, req.query); - const { data: pageJobs, pagination: meta } = matchSort - ? paginate( - sortJobsByMatch( - await enrichJobsWithProfileMatch( - req, - filteredJobs, - matchTechnologies, - { notifyHighMatches: false }, - ), - matchSort, - ), - pagination, - ) - : paginate(filteredJobs, pagination); - if (matchSort) { - await notifyHighMatchJobs(req, pageJobs as MatchedJob[]); - } - const jobs = matchSort - ? pageJobs - : await enrichJobsWithProfileMatch( - req, - pageJobs, - matchTechnologies, - ); - - return res.json({ - total: meta.total, - page: meta.page, - limit: meta.limit, - totalPages: meta.totalPages, - hasNext: meta.hasNext, - hasPrev: meta.hasPrev, - jobs, - source: `${source}:legacy_post_filter_fallback`, - }); - } - - const indexedJobs = await cacheGetJobsByIds(ids); - const filteredJobs = filterJobs(indexedJobs, req.query); - const { data: pageJobs, pagination: meta } = matchSort - ? paginate( - sortJobsByMatch( - await enrichJobsWithProfileMatch( - req, - filteredJobs, - matchTechnologies, - { notifyHighMatches: false }, - ), - matchSort, - ), - pagination, - ) - : paginate(filteredJobs, pagination); - if (matchSort) { - await notifyHighMatchJobs(req, pageJobs as MatchedJob[]); - } - const jobs = matchSort - ? pageJobs - : await enrichJobsWithProfileMatch( - req, - pageJobs, - matchTechnologies, - ); - - return res.json({ - total: meta.total, - page: meta.page, - limit: meta.limit, - totalPages: meta.totalPages, - hasNext: meta.hasNext, - hasPrev: meta.hasPrev, - jobs, - source: `${source}:verified`, - }); - } else { - const legacy = await legacyResolveIds(keywordsArray); - ids = legacy.ids; - source = legacy.source; - } - - if (matchSort) { - const allJobs = await cacheGetJobsByIds(ids); - const matchedJobs = await enrichJobsWithProfileMatch( - req, - allJobs, - matchTechnologies, - { notifyHighMatches: false }, - ); - const sortedJobs = sortJobsByMatch(matchedJobs, matchSort); - const { data: jobs, pagination: meta } = paginate( - sortedJobs, - pagination, - ); - await notifyHighMatchJobs(req, jobs as MatchedJob[]); - - return res.json({ - total: meta.total, - page: meta.page, - limit: meta.limit, - totalPages: meta.totalPages, - hasNext: meta.hasNext, - hasPrev: meta.hasPrev, - jobs, - source: `${source}:match_sorted_${matchSort}`, - }); - } - - const { data: pageIds, pagination: meta } = paginate(ids, pagination); - const pageJobs = await cacheGetJobsByIds(pageIds); - const jobs = await enrichJobsWithProfileMatch( - req, - pageJobs, - matchTechnologies, - ); - - return res.json({ - total: meta.total, - page: meta.page, - limit: meta.limit, - totalPages: meta.totalPages, - hasNext: meta.hasNext, - hasPrev: meta.hasPrev, - jobs, - source, - }); - } catch (error) { - logWarn("Erro ao buscar vagas no ecossistema Valkey", { - error: (error as Error).message, - }); - return res.status(500).json({ - message: "Erro ao recuperar vagas em memória.", - error: (error as Error).message, - }); - } -}); +jobsRoutes.get("/search", searchJobsController); diff --git a/backend/src/routes/savedJobs.routes.ts b/backend/src/routes/savedJobs.routes.ts index e253090..3d26c13 100644 --- a/backend/src/routes/savedJobs.routes.ts +++ b/backend/src/routes/savedJobs.routes.ts @@ -4,6 +4,7 @@ import { SavedJobsController } from "../modules/savedJobs/savedJobs.controller"; import { SavedJobsService } from "../modules/savedJobs/savedJobs.service"; import { createSavedJobSchema, + savedJobParamsSchema, updateSavedJobSchema, } from "../modules/savedJobs/schemas/savedJobs.schemas"; @@ -14,21 +15,29 @@ const controller = new SavedJobsController(service); router.get("/", (req, res, next) => { controller.getAll(req, res).catch(next); }); -router.get("/:id", (req, res, next) => { - controller.getById(req, res).catch(next); -}); +router.get( + "/:id", + validate({ params: savedJobParamsSchema }), + (req, res, next) => { + controller.getById(req, res).catch(next); + }, +); router.post("/", validate({ body: createSavedJobSchema }), (req, res, next) => { controller.create(req, res).catch(next); }); router.patch( "/:id", - validate({ body: updateSavedJobSchema }), + validate({ params: savedJobParamsSchema, body: updateSavedJobSchema }), (req, res, next) => { controller.update(req, res).catch(next); }, ); -router.delete("/:id", (req, res, next) => { - controller.delete(req, res).catch(next); -}); +router.delete( + "/:id", + validate({ params: savedJobParamsSchema }), + (req, res, next) => { + controller.delete(req, res).catch(next); + }, +); export { router as savedJobsRoutes }; diff --git a/backend/tests/integration/routes/admin.routes.test.ts b/backend/tests/integration/routes/admin.routes.test.ts index d3d75ef..8e527b3 100644 --- a/backend/tests/integration/routes/admin.routes.test.ts +++ b/backend/tests/integration/routes/admin.routes.test.ts @@ -42,6 +42,7 @@ const mocks = vi.hoisted(() => ({ scrapersJobs: vi.fn((_req, res) => res.json({ jobs: [], total: 0 })), scrapersJobsCount: vi.fn((_req, res) => res.json({ total: 0 })), scrapersTrigger: vi.fn((_req, res) => res.status(202).json({ ok: true })), + scrapersTriggerOne: vi.fn((_req, res) => res.status(202).json({ ok: true })), health: vi.fn((_req, res) => res.json({ status: "ok" })), metrics: vi.fn((_req, res) => res.json({ requestRatePerMinute: 1 })), dashboards: vi.fn((_req, res) => res.json({ dashboards: [] })), @@ -68,6 +69,7 @@ vi.mock("../../../src/routes/admin.context", () => ({ listJobs: mocks.scrapersJobs, jobsCount: mocks.scrapersJobsCount, trigger: mocks.scrapersTrigger, + triggerOne: mocks.scrapersTriggerOne, }, observabilityCtrl: { getHealth: mocks.health, @@ -156,6 +158,7 @@ describe("Integration - Admin Routes", () => { await request(app).patch("/admin/users/user-2/block").expect(200); await request(app).post("/admin/users/user-2/reset").expect(200); await request(app).post("/admin/scrapers/run").expect(202); + await request(app).post("/admin/scrapers/go-scraper/run").expect(202); await request(app).get("/admin/observability/metrics").expect(200); await request(app).get("/admin/audit").expect(200); await request(app).get("/admin/users").expect(200); @@ -163,6 +166,7 @@ describe("Integration - Admin Routes", () => { expect(mocks.blockUser).toHaveBeenCalled(); expect(mocks.resetPassword).toHaveBeenCalled(); expect(mocks.scrapersTrigger).toHaveBeenCalled(); + expect(mocks.scrapersTriggerOne).toHaveBeenCalled(); expect(mocks.metrics).toHaveBeenCalled(); expect(mocks.audit).toHaveBeenCalled(); expect(mocks.usersList).toHaveBeenCalled(); diff --git a/backend/tests/integration/routes/savedJobs.routes.test.ts b/backend/tests/integration/routes/savedJobs.routes.test.ts index 07bbb09..4ffada9 100644 --- a/backend/tests/integration/routes/savedJobs.routes.test.ts +++ b/backend/tests/integration/routes/savedJobs.routes.test.ts @@ -39,8 +39,10 @@ const fixtureSession = { destroy: vi.fn().mockResolvedValue(undefined), }; +const fixtureJobId = "e4b095ff-6439-4112-b837-024a50f838b0"; + const fixtureJob = { - id: "job-1", + id: fixtureJobId, userId: "user_abc", jobLink: "https://example.com/job/1", jobTitle: "Engenheiro de Software", @@ -132,24 +134,26 @@ describe("Integration - SavedJobs Routes", () => { describe("GET /:id", () => { it("retorna 200 e a vaga encontrada", async () => { - const res = await request(app).get(`${BASE}/job-1`).expect(200); + const res = await request(app).get(`${BASE}/${fixtureJobId}`).expect(200); - expect(res.body).toHaveProperty("id", "job-1"); + expect(res.body).toHaveProperty("id", fixtureJobId); }); it("chama getById com userId e jobId corretos", async () => { - await request(app).get(`${BASE}/job-1`); + await request(app).get(`${BASE}/${fixtureJobId}`); expect(mockSavedJobsService.getById).toHaveBeenCalledWith( "user_abc", - "job-1", + fixtureJobId, ); }); it("retorna 404 quando vaga não existe", async () => { mockSavedJobsService.getById.mockResolvedValueOnce(undefined); - await request(app).get(`${BASE}/inexistente`).expect(404); + await request(app) + .get(`${BASE}/11111111-1111-4111-8111-111111111111`) + .expect(404); }); it("retorna 401 quando sessão não tem userId", async () => { @@ -157,7 +161,7 @@ describe("Integration - SavedJobs Routes", () => { userId: undefined, } as any); - await request(app).get(`${BASE}/job-1`).expect(401); + await request(app).get(`${BASE}/${fixtureJobId}`).expect(401); }); }); @@ -167,7 +171,7 @@ describe("Integration - SavedJobs Routes", () => { it("cria vaga e retorna 201", async () => { const res = await request(app).post(BASE).send(createPayload).expect(201); - expect(res.body).toHaveProperty("id", "job-1"); + expect(res.body).toHaveProperty("id", fixtureJobId); }); it("chama create com userId e body parseado pelo Zod", async () => { @@ -258,7 +262,7 @@ describe("Integration - SavedJobs Routes", () => { it("retorna 200 e a vaga com status atualizado", async () => { const res = await request(app) - .patch(`${BASE}/job-1`) + .patch(`${BASE}/${fixtureJobId}`) .send(updatePayload) .expect(200); @@ -266,35 +270,40 @@ describe("Integration - SavedJobs Routes", () => { }); it("chama update com userId, jobId e body corretos", async () => { - await request(app).patch(`${BASE}/job-1`).send(updatePayload); + await request(app) + .patch(`${BASE}/${fixtureJobId}`) + .send(updatePayload); expect(mockSavedJobsService.update).toHaveBeenCalledWith( "user_abc", - "job-1", + fixtureJobId, expect.objectContaining({ status: "applied" }), ); }); it("retorna 400 para status fora do enum", async () => { await request(app) - .patch(`${BASE}/job-1`) + .patch(`${BASE}/${fixtureJobId}`) .send({ status: "nao-existe" }) .expect(400); }); it("retorna 400 para jobLink inválido no update", async () => { await request(app) - .patch(`${BASE}/job-1`) + .patch(`${BASE}/${fixtureJobId}`) .send({ jobLink: "nao-e-url" }) .expect(400); }); it("aceita body vazio (todos os campos são opcionais no updateJobSchema)", async () => { - await request(app).patch(`${BASE}/job-1`).send({}).expect(200); + await request(app) + .patch(`${BASE}/${fixtureJobId}`) + .send({}) + .expect(200); expect(mockSavedJobsService.update).toHaveBeenCalledWith( "user_abc", - "job-1", + fixtureJobId, {}, ); }); @@ -305,7 +314,7 @@ describe("Integration - SavedJobs Routes", () => { ); const res = await request(app) - .patch(`${BASE}/job-1`) + .patch(`${BASE}/${fixtureJobId}`) .send(updatePayload) .expect(404); @@ -320,7 +329,10 @@ describe("Integration - SavedJobs Routes", () => { userId: undefined, } as any); - await request(app).patch(`${BASE}/job-1`).send(updatePayload).expect(401); + await request(app) + .patch(`${BASE}/${fixtureJobId}`) + .send(updatePayload) + .expect(401); }); }); @@ -328,17 +340,19 @@ describe("Integration - SavedJobs Routes", () => { describe("DELETE /:id", () => { it("retorna 204 sem body", async () => { - const res = await request(app).delete(`${BASE}/job-1`).expect(204); + const res = await request(app) + .delete(`${BASE}/${fixtureJobId}`) + .expect(204); expect(res.text).toBe(""); }); it("chama delete com userId e jobId corretos", async () => { - await request(app).delete(`${BASE}/job-1`); + await request(app).delete(`${BASE}/${fixtureJobId}`); expect(mockSavedJobsService.delete).toHaveBeenCalledWith( "user_abc", - "job-1", + fixtureJobId, ); }); @@ -347,13 +361,19 @@ describe("Integration - SavedJobs Routes", () => { userId: undefined, } as any); - await request(app).delete(`${BASE}/job-1`).expect(401); + await request(app).delete(`${BASE}/${fixtureJobId}`).expect(401); + }); + + it("retorna 400 quando id não é um UUID válido", async () => { + await request(app).delete(`${BASE}/${fixtureJobId}%22`).expect(400); + + expect(mockSavedJobsService.delete).not.toHaveBeenCalled(); }); it("retorna 500 quando delete lança erro", async () => { mockSavedJobsService.delete.mockRejectedValueOnce(new Error("db error")); - await request(app).delete(`${BASE}/job-1`).expect(500); + await request(app).delete(`${BASE}/${fixtureJobId}`).expect(500); }); }); }); diff --git a/backend/tests/unit/app.test.ts b/backend/tests/unit/app.test.ts index 1c6cb6a..0f0c457 100644 --- a/backend/tests/unit/app.test.ts +++ b/backend/tests/unit/app.test.ts @@ -18,6 +18,7 @@ const mocks = vi.hoisted(() => ({ dbInsertConflict: vi.fn(), getUserById: vi.fn(), createHighMatchIfMissing: vi.fn(), + jobSearchesInc: vi.fn(), })); vi.mock("../../src/lib/cache.js", () => ({ @@ -62,6 +63,13 @@ vi.mock("../../src/logger.js", () => ({ logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn() }, })); +vi.mock("../../src/metrics/metrics.js", () => ({ + register: { contentType: "text/plain", metrics: vi.fn().mockResolvedValue("") }, + httpRequestDuration: { startTimer: vi.fn(() => vi.fn()) }, + httpRequestsTotal: { inc: vi.fn() }, + jobSearchesTotal: { inc: mocks.jobSearchesInc }, +})); + vi.mock("../../src/routes/auth.routes.js", async () => { const { Router } = await import("express"); return { authRoutes: Router() }; @@ -315,6 +323,9 @@ describe("jobsApiApp", () => { expect(mocks.cacheSearchJobIds).toHaveBeenCalledWith({ keywords: ["React"], + family: [], + technology: [], + seniority: "", level: "Júnior", location: "Brasil", continent: "", @@ -322,9 +333,10 @@ describe("jobsApiApp", () => { state: "", city: "", type: ["Remoto"], - model: [], + model: ["Remoto"], contract: "", }); + expect(mocks.jobSearchesInc).toHaveBeenCalledWith({ has_keywords: "true" }); expect(mocks.cacheSearchKeywords).not.toHaveBeenCalled(); expect(mocks.cacheGetJobsByIds).toHaveBeenCalledWith(["id-structured"]); expect(res.body.jobs).toEqual([ diff --git a/backend/tests/unit/libs/cache.test.ts b/backend/tests/unit/libs/cache.test.ts index 9e9f19b..abcbb46 100644 --- a/backend/tests/unit/libs/cache.test.ts +++ b/backend/tests/unit/libs/cache.test.ts @@ -25,6 +25,16 @@ vi.mock("../logger", () => ({ }, })); +const metricMocks = vi.hoisted(() => ({ + cacheInc: vi.fn(), +})); + +vi.mock("../../../src/metrics/metrics", () => ({ + cacheOperationsTotal: { + inc: metricMocks.cacheInc, + }, +})); + // Mock do pacote redis vi.mock("redis", () => { const mockClient = { @@ -89,6 +99,10 @@ describe("Valkey Cache Lib", () => { expect(mockClientInstance.get).toHaveBeenCalledWith("user:profile:123"); expect(result).toEqual({ id: 1, name: "Test" }); + expect(metricMocks.cacheInc).toHaveBeenCalledWith({ + operation: "get", + result: "hit", + }); }); it("deve retornar uma string pura se falhar no JSON.parse", async () => { @@ -97,6 +111,18 @@ describe("Valkey Cache Lib", () => { expect(result).toBe("string_pura"); }); + it("deve registrar miss no cacheGet quando a chave nao existir", async () => { + mockClientInstance.get.mockResolvedValue(null); + + const result = await cacheGet("profile:missing"); + + expect(result).toBeNull(); + expect(metricMocks.cacheInc).toHaveBeenCalledWith({ + operation: "get", + result: "miss", + }); + }); + it("deve salvar com TTL se fornecido", async () => { await cacheSet("profile:123", { role: "developer" }, 60); expect(mockClientInstance.set).toHaveBeenCalledWith( @@ -104,6 +130,10 @@ describe("Valkey Cache Lib", () => { JSON.stringify({ role: "developer" }), { EX: 60 }, ); + expect(metricMocks.cacheInc).toHaveBeenCalledWith({ + operation: "set", + result: "ok", + }); }); it("deve salvar sem TTL se o valor for menor ou igual a 0", async () => { @@ -117,6 +147,10 @@ describe("Valkey Cache Lib", () => { it("deve deletar uma chave no cacheDel", async () => { await cacheDel("profile:123"); expect(mockClientInstance.del).toHaveBeenCalledWith("user:profile:123"); + expect(metricMocks.cacheInc).toHaveBeenCalledWith({ + operation: "delete", + result: "ok", + }); }); }); @@ -169,9 +203,17 @@ describe("Valkey Cache Lib", () => { // "UX/UI Designer" -> "ux ui designer" expect(mockClientInstance.sUnion).toHaveBeenCalledWith([ "scraper:jobs:keyword:ux ui designer", + "scraper:jobs:technology:ux ui designer", + "scraper:jobs:family:ux ui designer", "scraper:jobs:keyword:ux", + "scraper:jobs:technology:ux", + "scraper:jobs:family:ux", "scraper:jobs:keyword:ui", + "scraper:jobs:technology:ui", + "scraper:jobs:family:ui", "scraper:jobs:keyword:designer", + "scraper:jobs:technology:designer", + "scraper:jobs:family:designer", ]); expect(result).toEqual(["job_1"]); }); @@ -183,8 +225,14 @@ describe("Valkey Cache Lib", () => { expect(mockClientInstance.sUnion).toHaveBeenCalledWith([ "scraper:jobs:keyword:go", + "scraper:jobs:technology:go", + "scraper:jobs:family:go", "scraper:jobs:keyword:c#", + "scraper:jobs:technology:c#", + "scraper:jobs:family:c#", "scraper:jobs:keyword:c", + "scraper:jobs:technology:c", + "scraper:jobs:family:c", ]); expect(result).toEqual(["job_1", "job_2"]); }); @@ -194,12 +242,18 @@ describe("Valkey Cache Lib", () => { it("deve montar chaves normalizadas para filtros estruturados", () => { expect( cacheJobIndexKeys({ + family: "Front-end", + technology: "React Native", + seniority: "Sênior", level: "Júnior", location: "Brasil", type: "Híbrido", contract: "PJ", }), ).toEqual([ + "scraper:jobs:family:front end", + "scraper:jobs:technology:react native", + "scraper:jobs:seniority:senior", "scraper:jobs:level:junior", "scraper:jobs:location:brasil", "scraper:jobs:model:hibrido", @@ -241,10 +295,20 @@ describe("Valkey Cache Lib", () => { "SUNIONSTORE", expect.stringMatching(/^scraper:jobs:search:/), "scraper:jobs:keyword:node.js", + "scraper:jobs:technology:node.js", + "scraper:jobs:family:node.js", "scraper:jobs:keyword:node js", + "scraper:jobs:technology:node js", + "scraper:jobs:family:node js", "scraper:jobs:keyword:nodejs", + "scraper:jobs:technology:nodejs", + "scraper:jobs:family:nodejs", "scraper:jobs:keyword:node", + "scraper:jobs:technology:node", + "scraper:jobs:family:node", "scraper:jobs:keyword:js", + "scraper:jobs:technology:js", + "scraper:jobs:family:js", ]); expect(mockClientInstance.sendCommand).toHaveBeenNthCalledWith(2, [ "SINTER", @@ -270,7 +334,11 @@ describe("Valkey Cache Lib", () => { "SUNIONSTORE", expect.stringMatching(/^scraper:jobs:search:/), "scraper:jobs:keyword:react", + "scraper:jobs:technology:react", + "scraper:jobs:family:react", "scraper:jobs:keyword:node", + "scraper:jobs:technology:node", + "scraper:jobs:family:node", ]); expect(mockClientInstance.expire).toHaveBeenCalledWith( expect.stringMatching(/^scraper:jobs:search:/), diff --git a/backend/tests/unit/modules/admin/dashboard.service.test.ts b/backend/tests/unit/modules/admin/dashboard.service.test.ts index 275d1cc..9533499 100644 --- a/backend/tests/unit/modules/admin/dashboard.service.test.ts +++ b/backend/tests/unit/modules/admin/dashboard.service.test.ts @@ -106,6 +106,7 @@ describe("DashboardService", () => { running: false, jobsCollected: 5, }); + scrapersService.getJobsCount.mockRejectedValueOnce(new Error("down")); const service = new DashboardService( repository as any, diff --git a/backend/tests/unit/modules/admin/scrapers.test.ts b/backend/tests/unit/modules/admin/scrapers.test.ts index 8d625d6..9f3bc0d 100644 --- a/backend/tests/unit/modules/admin/scrapers.test.ts +++ b/backend/tests/unit/modules/admin/scrapers.test.ts @@ -167,6 +167,23 @@ describe("ScrapersService", () => { ); }); + it("triggers a known scraper by name and rejects unknown scrapers", async () => { + vi.spyOn(scraperClient, "triggerScrape").mockResolvedValue({ + ok: true, + message: "started", + }); + + const service = new ScrapersService(); + + await expect(service.triggerScraper("go-scraper")).resolves.toEqual({ + ok: true, + message: "started", + }); + await expect(service.triggerScraper("lever")).rejects.toThrow( + "scraper desconhecido", + ); + }); + it("returns null jobsCollected when count fails", async () => { vi.spyOn(scraperClient, "getStatus").mockResolvedValue({ running: true }); vi.spyOn(scraperClient, "getJobsCount").mockRejectedValue(new Error("down")); @@ -198,6 +215,7 @@ describe("ScrapersService", () => { describe("ScrapersController", () => { const service = { triggerScrape: vi.fn(), + triggerScraper: vi.fn(), getStatus: vi.fn(), listScrapers: vi.fn(), getJobs: vi.fn(), @@ -210,6 +228,7 @@ describe("ScrapersController", () => { beforeEach(() => { vi.clearAllMocks(); service.triggerScrape.mockResolvedValue({ ok: true, message: "started" }); + service.triggerScraper.mockResolvedValue({ ok: true, message: "started" }); service.getStatus.mockResolvedValue({ running: false }); service.listScrapers.mockResolvedValue([{ name: "go-scraper", running: false }]); service.getJobs.mockResolvedValue({ jobs: [], total: 0 }); @@ -232,6 +251,42 @@ describe("ScrapersController", () => { expect(conflictRes.status).toHaveBeenCalledWith(409); }); + it("triggers one scraper with 202 and maps failures", async () => { + const res = response(); + await controller.triggerOne( + { ...req, params: { id: "go-scraper" } } as unknown as Request, + res, + ); + + expect(res.status).toHaveBeenCalledWith(202); + expect(service.triggerScraper).toHaveBeenCalledWith("go-scraper"); + expect(auditService.fromRequest).toHaveBeenCalledWith( + expect.anything(), + "scrapers.trigger", + { type: "scrapers", id: "go-scraper" }, + ); + + service.triggerScraper.mockRejectedValueOnce( + new ScraperAlreadyRunningError("busy"), + ); + const conflictRes = response(); + await controller.triggerOne( + { ...req, params: { id: "go-scraper" } } as unknown as Request, + conflictRes, + ); + expect(conflictRes.status).toHaveBeenCalledWith(409); + + service.triggerScraper.mockRejectedValueOnce( + new Error("scraper desconhecido: lever"), + ); + const missingRes = response(); + await controller.triggerOne( + { ...req, params: { id: "lever" } } as unknown as Request, + missingRes, + ); + expect(missingRes.status).toHaveBeenCalledWith(404); + }); + it("returns read endpoints and audits each read", async () => { await controller.status(req, response()); await controller.list(req, response()); diff --git a/backend/tests/unit/modules/savedJobs/savedJobs.service.test.ts b/backend/tests/unit/modules/savedJobs/savedJobs.service.test.ts index 8247dc4..012293e 100644 --- a/backend/tests/unit/modules/savedJobs/savedJobs.service.test.ts +++ b/backend/tests/unit/modules/savedJobs/savedJobs.service.test.ts @@ -321,7 +321,9 @@ describe("SavedJobsService", () => { describe("delete", () => { it("executa delete sem lançar erro", async () => { tx.delete.mockReturnValue({ - where: vi.fn().mockResolvedValue(undefined), + where: vi.fn().mockReturnValue({ + returning: vi.fn().mockResolvedValue([{ id: "job-1" }]), + }), }); await expect(service.delete("user-1", "job-1")).resolves.toBeUndefined(); @@ -329,7 +331,9 @@ describe("SavedJobsService", () => { }); it("deleta usando jobId e userId para impedir remoção cross-user", async () => { - const whereMock = vi.fn().mockResolvedValue(undefined); + const whereMock = vi.fn().mockReturnValue({ + returning: vi.fn().mockResolvedValue([{ id: "job-1" }]), + }); tx.delete.mockReturnValue({ where: whereMock }); await service.delete("user-1", "job-1"); @@ -339,5 +343,18 @@ describe("SavedJobsService", () => { { column: expect.anything(), value: "user-1" }, ]); }); + + it("lança NOT_FOUND quando nenhuma vaga é removida", async () => { + tx.delete.mockReturnValue({ + where: vi.fn().mockReturnValue({ + returning: vi.fn().mockResolvedValue([]), + }), + }); + + await expect(service.delete("user-1", "job-1")).rejects.toMatchObject({ + code: "NOT_FOUND", + message: "Vaga não encontrada", + }); + }); }); }); diff --git a/backend/tests/unit/utils/config.test.ts b/backend/tests/unit/utils/config.test.ts index f43da81..275d378 100644 --- a/backend/tests/unit/utils/config.test.ts +++ b/backend/tests/unit/utils/config.test.ts @@ -81,7 +81,7 @@ describe("getConfig", () => { const config = getConfig(); expect(config.headless).toBe(false); - expect(config.remoteOnly).toBe(true); + expect(config.remoteOnly).toBe(false); expect(config.waitBetweenSearchesMs).toBe(5000); expect(config.pageTimeoutMs).toBe(10000); expect(config.maxPagesPerKeyword).toBe(5); @@ -98,9 +98,9 @@ describe("getConfig", () => { expect(config.searchLanguage).toBe("pt"); }); - it("retorna remoteOnly true e jobTypes padrão", () => { + it("retorna remoteOnly false e jobTypes padrão", () => { const config = getConfig(); - expect(config.remoteOnly).toBe(true); + expect(config.remoteOnly).toBe(false); expect(config.jobTypes).toBe("C,F"); }); diff --git a/backend/tsconfig.json b/backend/tsconfig.json index 84e28c4..999b574 100644 --- a/backend/tsconfig.json +++ b/backend/tsconfig.json @@ -3,6 +3,7 @@ "target": "ES2022", "module": "CommonJS", "moduleResolution": "Node", + "ignoreDeprecations": "6.0", "outDir": "./dist", "rootDir": "./src", "strict": true, diff --git a/docker-compose.infra.yml b/docker-compose.infra.yml index becef28..cb03bd2 100644 --- a/docker-compose.infra.yml +++ b/docker-compose.infra.yml @@ -37,4 +37,4 @@ volumes: networks: vagas-net: - name: vagas-net + name: vagas-net \ No newline at end of file diff --git a/docker-compose.yml b/docker-compose.yml index ec32820..b602607 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -11,6 +11,21 @@ services: environment: - GO_SCRAPER_ADDR=:8081 - VALKEY_URL=redis://valkey:6379/0 + - GUPY_ENABLED=${GUPY_ENABLED:-true} + - GUPY_RAW_DISCOVERY_ENABLED=${GUPY_RAW_DISCOVERY_ENABLED:-true} + - GUPY_FULL_SWEEP_ENABLED=${GUPY_FULL_SWEEP_ENABLED:-true} + - GUPY_FULL_REMOTE_SWEEP_ENABLED=${GUPY_FULL_REMOTE_SWEEP_ENABLED:-true} + - INHIRE_ENABLED=${INHIRE_ENABLED:-true} + - INHIRE_TENANTS_FILE=/app/internal/interfaces/inhireTenants.json + - INHIRE_ENRICH_DETAILS=${INHIRE_ENRICH_DETAILS:-false} + - INHIRE_DETAILS_MODE=${INHIRE_DETAILS_MODE:-ambiguous} + - INHIRE_DETAILS_CONCURRENCY=${INHIRE_DETAILS_CONCURRENCY:-8} + - INHIRE_DETAILS_TIMEOUT_MS=${INHIRE_DETAILS_TIMEOUT_MS:-10000} + - GREENHOUSE_ENABLED=${GREENHOUSE_ENABLED:-true} + - GREENHOUSE_COMPANIES_FILE=/app/internal/interfaces/greenhouseCompanies.json + - LEVER_ENABLED=${LEVER_ENABLED:-false} + - LEVER_COMPANIES_FILE=/app/internal/interfaces/leverCompanies.json + - LEVER_INCLUDE_ALL_JOBS=${LEVER_INCLUDE_ALL_JOBS:-true} ports: - "8081:8081" healthcheck: diff --git a/front_admin/README.md b/front_admin/README.md index c3f4169..8908cef 100644 --- a/front_admin/README.md +++ b/front_admin/README.md @@ -1,6 +1,6 @@ # Front Admin -Painel administrativo do , separado da experiência principal para reduzir atrito no onboarding de contribuidores. +Painel administrativo do Cand!Date!, separado da experiência principal para reduzir atrito no onboarding de contribuidores. Use este workspace quando a tarefa envolver operação da plataforma, gestão de usuários, permissões, scrapers, auditoria, observabilidade ou configurações administrativas. @@ -34,7 +34,7 @@ Para executar apenas o painel: npm run dev --workspace=front_admin ``` -Quando executado junto com o frontend principal, use a porta http://localhost:5174 para o admin. +Quando executado junto com o frontend principal, use a porta para o admin. ## Variáveis de ambiente @@ -79,6 +79,10 @@ CORS_ALLOWED_ORIGINS=http://localhost:5173,http://localhost:5174 - `/audit`: auditoria. - `/settings`: configurações. +## Scrapers e Observabilidade + +O painel consome as rotas administrativas do backend para disparar o scraper completo ou uma fonte conhecida, consultar status, exibir contagem de vagas coletadas e mostrar dashboards de métricas. A fonte `go-scraper` é a integração operacional atual para execução manual. + ## Comandos ```bash diff --git a/front_admin/src/app/layouts/MainLayout/Header/components/NotificationButton.tsx b/front_admin/src/app/layouts/MainLayout/Header/components/NotificationButton.tsx index 023d1c9..efecd33 100644 --- a/front_admin/src/app/layouts/MainLayout/Header/components/NotificationButton.tsx +++ b/front_admin/src/app/layouts/MainLayout/Header/components/NotificationButton.tsx @@ -1,6 +1,8 @@ import { AlertTriangle, Bell, CheckCircle2, XCircle } from "lucide-react"; -import { useState } from "react"; +import { useEffect, useRef, useState } from "react"; import { useNotifications } from "../../../../../components/notifications/useNotifications"; +import { dashboardApi } from "../../../../../lib/api/dashboard.api"; +import type { DashboardOverview } from "../../../../../lib/api/types"; interface Notification { tone: "success" | "warning" | "error"; @@ -9,27 +11,6 @@ interface Notification { timeAgo: string; } -const NOTIFICATIONS: Notification[] = [ - { - tone: "success", - title: "LinkedIn Scraper atingiu a meta", - description: "2.543 vagas processadas com sucesso hoje.", - timeAgo: "Há 2 minutos", - }, - { - tone: "warning", - title: "SLA do Go Scraper normalizado", - description: "Retornou ao estado operacional ideal (98.7%).", - timeAgo: "Há 15 minutos", - }, - { - tone: "error", - title: "Falha de auditoria detectada", - description: "A tabela audit_logs ainda não foi criada no banco.", - timeAgo: "Há 28 minutos", - }, -]; - const NOTIFICATION_STYLES: Record< Notification["tone"], { @@ -63,10 +44,131 @@ const NOTIFICATION_STYLES: Record< }, }; +const SERVICE_LABELS = { + postgres: "Postgres", + valkey: "Valkey", + scraper: "Scraper", +} as const; + +function formatSnapshot(value: string): string { + const date = new Date(value); + if (Number.isNaN(date.getTime())) return "Snapshot recente"; + + return new Intl.DateTimeFormat("pt-BR", { + hour: "2-digit", + minute: "2-digit", + second: "2-digit", + }).format(date); +} + +function buildNotifications(overview: DashboardOverview): Notification[] { + const notifications: Notification[] = []; + const timeAgo = `Snapshot ${formatSnapshot(overview.generatedAt)}`; + + for (const [key, service] of Object.entries(overview.services.services)) { + if (service.status === "ok") continue; + + const label = SERVICE_LABELS[key as keyof typeof SERVICE_LABELS] ?? key; + notifications.push({ + tone: service.status === "down" ? "error" : "warning", + title: `${label} ${service.status === "down" ? "indisponível" : "degradado"}`, + description: service.error + ? service.error + : `Latência atual: ${service.latencyMs ?? 0}ms.`, + timeAgo, + }); + } + + for (const scraper of overview.scrapers) { + if (scraper.running) { + notifications.push({ + tone: "success", + title: `${scraper.name} em execução`, + description: "O scraper está rodando neste momento.", + timeAgo, + }); + } else if (scraper.status === "down") { + notifications.push({ + tone: "error", + title: `${scraper.name} indisponível`, + description: "O backend não conseguiu consultar o scraper.", + timeAgo, + }); + } + } + + if (notifications.length === 0) { + notifications.push({ + tone: "success", + title: "Operação saudável", + description: `${overview.stats.totalCollectedJobs.toLocaleString("pt-BR")} vagas no índice e serviços online.`, + timeAgo, + }); + } + + return notifications.slice(0, 5); +} + export function NotificationButton() { const [isOpen, setIsOpen] = useState(false); + const [notifications, setNotifications] = useState([]); + const [hasUnread, setHasUnread] = useState(false); + const [isLoading, setIsLoading] = useState(false); + const containerRef = useRef(null); const { notify } = useNotifications(); + useEffect(() => { + if (!isOpen) return; + + const handlePointerDown = (event: PointerEvent) => { + const target = event.target; + if (!(target instanceof Node)) return; + if (!containerRef.current?.contains(target)) setIsOpen(false); + }; + + document.addEventListener("pointerdown", handlePointerDown); + + return () => { + document.removeEventListener("pointerdown", handlePointerDown); + }; + }, [isOpen]); + + useEffect(() => { + if (!isOpen) return; + + let ignore = false; + setIsLoading(true); + + dashboardApi + .getOverview() + .then((overview) => { + if (!ignore) { + setNotifications(buildNotifications(overview)); + setHasUnread(false); + } + }) + .catch(() => { + if (!ignore) { + setNotifications([ + { + tone: "error", + title: "Dashboard indisponível", + description: "Não foi possível carregar notificações reais agora.", + timeAgo: "Agora", + }, + ]); + setHasUnread(false); + } + }) + .finally(() => { + if (!ignore) setIsLoading(false); + }); + + return () => { + ignore = true; + }; + }, [isOpen]); + const openNotification = (notification: Notification) => { notify({ tone: notification.tone, @@ -76,14 +178,21 @@ export function NotificationButton() { }; return ( -
+
{isOpen && ( @@ -91,11 +200,11 @@ export function NotificationButton() {
Notificações - {NOTIFICATIONS.length} Novas + {isLoading ? "Carregando" : `${notifications.length} reais`}
- {NOTIFICATIONS.map((n, i) => ( + {notifications.map((n, i) => (
diff --git a/front_admin/src/modules/scrapers/components/ScraperGrid/ScraperGrid.tsx b/front_admin/src/modules/scrapers/components/ScraperGrid/ScraperGrid.tsx index 220d105..41dbf40 100644 --- a/front_admin/src/modules/scrapers/components/ScraperGrid/ScraperGrid.tsx +++ b/front_admin/src/modules/scrapers/components/ScraperGrid/ScraperGrid.tsx @@ -3,22 +3,15 @@ import { ScraperCard } from "./ScraperCard"; interface ScraperGridProps { scrapers: Scraper[]; - isStarting: boolean; + startingScraperId: string | null; onToggle: (id: string) => void; - onStartAll: () => void; - onPauseAll: () => void; } export function ScraperGrid({ scrapers, - isStarting, + startingScraperId, onToggle, - onStartAll, - onPauseAll, }: ScraperGridProps) { - const hasRunningScraper = scrapers.some((scraper) => scraper.active); - const startDisabled = isStarting || hasRunningScraper; - return (
@@ -31,30 +24,9 @@ export function ScraperGrid({ de scraping de vagas

-
- - -
+ + Rode cada scraper pelo card +
@@ -64,7 +36,12 @@ export function ScraperGrid({
) : ( scrapers.map((scraper) => ( - + )) )}
diff --git a/front_admin/src/modules/scrapers/hooks/useScrapers.ts b/front_admin/src/modules/scrapers/hooks/useScrapers.ts index f9f57fa..922a7fe 100644 --- a/front_admin/src/modules/scrapers/hooks/useScrapers.ts +++ b/front_admin/src/modules/scrapers/hooks/useScrapers.ts @@ -180,7 +180,7 @@ export function useScrapers() { const [lastUpdatedAt, setLastUpdatedAt] = useState(null); const [isLoading, setIsLoading] = useState(true); const [isRefreshing, setIsRefreshing] = useState(false); - const [isStarting, setIsStarting] = useState(false); + const [startingScraperId, setStartingScraperId] = useState(null); const [isClearingJobsCache, setIsClearingJobsCache] = useState(false); const [error, setError] = useState(null); @@ -284,23 +284,56 @@ export function useScrapers() { lastUpdatedAt, }; - const toggleScraper = (id: string) => { + const runScraper = async (id: string) => { const scraper = scrapers.find((item) => item.id === id); if (!scraper) return; - addLog( - `Acao individual para ${scraper.name} ainda nao esta disponivel no backend.`, - ); - notify({ - tone: "warning", - title: "Ação indisponível", - description: - "O backend ainda não possui endpoint para pausar ou ativar um scraper individual.", - }); + if (scraper.active) { + addLog(`${scraper.name} ja esta em execucao.`); + notify({ + tone: "warning", + title: "Scraper já em execução", + description: `${scraper.name} já está rodando neste momento.`, + }); + return; + } + + setStartingScraperId(id); + try { + const result = await scrapersApi.triggerOne(id); + addLog(result.message || `${scraper.name} iniciado.`); + notify({ + tone: "success", + title: "Scraper iniciado", + description: result.message || `${scraper.name} começou a executar.`, + }); + await refresh({ includeJobs: true }); + } catch (error) { + if (error instanceof ApiError && error.status === 409) { + addLog(`${scraper.name} ja esta em execucao.`); + notify({ + tone: "warning", + title: "Scraper já em execução", + description: "A execução atual ainda não terminou.", + }); + await refresh({ includeJobs: true }); + return; + } + + setError(`Nao foi possivel iniciar ${scraper.name}.`); + addLog(`Falha ao solicitar execucao de ${scraper.name}.`); + notify({ + tone: "error", + title: "Erro ao iniciar scraper", + description: `O backend não conseguiu disparar ${scraper.name}.`, + }); + } finally { + setStartingScraperId(null); + } }; const startAll = async () => { - setIsStarting(true); + setStartingScraperId("__all__"); try { const result = await scrapersApi.trigger(); addLog(result.message || "Execucao dos scrapers iniciada."); @@ -331,7 +364,7 @@ export function useScrapers() { "O backend não conseguiu disparar a execução dos scrapers.", }); } finally { - setIsStarting(false); + setStartingScraperId(null); } }; @@ -386,11 +419,12 @@ export function useScrapers() { logs, isLoading, isRefreshing, - isStarting, + isStarting: startingScraperId !== null, + startingScraperId, isClearingJobsCache, error, refresh: reloadJobs, - toggleScraper, + toggleScraper: runScraper, startAll, pauseAll, clearJobsCache, diff --git a/front_admin/tests/app/layouts/MainLayout.test.tsx b/front_admin/tests/app/layouts/MainLayout.test.tsx index bf86a85..59a7246 100644 --- a/front_admin/tests/app/layouts/MainLayout.test.tsx +++ b/front_admin/tests/app/layouts/MainLayout.test.tsx @@ -2,6 +2,7 @@ import { fireEvent, screen } from "@testing-library/react"; import { Route, Routes } from "react-router-dom"; import { beforeEach, describe, expect, it, vi } from "vitest"; import { MainLayout } from "../../../src/app/layouts/MainLayout"; +import { dashboardApi } from "../../../src/lib/api/dashboard.api"; import { useAuth } from "../../../src/modules/auth/hooks/useAuth"; import { renderWithProviders } from "../../test-utils"; @@ -9,6 +10,12 @@ vi.mock("../../../src/modules/auth/hooks/useAuth", () => ({ useAuth: vi.fn(), })); +vi.mock("../../../src/lib/api/dashboard.api", () => ({ + dashboardApi: { + getOverview: vi.fn(), + }, +})); + describe("MainLayout", () => { const logout = vi.fn(); @@ -29,6 +36,33 @@ describe("MainLayout", () => { logout, hasPermission: vi.fn(), }); + vi.mocked(dashboardApi.getOverview).mockResolvedValue({ + stats: { + totalUsers: 10, + activeUsers: 10, + totalCollectedJobs: 2543, + jobsCollectedToday: 12, + }, + scrapers: [ + { + name: "go-scraper", + status: "running", + running: true, + lastRunAt: null, + jobsCollected: 2543, + }, + ], + services: { + status: "ok", + timestamp: "2026-01-01T10:00:00.000Z", + services: { + postgres: { status: "ok", latencyMs: 8 }, + valkey: { status: "ok", latencyMs: 6 }, + scraper: { status: "ok", latencyMs: 12 }, + }, + }, + generatedAt: "2026-01-01T10:00:00.000Z", + }); }); it("renders sidebar, header controls and outlet", async () => { @@ -56,9 +90,10 @@ describe("MainLayout", () => { fireEvent.click(screen.getByRole("button", { name: "Abrir notificações" })); expect(screen.getByText("Notificações")).toBeInTheDocument(); - fireEvent.click(screen.getByText("LinkedIn Scraper atingiu a meta")); + expect(await screen.findByText("go-scraper em execução")).toBeInTheDocument(); + fireEvent.click(screen.getByText("go-scraper em execução")); expect( - await screen.findAllByText("2.543 vagas processadas com sucesso hoje."), + await screen.findAllByText("O scraper está rodando neste momento."), ).toHaveLength(2); fireEvent.click(screen.getByText("Ada Lovelace")); diff --git a/front_admin/tests/modules/dashboard/DashboardPage.test.tsx b/front_admin/tests/modules/dashboard/DashboardPage.test.tsx index 2b2a824..52e6505 100644 --- a/front_admin/tests/modules/dashboard/DashboardPage.test.tsx +++ b/front_admin/tests/modules/dashboard/DashboardPage.test.tsx @@ -24,10 +24,10 @@ const dashboardState = { { id: "adzuna", name: "Adzuna", - status: "Online", + status: "Disponivel", lastRun: "Hoje", collected24h: 1540, - active: true, + active: false, }, ], chartPoints: [ @@ -63,13 +63,13 @@ describe("DashboardPage", () => { expect(screen.getByText("Monitoramento em tempo real")).toBeInTheDocument(); expect(screen.getByText("Total de Usuários")).toBeInTheDocument(); - expect(screen.getByText("Visão Geral da Plataforma")).toBeInTheDocument(); + expect(screen.getByText("Crescimento do índice")).toBeInTheDocument(); expect(screen.getByText("Status dos Serviços")).toBeInTheDocument(); fireEvent.click(screen.getByRole("button", { name: "Atualizar" })); expect(dashboardState.refresh).toHaveBeenCalledTimes(1); - fireEvent.click(screen.getByTitle("Pausar Scraper")); + fireEvent.click(screen.getByTitle("Iniciar Scraper")); expect(dashboardState.toggleScraper).toHaveBeenCalledWith("adzuna"); }); diff --git a/front_admin/tests/modules/dashboard/components/PlatformChart.test.tsx b/front_admin/tests/modules/dashboard/components/PlatformChart.test.tsx index 9308349..97aad3c 100644 --- a/front_admin/tests/modules/dashboard/components/PlatformChart.test.tsx +++ b/front_admin/tests/modules/dashboard/components/PlatformChart.test.tsx @@ -3,10 +3,10 @@ import { describe, expect, it } from "vitest"; import { PlatformChart } from "../../../../src/modules/dashboard/components/PlatformOverview/PlatformChart"; describe("PlatformChart", () => { - it("renders empty, single and dense point states", () => { + it("renders empty, single and dense point states with index metrics", () => { const { rerender, container } = render(); expect( - screen.getByText("Aguardando novos snapshots para formar a tendência"), + screen.getByText("Aguardando o primeiro snapshot do dashboard"), ).toBeInTheDocument(); rerender( @@ -21,7 +21,12 @@ describe("PlatformChart", () => { ]} />, ); - expect(screen.getByText("10:00")).toBeInTheDocument(); + expect(screen.getByText("Total indexado")).toBeInTheDocument(); + expect(screen.getAllByText("5")).toHaveLength(2); + expect(screen.getByText("Índice estável")).toBeInTheDocument(); + expect( + screen.getByText("Mais um snapshot é necessário para calcular variação."), + ).toBeInTheDocument(); rerender( { }))} />, ); - expect(container.querySelectorAll("circle")).toHaveLength(16); + expect(screen.getByText("Estado do índice")).toBeInTheDocument(); + expect(screen.getByText("+1")).toBeInTheDocument(); + expect(container.querySelectorAll("circle")).toHaveLength(0); }); }); diff --git a/front_admin/tests/modules/dashboard/components/ScraperTable.test.tsx b/front_admin/tests/modules/dashboard/components/ScraperTable.test.tsx index 364404d..97718e1 100644 --- a/front_admin/tests/modules/dashboard/components/ScraperTable.test.tsx +++ b/front_admin/tests/modules/dashboard/components/ScraperTable.test.tsx @@ -6,9 +6,14 @@ import { ScraperTable } from "../../../../src/modules/dashboard/components/Runni describe("ScraperTable", () => { it("renders empty state and active rows", async () => { const onToggle = vi.fn(); + const onConfigure = vi.fn(); const { rerender } = render( - + , ); @@ -28,6 +33,7 @@ describe("ScraperTable", () => { }, ]} onToggle={onToggle} + onConfigure={onConfigure} /> , ); @@ -35,8 +41,7 @@ describe("ScraperTable", () => { fireEvent.click(screen.getByTitle("Iniciar Scraper")); expect(onToggle).toHaveBeenCalledWith("s1"); - fireEvent.click(screen.getByRole("button", { name: "" })); - expect(await screen.findByText("Configurações em desenvolvimento")) - .toBeInTheDocument(); + fireEvent.click(screen.getByTitle("Ver detalhes de Scraper")); + expect(onConfigure).toHaveBeenCalledWith("s1"); }); }); diff --git a/front_admin/tests/modules/dashboard/hooks/dashboard-hooks.test.tsx b/front_admin/tests/modules/dashboard/hooks/dashboard-hooks.test.tsx index af201a7..1d38c63 100644 --- a/front_admin/tests/modules/dashboard/hooks/dashboard-hooks.test.tsx +++ b/front_admin/tests/modules/dashboard/hooks/dashboard-hooks.test.tsx @@ -1,5 +1,7 @@ import { act, renderHook, waitFor } from "@testing-library/react"; +import type { ReactNode } from "react"; import { beforeEach, describe, expect, it, vi } from "vitest"; +import { NotificationProvider } from "../../../../src/components/notifications/NotificationProvider"; import { dashboardService } from "../../../../src/modules/dashboard/services/dashboard.service"; import { useDashboard } from "../../../../src/modules/dashboard/hooks/useDashboard"; import { useDashboardMetrics } from "../../../../src/modules/dashboard/hooks/useDashboardMetrics"; @@ -38,6 +40,10 @@ const scrapers = [ }, ]; +function wrapper({ children }: { children: ReactNode }) { + return {children}; +} + describe("dashboard hooks", () => { beforeEach(() => { vi.mocked(dashboardService.getOverview).mockResolvedValue({ @@ -51,10 +57,11 @@ describe("dashboard hooks", () => { vi.mocked(dashboardService.getResources).mockResolvedValue(resources); vi.mocked(dashboardService.getServices).mockResolvedValue(services); vi.mocked(dashboardService.getScrapersSummary).mockResolvedValue(scrapers); + vi.mocked(dashboardService.toggleScraper).mockResolvedValue(undefined); }); it("loads dashboard overview and toggles scrapers", async () => { - const { result } = renderHook(() => useDashboard()); + const { result } = renderHook(() => useDashboard(), { wrapper }); await waitFor(() => expect(result.current.isLoading).toBe(false)); expect(result.current.chartPoints).toHaveLength(1); @@ -62,11 +69,14 @@ describe("dashboard hooks", () => { result.current.toggleScraper("adzuna"); expect(dashboardService.toggleScraper).toHaveBeenCalledWith("adzuna", false); + await waitFor(() => + expect(dashboardService.getOverview).toHaveBeenCalledTimes(2), + ); result.current.toggleScraper("missing"); expect(dashboardService.toggleScraper).toHaveBeenCalledTimes(1); - vi.mocked(dashboardService.getOverview).mockResolvedValueOnce({ + vi.mocked(dashboardService.getOverview).mockResolvedValue({ stats: { ...stats, totalJobs: { value: 101, trend: "ok", positive: true } }, resources, services, @@ -83,7 +93,7 @@ describe("dashboard hooks", () => { it("handles dashboard refresh failures", async () => { vi.mocked(dashboardService.getOverview).mockRejectedValueOnce(new Error("fail")); - const { result } = renderHook(() => useDashboard()); + const { result } = renderHook(() => useDashboard(), { wrapper }); await waitFor(() => expect(result.current.isLoading).toBe(false)); expect(result.current.error).toBe( diff --git a/front_admin/tests/modules/dashboard/services/dashboard.service.test.ts b/front_admin/tests/modules/dashboard/services/dashboard.service.test.ts index e8ed45f..7679d16 100644 --- a/front_admin/tests/modules/dashboard/services/dashboard.service.test.ts +++ b/front_admin/tests/modules/dashboard/services/dashboard.service.test.ts @@ -1,5 +1,6 @@ import { beforeEach, describe, expect, it, vi } from "vitest"; import { dashboardApi } from "../../../../src/lib/api/dashboard.api"; +import { scrapersApi } from "../../../../src/lib/api/scrapers.api"; import { dashboardService } from "../../../../src/modules/dashboard/services/dashboard.service"; vi.mock("../../../../src/lib/api/dashboard.api", () => ({ @@ -8,6 +9,12 @@ vi.mock("../../../../src/lib/api/dashboard.api", () => ({ }, })); +vi.mock("../../../../src/lib/api/scrapers.api", () => ({ + scrapersApi: { + triggerOne: vi.fn(), + }, +})); + const backendOverview = { stats: { totalUsers: 10, @@ -46,6 +53,11 @@ const backendOverview = { describe("dashboardService", () => { beforeEach(() => { vi.mocked(dashboardApi.getOverview).mockResolvedValue(backendOverview); + vi.mocked(scrapersApi.triggerOne).mockResolvedValue({ + ok: true, + message: "iniciado", + scraper: "x", + }); }); it("maps backend overview to dashboard view model", async () => { @@ -71,7 +83,7 @@ describe("dashboardService", () => { }); }); - it("exposes segmented getters and noop toggle", async () => { + it("exposes segmented getters and real scraper trigger", async () => { await expect(dashboardService.getStats()).resolves.toMatchObject({ activeUsers: { value: 8 }, }); @@ -81,5 +93,9 @@ describe("dashboardService", () => { await expect(dashboardService.getServices()).resolves.toHaveLength(3); await expect(dashboardService.getScrapersSummary()).resolves.toHaveLength(2); await expect(dashboardService.toggleScraper("x", true)).resolves.toBeUndefined(); + expect(scrapersApi.triggerOne).toHaveBeenCalledWith("x"); + await expect(dashboardService.toggleScraper("x", false)).rejects.toThrow( + "pausar scraper", + ); }); }); diff --git a/front_admin/tests/modules/scrapers/ScrapersPage.test.tsx b/front_admin/tests/modules/scrapers/ScrapersPage.test.tsx index fc70d8f..8740858 100644 --- a/front_admin/tests/modules/scrapers/ScrapersPage.test.tsx +++ b/front_admin/tests/modules/scrapers/ScrapersPage.test.tsx @@ -62,12 +62,11 @@ const scraperState = { isLoading: false, isRefreshing: false, isStarting: false, + startingScraperId: null, isClearingJobsCache: false, error: null, refresh: vi.fn(), toggleScraper: vi.fn(), - startAll: vi.fn(), - pauseAll: vi.fn(), clearJobsCache: vi.fn(), clearLogs: vi.fn(), refreshIntervalMs: 15_000, @@ -106,14 +105,12 @@ describe("ScrapersPage", () => { fireEvent.click(screen.getByRole("button", { name: "Recarregar dados" })); fireEvent.click(screen.getByRole("button", { name: /limpar cache de vagas/i })); - fireEvent.click(screen.getByRole("button", { name: /iniciar todos/i })); - fireEvent.click(screen.getByRole("button", { name: /pausar todos/i })); + fireEvent.click(screen.getByRole("button", { name: "Run" })); fireEvent.click(screen.getByRole("button", { name: /limpar logs/i })); expect(scraperState.refresh).toHaveBeenCalledTimes(1); expect(scraperState.clearJobsCache).toHaveBeenCalledTimes(1); - expect(scraperState.startAll).toHaveBeenCalledTimes(1); - expect(scraperState.pauseAll).toHaveBeenCalledTimes(1); + expect(scraperState.toggleScraper).toHaveBeenCalledWith("lever"); expect(scraperState.clearLogs).toHaveBeenCalledTimes(1); }); diff --git a/front_admin/tests/modules/scrapers/components/ScraperGrid.test.tsx b/front_admin/tests/modules/scrapers/components/ScraperGrid.test.tsx index bf07e68..852c459 100644 --- a/front_admin/tests/modules/scrapers/components/ScraperGrid.test.tsx +++ b/front_admin/tests/modules/scrapers/components/ScraperGrid.test.tsx @@ -8,14 +8,12 @@ describe("ScraperGrid", () => { const { rerender } = render( , ); expect(screen.getByText("Nenhum scraper retornado pelo backend.")).toBeInTheDocument(); - expect(screen.getByRole("button", { name: "Iniciando..." })).toBeDisabled(); + expect(screen.getByText("Rode cada scraper pelo card")).toBeInTheDocument(); rerender( { sla: "Indisponivel", }, ]} - isStarting={false} + startingScraperId={null} onToggle={onToggle} - onStartAll={vi.fn()} - onPauseAll={vi.fn()} />, ); - expect(screen.getByRole("button", { name: "Em execução" })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Rodando" })).toBeDisabled(); - fireEvent.click(screen.getByRole("button", { name: "Desativar" })); - fireEvent.click(screen.getByRole("button", { name: "Ativar" })); + fireEvent.click(screen.getByRole("button", { name: "Run" })); - expect(onToggle).toHaveBeenNthCalledWith(1, "a"); - expect(onToggle).toHaveBeenNthCalledWith(2, "b"); + expect(onToggle).toHaveBeenCalledWith("b"); }); }); diff --git a/front_admin/tests/modules/scrapers/hooks/useScrapers.test.tsx b/front_admin/tests/modules/scrapers/hooks/useScrapers.test.tsx index d59a5e4..144e802 100644 --- a/front_admin/tests/modules/scrapers/hooks/useScrapers.test.tsx +++ b/front_admin/tests/modules/scrapers/hooks/useScrapers.test.tsx @@ -12,6 +12,7 @@ vi.mock("../../../../src/lib/api/scrapers.api", () => ({ jobsCount: vi.fn(), jobs: vi.fn(), trigger: vi.fn(), + triggerOne: vi.fn(), clearJobsCache: vi.fn(), }, })); @@ -77,6 +78,11 @@ describe("useScrapers", () => { ok: true, message: "Execução iniciada", }); + vi.mocked(scrapersApi.triggerOne).mockResolvedValue({ + ok: true, + message: "Execução individual iniciada", + scraper: "Adzuna", + }); vi.mocked(scrapersApi.clearJobsCache).mockResolvedValue({ ok: true, deleted: 4, @@ -98,8 +104,11 @@ describe("useScrapers", () => { expect(result.current.adapterStats[0].jobs).toBeGreaterThan(0); expect(result.current.jobPreviews[0].title).toBe("Frontend"); - act(() => result.current.toggleScraper("Adzuna")); - expect(result.current.logs[0].text).toContain("ainda nao esta disponivel"); + await act(async () => { + await result.current.toggleScraper("Adzuna"); + }); + expect(scrapersApi.triggerOne).toHaveBeenCalledWith("Adzuna"); + expect(result.current.logs[0].text).toBe("Execução individual iniciada"); act(() => result.current.pauseAll()); expect(result.current.logs[0].text).toContain("Pausar scrapers"); @@ -147,6 +156,18 @@ describe("useScrapers", () => { expect(result.current.error).toBe("Nao foi possivel iniciar os scrapers."); }); + it("handles individual scraper trigger failures", async () => { + const { result } = renderHook(() => useScrapers(), { wrapper }); + await waitFor(() => expect(result.current.isLoading).toBe(false)); + + vi.mocked(scrapersApi.triggerOne).mockRejectedValueOnce(new Error("fail")); + await act(async () => { + await result.current.toggleScraper("Adzuna"); + }); + + expect(result.current.error).toBe("Nao foi possivel iniciar Adzuna."); + }); + it("handles already running scraper trigger as warning", async () => { const { result } = renderHook(() => useScrapers(), { wrapper }); await waitFor(() => expect(result.current.isLoading).toBe(false)); diff --git a/frontend/README.md b/frontend/README.md index 89f4cfc..846ae4f 100644 --- a/frontend/README.md +++ b/frontend/README.md @@ -1,6 +1,6 @@ # Frontend -Aplicação web principal do , voltada para usuários finais. +Aplicação web principal do Cand!Date!, voltada para usuários finais. Ela concentra a landing page pública, autenticação, callback OAuth e dashboard de vagas com filtros, detalhes, vagas salvas, perfil e preferências. @@ -33,7 +33,7 @@ Para executar apenas este workspace: npm run dev --workspace=frontend ``` -Por padrão o Vite usa http://localhost:5173. +Por padrão o Vite usa . ## Variáveis de ambiente diff --git a/frontend/src/domains/new_dashboard/constants/initialData.ts b/frontend/src/domains/new_dashboard/constants/initialData.ts index a200cb2..f2a363e 100644 --- a/frontend/src/domains/new_dashboard/constants/initialData.ts +++ b/frontend/src/domains/new_dashboard/constants/initialData.ts @@ -209,8 +209,8 @@ export const initialUser: UserProfile = { export const initialPreferences: SearchPreferences = { keywords: ["React", "Frontend", "Fullstack"], searchLocation: "São Paulo, SP", - remoteOnly: true, - jobTypes: ["Remoto"], + remoteOnly: false, + jobTypes: [], emailNotifications: true, careerChecklist: [], }; diff --git a/package-lock.json b/package-lock.json index 61d1c8f..9bf79e8 100644 --- a/package-lock.json +++ b/package-lock.json @@ -35,7 +35,7 @@ "@commitlint/config-conventional": "^21.0.2", "@types/cors": "^2.8.19", "@types/express": "^5.0.6", - "@types/node": "^25.9.2", + "@types/node": "^25.9.5", "@types/pdfkit": "^0.17.6", "@types/pg": "^8.20.0", "@types/pino": "^7.0.5", @@ -47,6 +47,7 @@ "electron-builder": "^26.15.0", "husky": "^9.1.7", "lint-staged": "^17.0.7", + "tsx": "^4.23.1", "typescript": "^5.9.3" } }, @@ -18009,9 +18010,9 @@ "license": "0BSD" }, "node_modules/tsx": { - "version": "4.23.0", - "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.0.tgz", - "integrity": "sha512-eUdUIaCr963q2h5u3+QwvYp0+eqPvn+egeqZUm0hwERCqqx1E3kK5ehbGCvqSE5MQAULr67ww0cA3jKc3YkM1w==", + "version": "4.23.1", + "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.1.tgz", + "integrity": "sha512-GQHnkIfxyx1wYCOS/wonik5MVRZU9hi1TEZmzGZSCJB1y9YgoZ8H6itNE/u4suE+yLmOzuE4E5S4TZ/ZX2wcWQ==", "devOptional": true, "license": "MIT", "dependencies": { diff --git a/package.json b/package.json index 84482cd..2e3f9e4 100644 --- a/package.json +++ b/package.json @@ -31,7 +31,13 @@ "dist": "npm run build:frontend && electron-builder", "db:generate": "drizzle-kit generate", "db:migrate": "drizzle-kit migrate", - "db:push": "drizzle-kit push" + "db:push": "drizzle-kit push", + "discover:greenhouse": "tsx tools/greenhouse-discovery/discover.ts", + "discover:inhire": "tsx tools/inhire-discovery/discover.ts", + "discover:lever": "tsx tools/lever-discovery/discover.ts", + "typecheck:inhire": "tsc -p tools/inhire-discovery/tsconfig.json", + "typecheck:lever": "tsc -p tools/lever-discovery/tsconfig.json", + "discover:ats": "npm run discover:greenhouse && npm run discover:lever && npm run discover:inhire" }, "dependencies": { "@vitest/coverage-v8": "4.1.8", @@ -56,7 +62,7 @@ "@commitlint/config-conventional": "^21.0.2", "@types/cors": "^2.8.19", "@types/express": "^5.0.6", - "@types/node": "^25.9.2", + "@types/node": "^25.9.5", "@types/pdfkit": "^0.17.6", "@types/pg": "^8.20.0", "@types/pino": "^7.0.5", @@ -68,6 +74,7 @@ "electron-builder": "^26.15.0", "husky": "^9.1.7", "lint-staged": "^17.0.7", + "tsx": "^4.23.1", "typescript": "^5.9.3" }, "lint-staged": { diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index 43b5363..c690346 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -1,6 +1,7 @@ packages: - "frontend" - "backend" + - "front_admin" allowBuilds: "@scarf/scarf": true @@ -14,19 +15,19 @@ minimumReleaseAgeExclude: - xlsx@0.19.3 - xlsx@0.20.2 overrides: - electron@<35.7.5: '>=35.7.5' - electron@<38.8.6: '>=38.8.6' - electron@<39.8.1: '>=39.8.1' - electron@<39.8.5: '>=39.8.5' - electron@>=33.0.0-alpha.1 <39.8.5: '>=39.8.5' - esbuild@<=0.24.2: '>=0.25.0' - tar@<7.5.7: '>=7.5.7' - tar@<7.5.8: '>=7.5.8' - tar@<=7.5.10: '>=7.5.11' - tar@<=7.5.2: '>=7.5.3' - tar@<=7.5.3: '>=7.5.4' - tar@<=7.5.9: '>=7.5.10' - tmp@<0.2.6: '>=0.2.6' - vitest@<4.1.0: '>=4.1.0' + electron@<35.7.5: ">=35.7.5" + electron@<38.8.6: ">=38.8.6" + electron@<39.8.1: ">=39.8.1" + electron@<39.8.5: ">=39.8.5" + electron@>=33.0.0-alpha.1 <39.8.5: ">=39.8.5" + esbuild@<=0.24.2: ">=0.25.0" + tar@<7.5.7: ">=7.5.7" + tar@<7.5.8: ">=7.5.8" + tar@<=7.5.10: ">=7.5.11" + tar@<=7.5.2: ">=7.5.3" + tar@<=7.5.3: ">=7.5.4" + tar@<=7.5.9: ">=7.5.10" + tmp@<0.2.6: ">=0.2.6" + vitest@<4.1.0: ">=4.1.0" -"pnpm": { "allowBuilds": [ "@scarf/scarf", "electron", "esbuild" ] } +"pnpm": { "allowBuilds": ["@scarf/scarf", "electron", "esbuild"] } diff --git a/scraper-go/cmd/greenhouse-discover/main.go b/scraper-go/cmd/greenhouse-discover/main.go new file mode 100644 index 0000000..5680670 --- /dev/null +++ b/scraper-go/cmd/greenhouse-discover/main.go @@ -0,0 +1,277 @@ +package main + +import ( + "context" + "encoding/json" + "flag" + "fmt" + "net/http" + "net/url" + "os" + "sort" + "strings" + "time" + "unicode" + + "golang.org/x/text/transform" + "golang.org/x/text/unicode/norm" +) + +type boardResult struct { + Token string `json:"token"` + Name string `json:"name,omitempty"` + URL string `json:"url"` + APIURL string `json:"apiUrl"` + Valid bool `json:"valid"` + Jobs int `json:"jobs,omitempty"` + Status int `json:"status,omitempty"` + Error string `json:"error,omitempty"` + Source string `json:"source,omitempty"` + Company string `json:"company,omitempty"` +} + +type greenhouseJobsResponse struct { + Jobs []struct { + Title string `json:"title"` + } `json:"jobs"` + Meta struct { + Total int `json:"total"` + } `json:"meta"` +} + +func main() { + var ( + namesArg = flag.String("names", "", "nomes de empresas separados por vírgula, ex: Reddit,GitLab") + tokensArg = flag.String("tokens", "", "board tokens separados por vírgula, ex: reddit,gitlab") + filePath = flag.String("file", "", "arquivo JSON com array de nomes/tokens") + out = flag.String("out", "text", "formato de saída: text ou json") + timeout = flag.Duration("timeout", 10*time.Second, "timeout por token") + ) + flag.Parse() + + nameInputs := splitCSV(*namesArg) + exactTokens := splitCSV(*tokensArg) + if *filePath != "" { + fromFile, err := readStringArray(*filePath) + if err != nil { + fmt.Fprintf(os.Stderr, "erro ao ler arquivo: %v\n", err) + os.Exit(1) + } + exactTokens = append(exactTokens, fromFile...) + } + nameInputs = uniqueStrings(append(nameInputs, flag.Args()...)) + exactTokens = uniqueStrings(exactTokens) + + if len(nameInputs) == 0 && len(exactTokens) == 0 { + fmt.Fprintln(os.Stderr, "uso: greenhouse-discover -names Reddit,GitLab ou -file internal/interfaces/greenhouseCompanies.json") + os.Exit(2) + } + + client := &http.Client{Timeout: *timeout} + var results []boardResult + seenTokens := make(map[string]struct{}) + + for _, token := range exactTokens { + token = strings.TrimSpace(strings.ToLower(token)) + if token == "" { + continue + } + if _, seen := seenTokens[token]; seen { + continue + } + seenTokens[token] = struct{}{} + + ctx, cancel := context.WithTimeout(context.Background(), *timeout) + result := checkBoard(ctx, client, token, token) + cancel() + results = append(results, result) + } + + for _, input := range nameInputs { + for _, token := range candidateTokens(input) { + if _, seen := seenTokens[token]; seen { + continue + } + seenTokens[token] = struct{}{} + + ctx, cancel := context.WithTimeout(context.Background(), *timeout) + result := checkBoard(ctx, client, input, token) + cancel() + results = append(results, result) + } + } + + sort.Slice(results, func(i, j int) bool { + if results[i].Valid != results[j].Valid { + return results[i].Valid + } + return results[i].Token < results[j].Token + }) + + switch strings.ToLower(strings.TrimSpace(*out)) { + case "json": + encoder := json.NewEncoder(os.Stdout) + encoder.SetIndent("", " ") + if err := encoder.Encode(results); err != nil { + fmt.Fprintf(os.Stderr, "erro ao escrever json: %v\n", err) + os.Exit(1) + } + default: + printText(results) + } +} + +func checkBoard(ctx context.Context, client *http.Client, company, token string) boardResult { + apiURL := fmt.Sprintf("https://boards-api.greenhouse.io/v1/boards/%s/jobs", url.PathEscape(token)) + result := boardResult{ + Token: token, + URL: fmt.Sprintf("https://job-boards.greenhouse.io/%s", token), + APIURL: apiURL, + Company: company, + Source: "boards-api", + } + + req, err := http.NewRequestWithContext(ctx, http.MethodGet, apiURL, nil) + if err != nil { + result.Error = err.Error() + return result + } + req.Header.Set("Accept", "application/json") + req.Header.Set("User-Agent", "JobsScraper/greenhouse-discover") + + resp, err := client.Do(req) + if err != nil { + result.Error = err.Error() + return result + } + defer resp.Body.Close() + + result.Status = resp.StatusCode + if resp.StatusCode != http.StatusOK { + return result + } + + var data greenhouseJobsResponse + if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { + result.Error = err.Error() + return result + } + + result.Valid = true + result.Jobs = data.Meta.Total + if result.Jobs == 0 { + result.Jobs = len(data.Jobs) + } + + return result +} + +func candidateTokens(input string) []string { + normalized := normalize(input) + words := strings.Fields(normalized) + if len(words) == 0 { + return nil + } + + joined := strings.Join(words, "") + hyphenated := strings.Join(words, "-") + underscored := strings.Join(words, "_") + + return uniqueStrings([]string{ + strings.TrimSpace(strings.ToLower(input)), + joined, + hyphenated, + underscored, + joined + "careers", + joined + "jobs", + }) +} + +func normalize(input string) string { + t := transform.Chain(norm.NFD, transform.RemoveFunc(func(r rune) bool { + return unicode.Is(unicode.Mn, r) + }), norm.NFC) + + value, _, _ := transform.String(t, strings.ToLower(input)) + + var b strings.Builder + for _, r := range value { + if unicode.IsLetter(r) || unicode.IsNumber(r) { + b.WriteRune(r) + } else { + b.WriteRune(' ') + } + } + + return strings.Join(strings.Fields(b.String()), " ") +} + +func readStringArray(path string) ([]string, error) { + data, err := os.ReadFile(path) + if err != nil { + return nil, err + } + + var values []string + if err := json.Unmarshal(data, &values); err != nil { + return nil, err + } + + return values, nil +} + +func splitCSV(value string) []string { + parts := strings.Split(value, ",") + out := make([]string, 0, len(parts)) + for _, part := range parts { + part = strings.TrimSpace(part) + if part != "" { + out = append(out, part) + } + } + return out +} + +func uniqueStrings(values []string) []string { + seen := make(map[string]struct{}, len(values)) + out := make([]string, 0, len(values)) + for _, value := range values { + value = strings.TrimSpace(value) + if value == "" { + continue + } + key := strings.ToLower(value) + if _, ok := seen[key]; ok { + continue + } + seen[key] = struct{}{} + out = append(out, value) + } + return out +} + +func printText(results []boardResult) { + validCount := 0 + for _, result := range results { + if !result.Valid { + continue + } + validCount++ + fmt.Printf("OK %-32s %5d vagas %s\n", result.Token, result.Jobs, result.URL) + } + + if validCount > 0 { + fmt.Println() + } + + for _, result := range results { + if result.Valid { + continue + } + detail := fmt.Sprintf("status=%d", result.Status) + if result.Error != "" { + detail = "erro=" + result.Error + } + fmt.Printf("MISS %-32s %s\n", result.Token, detail) + } +} diff --git a/scraper-go/cmd/server/adapters.go b/scraper-go/cmd/server/adapters.go index 933fc93..b084daf 100644 --- a/scraper-go/cmd/server/adapters.go +++ b/scraper-go/cmd/server/adapters.go @@ -1,12 +1,11 @@ package main import ( - "context" - "log" "log/slog" "os" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" "github.com/joho/godotenv" "github.com/redis/go-redis/v9" ) @@ -30,62 +29,23 @@ func loadEnv() { slog.Warn("arquivo .env não encontrado, usando variáveis do sistema") } -func resolveInterfacesPath(filename string) string { - if v := os.Getenv("INTERFACES_DIR"); v != "" { - return v + "/" + filename - } - candidates := []string{ - "internal/interfaces/" + filename, - "../internal/interfaces/" + filename, - "../../internal/interfaces/" + filename, - } - for _, c := range candidates { - if _, err := os.Stat(c); err == nil { - return c - } - } - return "internal/interfaces/" + filename -} - -func buildAdapters(rdb *redis.Client) []adapters.Adapter { - all := make([]adapters.Adapter, 0) - - all = append(all, adapters.NewLinkedIn()) - - all = append(all, adapters.NewAdzuna( - os.Getenv("ADZUNA_APP_ID"), - os.Getenv("ADZUNA_APP_KEY"), - "br", - )) - - if err := os.Setenv("GREENHOUSE_COMPANIES_FILE", resolveInterfacesPath("greenhouseCompanies.json")); err != nil { - slog.Warn("falha ao setar GREENHOUSE_COMPANIES_FILE", "error", err) - } - greenhouseSlugs, err := adapters.FetchGreenhouseSlugs(context.Background()) - if err != nil { - slog.Warn("falha ao buscar slugs do Greenhouse", "error", err) - } - for _, slug := range greenhouseSlugs { - all = append(all, adapters.NewGreenhouse(slug, slug)) - } - - if err := os.Setenv("LEVER_COMPANIES_FILE", resolveInterfacesPath("leverCompanies.json")); err != nil { - slog.Warn("falha ao setar LEVER_COMPANIES_FILE", "error", err) - } - leverCompanies, err := adapters.FetchLeverSlugs(context.Background()) - if err != nil { - slog.Warn("falha ao carregar leverCompanies.json", "error", err) - } else { - for _, c := range leverCompanies { - all = append(all, adapters.NewLever(c.Slug, c.Name)) - } - } - - if joobleKey := os.Getenv("JOOBLE_API_KEY"); joobleKey != "" { - all = append(all, adapters.NewJooble(joobleKey, rdb)) - } else { - log.Println("AVISO: JOOBLE_API_KEY não encontrada") - } - - return all +// func resolveInterfacesPath(filename string) string { +// if v := os.Getenv("INTERFACES_DIR"); v != "" { +// return v + "/" + filename +// } +// candidates := []string{ +// "internal/interfaces/" + filename, +// "../internal/interfaces/" + filename, +// "../../internal/interfaces/" + filename, +// } +// for _, c := range candidates { +// if _, err := os.Stat(c); err == nil { +// return c +// } +// } +// return "internal/interfaces/" + filename +// } + +func buildAdapters(rdb *redis.Client) []ports.JobSource { + return adapters.GetAdapters(rdb) } diff --git a/scraper-go/cmd/server/admin_handlers.go b/scraper-go/cmd/server/admin_handlers.go index 7de843c..be67d0b 100644 --- a/scraper-go/cmd/server/admin_handlers.go +++ b/scraper-go/cmd/server/admin_handlers.go @@ -1,9 +1,11 @@ package main import ( + "context" "encoding/json" "net/http" "strconv" + "time" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cronjob" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" @@ -13,7 +15,9 @@ import ( // retorna 409 se já houver uma execução em andamento. func handleTriggerScrape(scheduler *cronjob.Scheduler) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { - if err := scheduler.RunNow(r.Context()); err != nil { + // A execução manual precisa sobreviver ao fim da request HTTP; se usarmos + // r.Context(), o scraper nasce com contexto cancelado assim que respondemos. + if err := scheduler.RunNow(context.Background()); err != nil { if err == cronjob.ErrAlreadyRunning { w.WriteHeader(http.StatusConflict) json.NewEncoder(w).Encode(map[string]any{ @@ -37,9 +41,19 @@ func handleTriggerScrape(scheduler *cronjob.Scheduler) http.HandlerFunc { // handleScraperStatus retorna o estado atual do scheduler via GET /admin/scrape/status func handleScraperStatus(scheduler *cronjob.Scheduler) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { + running, lastRunAt, jobsCollected := scheduler.Snapshot() + var lastRunAtValue *string + if !lastRunAt.IsZero() { + formatted := lastRunAt.Format(time.RFC3339) + lastRunAtValue = &formatted + } + w.Header().Set("Content-Type", "application/json") json.NewEncoder(w).Encode(map[string]any{ - "running": scheduler.IsRunning(), + "name": "go-scraper", + "running": running, + "lastRunAt": lastRunAtValue, + "jobsCollected": jobsCollected, }) } } diff --git a/scraper-go/cmd/server/handlers.go b/scraper-go/cmd/server/handlers.go index 721a8ae..98448aa 100644 --- a/scraper-go/cmd/server/handlers.go +++ b/scraper-go/cmd/server/handlers.go @@ -8,11 +8,11 @@ import ( "github.com/redis/go-redis/v9" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cache" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/pipeline" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" ) const ( @@ -20,9 +20,9 @@ const ( scrapeTimeout = 15 * time.Minute ) -func handleScrape(adapterList []adapters.Adapter, kwStore *keywords.Store, c cache.Cache, rdb *redis.Client) http.HandlerFunc { +func handleScrape(adapterList []ports.JobSource, kwStore *keywords.Store, c cache.Cache, rdb *redis.Client) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { - var req models.ScrapeRequest + var req domain.ScrapeRequest if err := json.NewDecoder(r.Body).Decode(&req); err != nil { http.Error(w, "invalid json body", http.StatusBadRequest) return @@ -41,16 +41,24 @@ func handleScrape(adapterList []adapters.Adapter, kwStore *keywords.Store, c cac defer cancel() config := pipeline.SearchConfig{ - Keywords: req.Keywords, - SearchLocation: req.SearchLocation, - JobTypes: req.JobTypes, - TimeFilter: req.TimeFilter, - RemoteOnly: req.RemoteOnly, + Keywords: req.Keywords, + SearchLocation: req.SearchLocation, + SearchGeoID: req.SearchGeoID, + SearchLanguage: req.SearchLanguage, + JobTypes: req.JobTypes, + TimeFilter: req.TimeFilter, + RemoteOnly: req.RemoteOnly, + Sources: req.Sources, + ResultsPerPage: req.ResultsPerPage, + MaxPagesPerKeyword: req.MaxPagesPerKeyword, + WaitBetweenSearchesMs: req.WaitBetweenSearchesMs, + PageTimeoutMs: req.PageTimeoutMs, + MaxConcurrency: req.MaxConcurrency, } start := time.Now() - result, err := pipeline.SearchJobs(ctx, c, config, scrapeTTL, rdb) // ← rdb aqui + result, err := pipeline.SearchJobs(ctx, c, config, adapterList, scrapeTTL, rdb) if err != nil { http.Error(w, "Erro ao buscar vagas.", http.StatusInternalServerError) return @@ -60,7 +68,7 @@ func handleScrape(adapterList []adapters.Adapter, kwStore *keywords.Store, c cac w.Header().Set("Content-Type", "application/json") w.Header().Set("X-Cache", cacheHeader(result.FromCache)) - json.NewEncoder(w).Encode(models.ScrapeResponse{ + json.NewEncoder(w).Encode(domain.ScrapeResponse{ Jobs: result.Jobs, Total: result.Total, CachedAt: result.CachedAt.UTC().Format(time.RFC3339), diff --git a/scraper-go/cmd/server/server.go b/scraper-go/cmd/server/server.go index 6ee4bf6..fabd9e4 100644 --- a/scraper-go/cmd/server/server.go +++ b/scraper-go/cmd/server/server.go @@ -10,16 +10,16 @@ import ( "syscall" "time" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cache" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cronjob" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" "github.com/prometheus/client_golang/prometheus/promhttp" "github.com/redis/go-redis/v9" ) -func run(adapterList []adapters.Adapter) { +func run(adapterList []ports.JobSource) { addr := os.Getenv("GO_SCRAPER_ADDR") if addr == "" { addr = ":8081" @@ -44,7 +44,7 @@ func run(adapterList []adapters.Adapter) { // ── Scheduler (cronjob) ── schedulerCfg := cronjob.DefaultConfig() - scheduler := cronjob.New(schedulerCfg, kwStore, jobStore, rdb) + scheduler := cronjob.New(schedulerCfg, kwStore, jobStore, adapterList, rdb) scheduler.OnComplete = func(kws []string, scraped, saved int, duration time.Duration) { printSummary(len(adapterList), kws, scraped, duration) @@ -119,13 +119,33 @@ func newRedisClient() (*redis.Client, error) { rdb := redis.NewClient(opts) - ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() - if err := rdb.Ping(ctx).Err(); err != nil { + if err := waitForRedisPing(ctx, rdb); err != nil { _ = rdb.Close() return nil, err } return rdb, nil } + +func waitForRedisPing(ctx context.Context, rdb *redis.Client) error { + var lastErr error + ticker := time.NewTicker(time.Second) + defer ticker.Stop() + + for { + if err := rdb.Ping(ctx).Err(); err == nil { + return nil + } else { + lastErr = err + } + + select { + case <-ctx.Done(): + return lastErr + case <-ticker.C: + } + } +} diff --git a/scraper-go/internal/adapters/adapter.go b/scraper-go/internal/adapters/adapter.go index 686d318..15e4c20 100644 --- a/scraper-go/internal/adapters/adapter.go +++ b/scraper-go/internal/adapters/adapter.go @@ -1,17 +1,6 @@ package adapters -import ( - "context" +import "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" -) - -// Adapter é a interface que todo scraper de fonte deve implementar. -type Adapter interface { - // SourceName retorna o identificador da fonte (ex: "linkedin", "Adzuna:br"). - SourceName() string - - // Search busca vagas para uma keyword e configuração específica. - // Deve respeitar o contexto (timeout / cancelamento). - Search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) -} +type Adapter = ports.JobSource +type BatchAdapter = ports.BatchJobSource diff --git a/scraper-go/internal/adapters/adapterutil/helpers.go b/scraper-go/internal/adapters/adapterutil/helpers.go new file mode 100644 index 0000000..3bbea82 --- /dev/null +++ b/scraper-go/internal/adapters/adapterutil/helpers.go @@ -0,0 +1,107 @@ +package adapterutil + +import ( + "html" + "strings" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func NonEmptyStrings(values []string) []string { + out := make([]string, 0, len(values)) + for _, value := range values { + value = strings.TrimSpace(value) + if value != "" { + out = append(out, value) + } + } + return out +} + +func ContainsNormalized(text, query string) bool { + return strings.Contains(NormalizeText(text), NormalizeText(query)) +} + +func MatchesKeyword(text, keyword string) bool { + normalizedText := " " + NormalizeText(text) + " " + terms := strings.Fields(NormalizeText(keyword)) + if len(terms) == 0 { + return true + } + + for _, term := range terms { + if term == "go" { + if strings.Contains(normalizedText, " go ") || strings.Contains(normalizedText, " golang ") { + continue + } + return false + } + if !strings.Contains(normalizedText, " "+term+" ") { + return false + } + } + + return true +} + +func NormalizeText(value string) string { + value = strings.ToLower(html.UnescapeString(value)) + value = strings.ReplaceAll(value, "/", " ") + value = strings.ReplaceAll(value, "-", " ") + return strings.Join(strings.Fields(value), " ") +} + +func UniqueTrimmedStrings(values []string) []string { + seen := make(map[string]struct{}, len(values)) + out := make([]string, 0, len(values)) + for _, value := range values { + value = strings.TrimSpace(value) + if value == "" { + continue + } + key := strings.ToLower(value) + if _, ok := seen[key]; ok { + continue + } + seen[key] = struct{}{} + out = append(out, value) + } + return out +} + +func RepeatedJobPage(seen map[string]struct{}, jobs []domain.Job) bool { + signature := JobPageSignature(jobs) + if signature == "" { + return false + } + if _, exists := seen[signature]; exists { + return true + } + seen[signature] = struct{}{} + return false +} + +func JobPageSignature(jobs []domain.Job) string { + var b strings.Builder + + for _, job := range jobs { + key := strings.TrimSpace(job.ID) + if key == "" { + key = strings.TrimSpace(job.URL) + } + if key == "" { + key = strings.Join([]string{ + strings.TrimSpace(job.Title), + strings.TrimSpace(job.Company), + strings.TrimSpace(job.Location), + }, "|") + } + if strings.Trim(key, "|") == "" { + continue + } + b.WriteString(key) + b.WriteByte('\n') + } + + return b.String() +} diff --git a/scraper-go/internal/adapters/adzuna.go b/scraper-go/internal/adapters/adzuna.go deleted file mode 100644 index 5a197d1..0000000 --- a/scraper-go/internal/adapters/adzuna.go +++ /dev/null @@ -1,197 +0,0 @@ -package adapters - -import ( - "context" - "encoding/json" - "fmt" - "net/http" - "net/url" - "strings" - "time" - - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" -) - -type AdzunaAdapter struct { - client *http.Client - appID string - appKey string - country string - // mu sync.Mutex - semaphore chan struct{} -} - -func NewAdzuna(appID, appKey, country string) *AdzunaAdapter { - return &AdzunaAdapter{ - client: &http.Client{Timeout: 60 * time.Second}, - appID: appID, - appKey: appKey, - country: strings.ToLower(strings.TrimSpace(country)), - semaphore: make(chan struct{}, 3), - } -} - -func (a *AdzunaAdapter) SourceName() string { - return fmt.Sprintf("Adzuna:%s", a.country) -} - -// buildURL espelha exatamente o buildAdzunaUrl do JS. -func (a *AdzunaAdapter) buildURL(keyword string, req models.ScrapeRequest, page int) string { - resultsPerPage := req.ResultsPerPage - if resultsPerPage <= 0 { - resultsPerPage = 20 - } - - endpoint := fmt.Sprintf( - "https://api.adzuna.com/v1/api/jobs/%s/search/%d", - a.country, page, - ) - - u, _ := url.Parse(endpoint) - q := u.Query() - q.Set("app_id", a.appID) - q.Set("app_key", a.appKey) - q.Set("results_per_page", fmt.Sprintf("%d", resultsPerPage)) - q.Set("what", keyword) - - if req.SearchLocation != "" { - q.Set("where", req.SearchLocation) - } - - // Comentado para espelhar o JS: - // if req.RemoteOnly { - // q.Set("work_from_home", "1") - // } - - u.RawQuery = q.Encode() - return u.String() -} - -func (a *AdzunaAdapter) Search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) { - // Adzuna é sensível a concorrência, então usamos um canal como semáforo para limitar. - a.semaphore <- struct{}{} - defer func() { <-a.semaphore }() - - maxPages := req.MaxPagesPerKeyword - if maxPages <= 0 { - maxPages = 3 - } - - // Intervalo entre requisições (aumentamos a segurança) - waitDuration := time.Duration(req.WaitBetweenSearchesMs) * time.Millisecond - if waitDuration <= 0 { - waitDuration = 2000 * time.Millisecond // Adzuna é sensível, 2s é mais seguro - } - - pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond - if pageTimeout <= 0 { - pageTimeout = 15 * time.Second - } - - var allJobs []models.Job - - for page := 1; page <= maxPages; page++ { - endpoint := a.buildURL(keyword, req, page) - - pageCtx, cancel := context.WithTimeout(ctx, pageTimeout) - jobs, err := a.fetchPage(pageCtx, endpoint, keyword) - cancel() - - if err != nil { - // Se der erro 400 ou 429, o log detalhado agora sairá no fetchPage - return nil, fmt.Errorf("adzuna erro na página %d: %w", page, err) - } - - if len(jobs) == 0 { - break - } - - allJobs = append(allJobs, jobs...) - - // Pausa obrigatória entre páginas. - // O semáforo limita QUANTAS keywords rodam juntas, - // e este sleep garante o respiro entre as PÁGINAS de cada keyword. - select { - case <-ctx.Done(): - return allJobs, ctx.Err() - case <-time.After(waitDuration): - // Continua para a próxima página ou libera para a próxima keyword - } - } - - return allJobs, nil -} - -type adzunaResponse struct { - Results []struct { - Title string `json:"title"` - Company struct { - DisplayName string `json:"display_name"` - } `json:"company"` - Location struct { - DisplayName string `json:"display_name"` - } `json:"location"` - RedirectURL string `json:"redirect_url"` - URL string `json:"url"` - SalaryMin float64 `json:"salary_min"` - SalaryMax float64 `json:"salary_max"` - Created string `json:"created"` - } `json:"results"` -} - -func (a *AdzunaAdapter) fetchPage(ctx context.Context, endpoint, keyword string) ([]models.Job, error) { - req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) - if err != nil { - return nil, err - } - req.Header.Set("Accept", "application/json") - - resp, err := a.client.Do(req) - if err != nil { - return nil, err - } - defer resp.Body.Close() - - if resp.StatusCode != http.StatusOK { - return nil, fmt.Errorf("status %d", resp.StatusCode) - } - - var data adzunaResponse - if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { - return nil, err - } - - if len(data.Results) == 0 { - return []models.Job{}, nil - } - - jobs := make([]models.Job, 0, len(data.Results)) - for _, r := range data.Results { - // Espelha o salario do JS: "min-max" ou vazio. - salario := "" - if r.SalaryMin != 0 || r.SalaryMax != 0 { - salario = fmt.Sprintf("%g-%g", r.SalaryMin, r.SalaryMax) - } - - link := strings.TrimSpace(r.RedirectURL) - if link == "" { - link = strings.TrimSpace(r.URL) - } - - jobs = append(jobs, models.Job{ - ID: link, - Title: strings.TrimSpace(r.Title), - Company: strings.TrimSpace(r.Company.DisplayName), - Location: strings.TrimSpace(r.Location.DisplayName), - URL: link, - Salary: salario, - PostedAt: r.Created, - Source: "Adzuna", - Sources: []string{"Adzuna"}, - Keyword: keyword, - Keywords: []string{keyword}, - }) - } - - return jobs, nil -} diff --git a/scraper-go/internal/adapters/adzuna/adapter.go b/scraper-go/internal/adapters/adzuna/adapter.go new file mode 100644 index 0000000..92918a1 --- /dev/null +++ b/scraper-go/internal/adapters/adzuna/adapter.go @@ -0,0 +1,372 @@ +package adzuna + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "log/slog" + "net/http" + "net/url" + "os" + "strconv" + "strings" + "sync" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +const ( + defaultAdzunaResultsPerPage = 20 + defaultAdzunaMaxPages = 5 + defaultAdzunaKeywordSlotSize = 30 +) + +type adzunaStatusError struct { + statusCode int +} + +func (e adzunaStatusError) Error() string { + return fmt.Sprintf("status %d", e.statusCode) +} + +type AdzunaAdapter struct { + client *http.Client + appID string + appKey string + country string + semaphore chan struct{} + mu sync.Mutex + nextOffset int +} + +func NewAdzuna(appID, appKey, country string) *AdzunaAdapter { + return &AdzunaAdapter{ + client: &http.Client{Timeout: 60 * time.Second}, + appID: appID, + appKey: appKey, + country: strings.ToLower(strings.TrimSpace(country)), + semaphore: make(chan struct{}, 3), + } +} + +func (a *AdzunaAdapter) SourceName() string { + return fmt.Sprintf("Adzuna:%s", a.country) +} + +func adzunaKeywordSlotSize() int { + value := strings.TrimSpace(os.Getenv("ADZUNA_KEYWORD_SLOT_SIZE")) + if value == "" { + return defaultAdzunaKeywordSlotSize + } + parsed, err := strconv.Atoi(value) + if err != nil || parsed <= 0 { + return defaultAdzunaKeywordSlotSize + } + return parsed +} + +// buildURL espelha exatamente o buildAdzunaUrl do JS. +func (a *AdzunaAdapter) buildURL(keyword string, req domain.ScrapeRequest, page int) string { + resultsPerPage := req.ResultsPerPage + if resultsPerPage <= 0 { + resultsPerPage = defaultAdzunaResultsPerPage + } + + endpoint := fmt.Sprintf( + "https://api.adzuna.com/v1/api/jobs/%s/search/%d", + a.country, page, + ) + + u, _ := url.Parse(endpoint) + q := u.Query() + q.Set("app_id", a.appID) + q.Set("app_key", a.appKey) + q.Set("results_per_page", fmt.Sprintf("%d", resultsPerPage)) + q.Set("what", keyword) + + if req.SearchLocation != "" { + q.Set("where", req.SearchLocation) + } + + // Comentado para espelhar o JS: + // if req.RemoteOnly { + // q.Set("work_from_home", "1") + // } + + u.RawQuery = q.Encode() + return u.String() +} + +func (a *AdzunaAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + a.semaphore <- struct{}{} + defer func() { <-a.semaphore }() + + return a.searchKeyword(ctx, keyword, req) +} + +func (a *AdzunaAdapter) SearchBatch(ctx context.Context, keywords []string, req domain.ScrapeRequest) ([]domain.Job, error) { + slot := a.nextKeywordSlot(keywords, adzunaKeywordSlotSize()) + if len(slot) == 0 { + return nil, nil + } + + if len(slot) < len(keywords) { + slog.Info("adzuna: usando slot rotativo de keywords", + "country", a.country, + "selected", len(slot), + "total", len(keywords), + ) + } + + type searchResult struct { + jobs []domain.Job + err error + } + + results := make(chan searchResult, len(slot)) + var wg sync.WaitGroup + + for _, keyword := range slot { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + + wg.Add(1) + go func(keyword string) { + defer wg.Done() + + select { + case a.semaphore <- struct{}{}: + defer func() { <-a.semaphore }() + case <-ctx.Done(): + results <- searchResult{err: ctx.Err()} + return + } + + jobs, err := a.searchKeyword(ctx, keyword, req) + results <- searchResult{jobs: jobs, err: err} + }(keyword) + } + + wg.Wait() + close(results) + + var allJobs []domain.Job + var firstErr error + for result := range results { + if result.err != nil { + if firstErr == nil { + firstErr = result.err + } + continue + } + allJobs = append(allJobs, result.jobs...) + } + if len(allJobs) > 0 { + return allJobs, nil + } + + return nil, firstErr +} + +func (a *AdzunaAdapter) nextKeywordSlot(keywords []string, slotSize int) []string { + if len(keywords) == 0 { + return nil + } + if slotSize <= 0 || slotSize >= len(keywords) { + return append([]string(nil), keywords...) + } + + a.mu.Lock() + defer a.mu.Unlock() + + offset := a.nextOffset % len(keywords) + a.nextOffset = (offset + slotSize) % len(keywords) + + slot := make([]string, 0, slotSize) + for i := 0; i < slotSize; i++ { + slot = append(slot, keywords[(offset+i)%len(keywords)]) + } + return slot +} + +func (a *AdzunaAdapter) searchKeyword(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + return nil, nil + } + + maxPages := req.MaxPagesPerKeyword + if maxPages <= 0 { + maxPages = defaultAdzunaMaxPages + } + + // Intervalo entre requisições (aumentamos a segurança) + waitDuration := time.Duration(req.WaitBetweenSearchesMs) * time.Millisecond + if waitDuration <= 0 { + waitDuration = 2000 * time.Millisecond // Adzuna é sensível, 2s é mais seguro + } + + pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond + if pageTimeout <= 0 { + pageTimeout = 15 * time.Second + } + + var allJobs []domain.Job + seenPages := make(map[string]struct{}) + + for page := 1; page <= maxPages; page++ { + endpoint := a.buildURL(keyword, req, page) + + pageCtx, cancel := context.WithTimeout(ctx, pageTimeout) + jobs, err := a.fetchPageWithRetry(pageCtx, endpoint, keyword) + cancel() + + if err != nil { + if len(allJobs) > 0 && shouldKeepAdzunaPartialResults(err) { + return allJobs, nil + } + return nil, fmt.Errorf("adzuna erro na página %d: %w", page, err) + } + + if len(jobs) == 0 { + break + } + if adapterutil.RepeatedJobPage(seenPages, jobs) { + break + } + + allJobs = append(allJobs, jobs...) + + // Pausa obrigatória entre páginas. + // O semáforo limita QUANTAS keywords rodam juntas, + // e este sleep garante o respiro entre as PÁGINAS de cada keyword. + select { + case <-ctx.Done(): + return allJobs, ctx.Err() + case <-time.After(waitDuration): + // Continua para a próxima página ou libera para a próxima keyword + } + } + + return allJobs, nil +} + +func (a *AdzunaAdapter) fetchPageWithRetry(ctx context.Context, endpoint, keyword string) ([]domain.Job, error) { + var lastErr error + + for attempt := 0; attempt < 3; attempt++ { + jobs, err := a.fetchPage(ctx, endpoint, keyword) + if err == nil { + return jobs, nil + } + + lastErr = err + if !isTransientAdzunaError(err) { + return nil, err + } + + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-time.After(time.Duration(attempt+1) * 1500 * time.Millisecond): + } + } + + return nil, lastErr +} + +func shouldKeepAdzunaPartialResults(err error) bool { + return isTransientAdzunaError(err) || errors.Is(err, context.DeadlineExceeded) +} + +func isTransientAdzunaError(err error) bool { + var statusErr adzunaStatusError + if errors.As(err, &statusErr) { + return statusErr.statusCode == http.StatusTooManyRequests || + statusErr.statusCode == http.StatusInternalServerError || + statusErr.statusCode == http.StatusBadGateway || + statusErr.statusCode == http.StatusServiceUnavailable || + statusErr.statusCode == http.StatusGatewayTimeout + } + + return errors.Is(err, context.DeadlineExceeded) +} + +type adzunaResponse struct { + Results []struct { + Title string `json:"title"` + Company struct { + DisplayName string `json:"display_name"` + } `json:"company"` + Location struct { + DisplayName string `json:"display_name"` + } `json:"location"` + RedirectURL string `json:"redirect_url"` + URL string `json:"url"` + SalaryMin float64 `json:"salary_min"` + SalaryMax float64 `json:"salary_max"` + Created string `json:"created"` + } `json:"results"` +} + +func (a *AdzunaAdapter) fetchPage(ctx context.Context, endpoint, keyword string) ([]domain.Job, error) { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, err + } + req.Header.Set("Accept", "application/json") + + resp, err := a.client.Do(req) + if err != nil { + return nil, err + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return nil, adzunaStatusError{statusCode: resp.StatusCode} + } + + var data adzunaResponse + if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { + return nil, err + } + + if len(data.Results) == 0 { + return []domain.Job{}, nil + } + + jobs := make([]domain.Job, 0, len(data.Results)) + for _, r := range data.Results { + // Espelha o salario do JS: "min-max" ou vazio. + salario := "" + if r.SalaryMin != 0 || r.SalaryMax != 0 { + salario = fmt.Sprintf("%g-%g", r.SalaryMin, r.SalaryMax) + } + + link := strings.TrimSpace(r.RedirectURL) + if link == "" { + link = strings.TrimSpace(r.URL) + } + + jobs = append(jobs, domain.Job{ + ID: link, + Title: strings.TrimSpace(r.Title), + Company: strings.TrimSpace(r.Company.DisplayName), + Location: strings.TrimSpace(r.Location.DisplayName), + URL: link, + Salary: salario, + PostedAt: r.Created, + Source: "Adzuna", + Sources: []string{"Adzuna"}, + Keyword: keyword, + Keywords: []string{keyword}, + }) + } + + return jobs, nil +} diff --git a/scraper-go/internal/adapters/adzuna/pagination_test.go b/scraper-go/internal/adapters/adzuna/pagination_test.go new file mode 100644 index 0000000..f17f24a --- /dev/null +++ b/scraper-go/internal/adapters/adzuna/pagination_test.go @@ -0,0 +1,140 @@ +package adzuna + +import ( + "context" + "fmt" + "net/http" + "strconv" + "strings" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestAdzunaSearchContinuesPastFormerDefaultPagesUntilEmpty(t *testing.T) { + calls := 0 + adapter := NewAdzuna("app", "key", "br") + adapter.client = testutil.HTTPClient(func(req *http.Request) (*http.Response, error) { + calls++ + parts := strings.Split(strings.Trim(req.URL.Path, "/"), "/") + page, err := strconv.Atoi(parts[len(parts)-1]) + if err != nil { + t.Fatalf("invalid page in path %q: %v", req.URL.Path, err) + } + if page > 4 { + return testutil.Response(`{"results":[]}`), nil + } + return testutil.Response(fmt.Sprintf(`{ + "results":[{ + "title":"Dev Go %d", + "company":{"display_name":"Acme"}, + "location":{"display_name":"Brasil"}, + "redirect_url":"https://example.com/adzuna/%d", + "created":"2026-07-27T00:00:00Z" + }] + }`, page, page)), nil + }) + + jobs, err := adapter.Search(context.Background(), "go", domain.ScrapeRequest{ + WaitBetweenSearchesMs: 1, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 4 { + t.Fatalf("expected 4 jobs from pages past old limit, got %d", len(jobs)) + } + if calls != 5 { + t.Fatalf("expected 5 calls including empty page, got %d", calls) + } +} + +func TestAdzunaSearchStopsOnRepeatedPage(t *testing.T) { + calls := 0 + adapter := NewAdzuna("app", "key", "br") + adapter.client = testutil.HTTPClient(func(req *http.Request) (*http.Response, error) { + calls++ + return testutil.Response(`{ + "results":[{ + "title":"Dev Go", + "company":{"display_name":"Acme"}, + "location":{"display_name":"Brasil"}, + "redirect_url":"https://example.com/adzuna/repeated", + "created":"2026-07-27T00:00:00Z" + }] + }`), nil + }) + + jobs, err := adapter.Search(context.Background(), "go", domain.ScrapeRequest{ + WaitBetweenSearchesMs: 1, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected only first repeated page to be kept, got %d jobs", len(jobs)) + } + if calls != 2 { + t.Fatalf("expected stop after detecting repeated second page, got %d calls", calls) + } +} + +func TestAdzunaSearchKeepsPartialResultsOnTransientFailure(t *testing.T) { + calls := 0 + adapter := NewAdzuna("app", "key", "br") + adapter.client = testutil.HTTPClient(func(req *http.Request) (*http.Response, error) { + calls++ + parts := strings.Split(strings.Trim(req.URL.Path, "/"), "/") + page, err := strconv.Atoi(parts[len(parts)-1]) + if err != nil { + t.Fatalf("invalid page in path %q: %v", req.URL.Path, err) + } + if page == 2 { + return testutil.StatusResponse(http.StatusServiceUnavailable, `{"error":"busy"}`), nil + } + return testutil.Response(`{ + "results":[{ + "title":"Dev Java", + "company":{"display_name":"Acme"}, + "location":{"display_name":"Brasil"}, + "redirect_url":"https://example.com/adzuna/java", + "created":"2026-07-27T00:00:00Z" + }] + }`), nil + }) + + jobs, err := adapter.Search(context.Background(), "java", domain.ScrapeRequest{ + WaitBetweenSearchesMs: 1, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected partial first page to be kept, got %d jobs", len(jobs)) + } + if calls != 4 { + t.Fatalf("expected first page plus 3 retry attempts on page 2, got %d calls", calls) + } +} + +func TestAdzunaSearchReturnsErrorWhenFirstPageFails(t *testing.T) { + adapter := NewAdzuna("app", "key", "br") + adapter.client = testutil.HTTPClient(func(req *http.Request) (*http.Response, error) { + return testutil.StatusResponse(http.StatusServiceUnavailable, `{"error":"busy"}`), nil + }) + + jobs, err := adapter.Search(context.Background(), "java", domain.ScrapeRequest{ + WaitBetweenSearchesMs: 1, + }) + + if err == nil { + t.Fatal("expected error when first page fails") + } + if len(jobs) != 0 { + t.Fatalf("expected no jobs when first page fails, got %d", len(jobs)) + } +} diff --git a/scraper-go/internal/adapters/greehouse.go b/scraper-go/internal/adapters/greehouse.go deleted file mode 100644 index ba43103..0000000 --- a/scraper-go/internal/adapters/greehouse.go +++ /dev/null @@ -1,182 +0,0 @@ -package adapters - -import ( - "context" - "encoding/json" - "fmt" - "log/slog" - "net/http" - "os" - "regexp" - "strings" - "time" - - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" -) - -var jsonLDPattern = regexp.MustCompile(`(?s)]+type="application/ld\+json"[^>]*>(.*?)`) - -type greenhouseListResponse struct { - Jobs []greenhouseListJob `json:"jobs"` -} - -type greenhouseListJob struct { - Title string `json:"title"` - Content string `json:"content"` - AbsoluteURL string `json:"absolute_url"` - UpdatedAt string `json:"updated_at"` - Location struct { - Name string `json:"name"` - } `json:"location"` -} - -type jobPosting struct { - Type string `json:"@type"` - Title string `json:"title"` - Description string `json:"description"` - DatePosted string `json:"datePosted"` - - HiringOrganization struct { - Name string `json:"name"` - } `json:"hiringOrganization"` - - JobLocation struct { - Address struct { - Locality string `json:"addressLocality"` - Region string `json:"addressRegion"` - Country string `json:"addressCountry"` - } `json:"address"` - } `json:"jobLocation"` - - BaseSalary *struct { - Value struct { - MinValue float64 `json:"minValue"` - MaxValue float64 `json:"maxValue"` - UnitText string `json:"unitText"` - } `json:"value"` - } `json:"baseSalary"` -} - -func extractJSONLD(html string) *jobPosting { - matches := jsonLDPattern.FindAllStringSubmatch(html, -1) - - for _, match := range matches { - if len(match) < 2 { - continue - } - - var posting jobPosting - if err := json.Unmarshal([]byte(strings.TrimSpace(match[1])), &posting); err != nil { - continue - } - - if posting.Type == "JobPosting" && posting.Title != "" { - return &posting - } - } - - return nil -} - -type GreenhouseAdapter struct { - client *http.Client - boardToken string - companyName string -} - -func NewGreenhouse(boardToken, companyName string) *GreenhouseAdapter { - return &GreenhouseAdapter{ - client: &http.Client{Timeout: 30 * time.Second}, - boardToken: boardToken, - companyName: companyName, - } -} - -func (a *GreenhouseAdapter) SourceName() string { - return fmt.Sprintf("Green House:%s", a.companyName) -} - -func (a *GreenhouseAdapter) Search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) { - pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond - if pageTimeout <= 0 { - pageTimeout = 15 * time.Second - } - - listCtx, cancel := context.WithTimeout(ctx, pageTimeout) - defer cancel() - - endpoint := fmt.Sprintf("https://boards-api.greenhouse.io/v1/boards/%s/jobs?content=true", a.boardToken) - - listReq, err := http.NewRequestWithContext(listCtx, http.MethodGet, endpoint, nil) - if err != nil { - return nil, err - } - - resp, err := a.client.Do(listReq) - if err != nil { - return nil, err - } - defer resp.Body.Close() - - if resp.StatusCode != http.StatusOK { - return nil, fmt.Errorf("status %d", resp.StatusCode) - } - - var data struct { - Jobs []struct { - Title string `json:"title"` - Content string `json:"content"` - AbsoluteURL string `json:"absolute_url"` - UpdatedAt string `json:"updated_at"` - Location struct { - Name string `json:"name"` - } `json:"location"` - } `json:"jobs"` - } - - if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { - return nil, err - } - - kwLower := strings.ToLower(keyword) - var jobs []models.Job - - for _, j := range data.Jobs { - if strings.Contains(strings.ToLower(j.Title), kwLower) { - jobs = append(jobs, models.Job{ - ID: strings.TrimSpace(j.AbsoluteURL), - Title: strings.TrimSpace(j.Title), - Description: strings.TrimSpace(j.Content), - Company: a.companyName, - Location: strings.TrimSpace(j.Location.Name), - URL: strings.TrimSpace(j.AbsoluteURL), - PostedAt: j.UpdatedAt, - Source: "Green House", - Sources: []string{"Green House"}, - Keyword: keyword, - Keywords: []string{keyword}, - }) - } - } - - return jobs, nil -} - -func FetchGreenhouseSlugs(ctx context.Context) ([]string, error) { - - filename := "./internal/interfaces/greenhouseCompanies.json" - - data, err := os.ReadFile(filename) - if err != nil { - return nil, fmt.Errorf("não foi possível ler o arquivo %s: %w", filename, err) - } - - var slugs []string - if err := json.Unmarshal(data, &slugs); err != nil { - return nil, fmt.Errorf("erro ao processar o JSON de empresas: %w", err) - } - - slog.Info("Slugs da Greenhouse carregados com sucesso", "total", len(slugs)) - - return slugs, nil -} diff --git a/scraper-go/internal/adapters/greenhouse/adapter.go b/scraper-go/internal/adapters/greenhouse/adapter.go new file mode 100644 index 0000000..8a2ae51 --- /dev/null +++ b/scraper-go/internal/adapters/greenhouse/adapter.go @@ -0,0 +1,455 @@ +package greenhouse + +import ( + "context" + "encoding/json" + "fmt" + "html" + "log/slog" + "net/http" + "os" + "regexp" + "strings" + "sync" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" +) + +// var jsonLDPattern = regexp.MustCompile(`(?s)]+type="application/ld\+json"[^>]*>(.*?)`) + +type greenhouseListResponse struct { + Jobs []greenhouseListJob `json:"jobs"` + Meta struct { + Total int `json:"total"` + } `json:"meta"` +} + +type greenhouseListJob struct { + ID int `json:"id"` + InternalJobID *int `json:"internal_job_id"` + Title string `json:"title"` + Content string `json:"content"` + AbsoluteURL string `json:"absolute_url"` + UpdatedAt string `json:"updated_at"` + Language string `json:"language"` + RequisitionID string `json:"requisition_id"` + Metadata []greenhouseMetadata `json:"metadata"` + Departments []greenhouseGroup `json:"departments"` + Offices []greenhouseOffice `json:"offices"` + Location struct { + Name string `json:"name"` + } `json:"location"` +} + +type greenhouseMetadata struct { + Name string `json:"name"` + Value any `json:"value"` +} + +type greenhouseGroup struct { + Name string `json:"name"` +} + +type greenhouseOffice struct { + Name string `json:"name"` + Location string `json:"location"` +} + +// type jobPosting struct { +// Type string `json:"@type"` +// Title string `json:"title"` +// Description string `json:"description"` +// DatePosted string `json:"datePosted"` + +// HiringOrganization struct { +// Name string `json:"name"` +// } `json:"hiringOrganization"` + +// JobLocation struct { +// Address struct { +// Locality string `json:"addressLocality"` +// Region string `json:"addressRegion"` +// Country string `json:"addressCountry"` +// } `json:"address"` +// } `json:"jobLocation"` + +// BaseSalary *struct { +// Value struct { +// MinValue float64 `json:"minValue"` +// MaxValue float64 `json:"maxValue"` +// UnitText string `json:"unitText"` +// } `json:"value"` +// } `json:"baseSalary"` +// } + +// func extractJSONLD(html string) *jobPosting { +// matches := jsonLDPattern.FindAllStringSubmatch(html, -1) + +// for _, match := range matches { +// if len(match) < 2 { +// continue +// } + +// var posting jobPosting +// if err := json.Unmarshal([]byte(strings.TrimSpace(match[1])), &posting); err != nil { +// continue +// } + +// if posting.Type == "JobPosting" && posting.Title != "" { +// return &posting +// } +// } + +// return nil +// } + +type GreenhouseAdapter struct { + client *http.Client + boardToken string + companyName string + + mu sync.Mutex + cachedJobs []greenhouseListJob + cacheErr error + loaded bool +} + +func NewGreenhouse(boardToken, companyName string) *GreenhouseAdapter { + return &GreenhouseAdapter{ + client: &http.Client{Timeout: 30 * time.Second}, + boardToken: boardToken, + companyName: companyName, + } +} + +func (a *GreenhouseAdapter) SourceName() string { + return fmt.Sprintf("Green House:%s", a.companyName) +} + +func (a *GreenhouseAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + return a.SearchBatch(ctx, []string{keyword}, req) +} + +func (a *GreenhouseAdapter) SearchBatch(ctx context.Context, keywords []string, req domain.ScrapeRequest) ([]domain.Job, error) { + rawJobs, err := a.fetchJobs(ctx, req) + if err != nil { + return nil, err + } + + var jobs []domain.Job + + for _, keyword := range keywords { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + + for _, j := range rawJobs { + if !a.matchesRequest(j, keyword, req) { + continue + } + + source := "Green House" + jobs = append(jobs, domain.Job{ + ID: greenhouseJobID(j), + Title: strings.TrimSpace(j.Title), + Description: greenhouseDescription(j), + Company: a.companyName, + Location: greenhouseLocation(j), + URL: strings.TrimSpace(j.AbsoluteURL), + Modality: greenhouseModality(j), + PostedAt: strings.TrimSpace(j.UpdatedAt), + Source: source, + Sources: []string{source}, + Keyword: keyword, + Keywords: []string{keyword}, + }) + } + } + + return jobs, nil +} + +func (a *GreenhouseAdapter) fetchJobs(ctx context.Context, req domain.ScrapeRequest) ([]greenhouseListJob, error) { + a.mu.Lock() + if a.loaded { + defer a.mu.Unlock() + return a.cachedJobs, a.cacheErr + } + defer a.mu.Unlock() + + pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond + if pageTimeout <= 0 { + pageTimeout = 15 * time.Second + } + + listCtx, cancel := context.WithTimeout(ctx, pageTimeout) + defer cancel() + + endpoint := fmt.Sprintf("https://boards-api.greenhouse.io/v1/boards/%s/jobs?content=true", a.boardToken) + + listReq, err := http.NewRequestWithContext(listCtx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, err + } + + listReq.Header.Set("User-Agent", "JobsScraper/1.0") + listReq.Header.Set("Accept", "application/json") + + resp, err := a.client.Do(listReq) + if err != nil { + a.cachedJobs = nil + a.cacheErr = err + a.loaded = true + return nil, err + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + err := fmt.Errorf("greenhouse: status inesperado %d para board '%s'", resp.StatusCode, a.boardToken) + a.cachedJobs = nil + a.cacheErr = err + a.loaded = true + return nil, err + } + + var data greenhouseListResponse + if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { + a.cachedJobs = nil + a.cacheErr = err + a.loaded = true + return nil, err + } + + a.cachedJobs = data.Jobs + a.cacheErr = nil + a.loaded = true + + return data.Jobs, nil +} + +func (a *GreenhouseAdapter) matchesRequest(job greenhouseListJob, keyword string, req domain.ScrapeRequest) bool { + if lang := strings.TrimSpace(req.SearchLanguage); lang != "" { + if job.Language != "" && !strings.EqualFold(job.Language, lang) { + return false + } + } + + if keyword != "" && !matchesGreenhouseKeyword(greenhouseSearchText(job), keyword) { + return false + } + + if req.RemoteOnly && !greenhouseIsRemote(job) { + return false + } + + if location := strings.TrimSpace(req.SearchLocation); location != "" && !greenhouseIsRemote(job) { + if !containsNormalized(greenhouseLocation(job), location) && !containsNormalized(greenhouseSearchText(job), location) { + return false + } + } + + return true +} + +func greenhouseSearchText(job greenhouseListJob) string { + parts := []string{ + job.Title, + job.Content, + job.Location.Name, + job.Language, + job.RequisitionID, + } + + for _, department := range job.Departments { + parts = append(parts, department.Name) + } + for _, office := range job.Offices { + parts = append(parts, office.Name, office.Location) + } + for _, metadata := range job.Metadata { + parts = append(parts, metadata.Name, fmt.Sprint(metadata.Value)) + } + + return strings.Join(parts, " ") +} + +func greenhouseJobID(job greenhouseListJob) string { + if job.AbsoluteURL != "" { + return strings.TrimSpace(job.AbsoluteURL) + } + if job.ID != 0 { + return fmt.Sprintf("greenhouse:%d", job.ID) + } + return strings.TrimSpace(job.Title) +} + +func greenhouseDescription(job greenhouseListJob) string { + sections := []string{cleanGreenhouseHTML(job.Content)} + + departments := greenhouseDepartmentNames(job) + if len(departments) > 0 { + sections = append(sections, "Departamentos: "+strings.Join(departments, ", ")) + } + + offices := greenhouseOfficeNames(job) + if len(offices) > 0 { + sections = append(sections, "Escritórios: "+strings.Join(offices, ", ")) + } + + metadata := greenhouseMetadataPairs(job) + if len(metadata) > 0 { + sections = append(sections, "Metadados: "+strings.Join(metadata, "; ")) + } + + return strings.Join(adapterutil.NonEmptyStrings(sections), "\n\n") +} + +func greenhouseDepartmentNames(job greenhouseListJob) []string { + names := make([]string, 0, len(job.Departments)) + for _, department := range job.Departments { + names = append(names, department.Name) + } + return adapterutil.UniqueTrimmedStrings(names) +} + +func greenhouseOfficeNames(job greenhouseListJob) []string { + names := make([]string, 0, len(job.Offices)) + for _, office := range job.Offices { + names = append(names, strings.Join(adapterutil.NonEmptyStrings([]string{office.Name, office.Location}), " - ")) + } + return adapterutil.UniqueTrimmedStrings(names) +} + +func greenhouseMetadataPairs(job greenhouseListJob) []string { + pairs := make([]string, 0, len(job.Metadata)) + for _, metadata := range job.Metadata { + name := strings.TrimSpace(metadata.Name) + value := strings.TrimSpace(fmt.Sprint(metadata.Value)) + if name == "" || value == "" || value == "" { + continue + } + pairs = append(pairs, name+": "+value) + } + return adapterutil.UniqueTrimmedStrings(pairs) +} + +func greenhouseLocation(job greenhouseListJob) string { + locations := []string{job.Location.Name} + for _, office := range job.Offices { + locations = append(locations, office.Location, office.Name) + } + return strings.Join(adapterutil.UniqueTrimmedStrings(locations), " | ") +} + +func greenhouseModality(job greenhouseListJob) string { + text := normalizeGreenhouseText(greenhouseSearchText(job)) + switch { + case strings.Contains(text, "hybrid") || strings.Contains(text, "hibrido"): + return "Híbrido" + case greenhouseIsRemote(job): + return "Remoto" + case strings.Contains(text, "onsite") || strings.Contains(text, "on site") || strings.Contains(text, "presencial"): + return "Presencial" + default: + return "" + } +} + +func greenhouseIsRemote(job greenhouseListJob) bool { + text := normalizeGreenhouseText(strings.Join([]string{ + job.Location.Name, + greenhouseSearchText(job), + }, " ")) + + return strings.Contains(text, "remote") || + strings.Contains(text, "remoto") || + strings.Contains(text, "anywhere") || + strings.Contains(text, "worldwide") || + strings.Contains(text, "global") +} + +func cleanGreenhouseHTML(value string) string { + value = html.UnescapeString(value) + value = regexp.MustCompile(`(?is)]*>.*?`).ReplaceAllString(value, " ") + value = regexp.MustCompile(`(?is)]*>.*?`).ReplaceAllString(value, " ") + value = regexp.MustCompile(`(?s)<[^>]+>`).ReplaceAllString(value, " ") + value = html.UnescapeString(value) + return strings.Join(strings.Fields(value), " ") +} + +func containsNormalized(text, query string) bool { + return strings.Contains(normalizeGreenhouseText(text), normalizeGreenhouseText(query)) +} + +func matchesGreenhouseKeyword(text, keyword string) bool { + normalizedText := " " + normalizeGreenhouseText(text) + " " + terms := strings.Fields(normalizeGreenhouseText(keyword)) + if len(terms) == 0 { + return true + } + + for _, term := range terms { + if term == "go" { + if strings.Contains(normalizedText, " go ") || strings.Contains(normalizedText, " golang ") { + continue + } + return false + } + if !strings.Contains(normalizedText, " "+term+" ") { + return false + } + } + + return true +} + +func normalizeGreenhouseText(value string) string { + value = strings.ToLower(html.UnescapeString(value)) + value = strings.ReplaceAll(value, "/", " ") + value = strings.ReplaceAll(value, "-", " ") + return strings.Join(strings.Fields(value), " ") +} + +func FetchGreenhouseSlugs(ctx context.Context) ([]string, error) { + + filename := os.Getenv("GREENHOUSE_COMPANIES_FILE") + if filename == "" { + filename = "./internal/interfaces/greenhouseCompanies.json" + } + + data, err := os.ReadFile(filename) + if err != nil { + return nil, fmt.Errorf("não foi possível ler o arquivo %s: %w", filename, err) + } + + var slugs []string + if err := json.Unmarshal(data, &slugs); err != nil { + return nil, fmt.Errorf("erro ao processar o JSON de empresas: %w", err) + } + + slog.Info("Slugs da Greenhouse carregados com sucesso", "total", len(slugs)) + + return slugs, nil +} + +func BuildGreenhouseAdapters(ctx context.Context) ([]ports.JobSource, error) { + slugs, err := FetchGreenhouseSlugs(ctx) + if err != nil { + return nil, err + } + + result := make([]ports.JobSource, 0, len(slugs)) + for _, slug := range slugs { + slug = strings.TrimSpace(slug) + if slug == "" { + continue + } + result = append(result, NewGreenhouse(slug, slug)) + } + + return result, nil +} diff --git a/scraper-go/internal/adapters/greenhouse/adapter_test.go b/scraper-go/internal/adapters/greenhouse/adapter_test.go new file mode 100644 index 0000000..47582ca --- /dev/null +++ b/scraper-go/internal/adapters/greenhouse/adapter_test.go @@ -0,0 +1,166 @@ +package greenhouse + +import ( + "context" + "fmt" + "net/http" + "strings" + "sync" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestGreenhouseSearchMatchesRichJobFields(t *testing.T) { + adapter := NewGreenhouse("acme", "Acme") + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.String() != "https://boards-api.greenhouse.io/v1/boards/acme/jobs?content=true" { + t.Fatalf("unexpected endpoint: %s", req.URL.String()) + } + return testResponse(`{ + "jobs": [ + { + "id": 123, + "title": "Backend Engineer", + "content": "

Build payment services with Java and PostgreSQL.

", + "absolute_url": "https://boards.greenhouse.io/acme/jobs/123", + "updated_at": "2026-07-27T00:00:00Z", + "language": "en", + "requisition_id": "ENG-123", + "location": {"name": "Remote - Brazil"}, + "departments": [{"name": "Platform Engineering"}], + "offices": [{"name": "LATAM", "location": "Brazil"}], + "metadata": [{"name": "Stack", "value": "Kafka"}] + }, + { + "id": 456, + "title": "Finance Analyst", + "content": "

Budget planning.

", + "absolute_url": "https://boards.greenhouse.io/acme/jobs/456", + "updated_at": "2026-07-27T00:00:00Z", + "language": "en", + "location": {"name": "New York"} + } + ], + "meta": {"total": 2} + }`), nil + }) + + jobs, err := adapter.Search(context.Background(), "kafka", domain.ScrapeRequest{ + SearchLanguage: "en", + SearchLocation: "Brazil", + RemoteOnly: true, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected one rich metadata match, got %d", len(jobs)) + } + + job := jobs[0] + if job.Title != "Backend Engineer" { + t.Fatalf("unexpected title: %q", job.Title) + } + if job.Location != "Remote - Brazil | Brazil | LATAM" { + t.Fatalf("unexpected location: %q", job.Location) + } + if job.Modality != "Remoto" { + t.Fatalf("expected remote modality, got %q", job.Modality) + } + if !strings.Contains(job.Description, "Build payment services") { + t.Fatalf("expected clean content in description: %q", job.Description) + } + if !strings.Contains(job.Description, "Departamentos: Platform Engineering") { + t.Fatalf("expected departments in description: %q", job.Description) + } + if !strings.Contains(job.Description, "Metadados: Stack: Kafka") { + t.Fatalf("expected metadata in description: %q", job.Description) + } +} + +func TestGreenhouseSearchMatchesDescriptionNotOnlyTitle(t *testing.T) { + adapter := NewGreenhouse("acme", "Acme") + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + return testResponse(`{ + "jobs": [{ + "id": 123, + "title": "Software Engineer", + "content": "Experience with Golang services.", + "absolute_url": "https://boards.greenhouse.io/acme/jobs/123", + "location": {"name": "Remote"} + }] + }`), nil + }) + + jobs, err := adapter.Search(context.Background(), "go", domain.ScrapeRequest{}) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected description keyword match, got %d", len(jobs)) + } +} + +func TestGreenhouseSearchCachesBoardListAcrossKeywords(t *testing.T) { + var ( + mu sync.Mutex + calls int + ) + + adapter := NewGreenhouse("acme", "Acme") + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + mu.Lock() + calls++ + mu.Unlock() + + return testResponse(`{ + "jobs": [{ + "id": 123, + "title": "Go Engineer", + "content": "Backend APIs", + "absolute_url": "https://boards.greenhouse.io/acme/jobs/123", + "location": {"name": "Remote"} + }] + }`), nil + }) + + var wg sync.WaitGroup + for _, keyword := range []string{"go", "backend", "engineer"} { + wg.Add(1) + go func(keyword string) { + defer wg.Done() + if _, err := adapter.Search(context.Background(), keyword, domain.ScrapeRequest{}); err != nil { + t.Errorf("Search(%q) returned error: %v", keyword, err) + } + }(keyword) + } + wg.Wait() + + mu.Lock() + defer mu.Unlock() + if calls != 1 { + t.Fatalf("expected one Greenhouse list call, got %d", calls) + } +} + +func TestGreenhouseSearchHandlesNonOKStatus(t *testing.T) { + adapter := NewGreenhouse("missing", "Missing") + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusNotFound, + Body: http.NoBody, + Header: make(http.Header), + }, nil + }) + + _, err := adapter.Search(context.Background(), "go", domain.ScrapeRequest{}) + if err == nil { + t.Fatal("expected error for non-OK response") + } + if !strings.Contains(fmt.Sprint(err), "board 'missing'") { + t.Fatalf("expected board token in error, got %v", err) + } +} diff --git a/scraper-go/internal/adapters/greenhouse/test_helpers_test.go b/scraper-go/internal/adapters/greenhouse/test_helpers_test.go new file mode 100644 index 0000000..97a8d12 --- /dev/null +++ b/scraper-go/internal/adapters/greenhouse/test_helpers_test.go @@ -0,0 +1,15 @@ +package greenhouse + +import ( + "net/http" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" +) + +func testHTTPClient(fn testutil.RoundTripFunc) *http.Client { + return testutil.HTTPClient(fn) +} + +func testResponse(body string) *http.Response { + return testutil.Response(body) +} diff --git a/scraper-go/internal/adapters/gupy/adapter.go b/scraper-go/internal/adapters/gupy/adapter.go new file mode 100644 index 0000000..032137b --- /dev/null +++ b/scraper-go/internal/adapters/gupy/adapter.go @@ -0,0 +1,870 @@ +package gupy + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "log/slog" + "net" + "net/http" + "net/url" + "os" + "sort" + "strconv" + "strings" + "sync" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +const ( + gupyDefaultBaseURL = "https://employability-portal.gupy.io/api/v1/jobs" + gupyDefaultPageLimit = 100 + gupyDefaultMaxOffset = 10000 + gupyDefaultBatchSize = 4 + gupyDefaultQueryLimit = 60 +) + +type GupyAdapter struct { + client *http.Client + baseURL string + pageLimit int + maxOffset int + batchSize int + mu sync.Mutex + nextQueryOffset int +} + +type gupyResponse struct { + Data []gupyJob `json:"data"` +} + +type gupyJob struct { + ID any `json:"id"` + Name string `json:"name"` + CareerPageName string `json:"careerPageName"` + CareerPageURL string `json:"careerPageUrl"` + JobURL string `json:"jobUrl"` + City string `json:"city"` + State string `json:"state"` + Country string `json:"country"` + WorkplaceType string `json:"workplaceType"` + IsRemoteWork bool `json:"isRemoteWork"` + PublishedDate string `json:"publishedDate"` + Description string `json:"description"` +} + +type gupyHTTPError struct { + statusCode int + keyword string + offset int +} + +func (e *gupyHTTPError) Error() string { + return fmt.Sprintf("gupy: status inesperado %d para keyword %q offset %d", e.statusCode, e.keyword, e.offset) +} + +func (e *gupyHTTPError) StatusCode() int { + return e.statusCode +} + +func NewGupy() *GupyAdapter { + return &GupyAdapter{ + client: &http.Client{ + Timeout: 30 * time.Second, + Transport: &http.Transport{ + MaxIdleConnsPerHost: 10, + IdleConnTimeout: 90 * time.Second, + }, + }, + baseURL: gupyDefaultBaseURL, + pageLimit: gupyDefaultPageLimit, + maxOffset: gupyDefaultMaxOffset, + batchSize: gupyDefaultBatchSize, + } +} + +func (a *GupyAdapter) SourceName() string { + return "Gupy" +} + +func (a *GupyAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + return nil, nil + } + + return a.SearchBatch(ctx, []string{keyword}, req) +} + +func (a *GupyAdapter) SearchBatch(ctx context.Context, keywords []string, req domain.ScrapeRequest) ([]domain.Job, error) { + queryKeywords := gupyExpandedQueries(keywords, req) + if len(queryKeywords) == 0 { + return nil, nil + } + + queries := make([]string, 0, len(queryKeywords)) + for query := range queryKeywords { + queries = append(queries, query) + } + sort.Strings(queries) + queries = a.nextQuerySlot(queries, gupyQueryLimit()) + + type collectedJob struct { + raw gupyJob + keywords []string + } + + collected := make(map[string]collectedJob) + order := make([]string, 0) + for _, query := range queries { + rawJobs, err := a.fetchAll(ctx, query, req) + if err != nil { + slog.Warn("gupy: query ignorada por erro", + "query", query, + "error", err, + ) + continue + } + + for _, raw := range rawJobs { + if !gupyMatchesRequest(raw, req) { + continue + } + + job := gupyToJob(raw, queryKeywords[query]) + if job.URL == "" || job.Title == "" { + continue + } + if !gupyLooksLikeTechnologyJob(job, queryKeywords[query]) { + continue + } + + key := gupyDedupeKey(raw) + if key == "" { + key = job.URL + } + + current, ok := collected[key] + if !ok { + collected[key] = collectedJob{raw: raw, keywords: queryKeywords[query]} + order = append(order, key) + continue + } + + current.keywords = adapterutil.UniqueTrimmedStrings(append(current.keywords, queryKeywords[query]...)) + collected[key] = current + } + } + + jobs := make([]domain.Job, 0, len(order)) + for _, key := range order { + collectedJob := collected[key] + keywords := adapterutil.UniqueTrimmedStrings(collectedJob.keywords) + if len(keywords) == 0 { + continue + } + jobs = append(jobs, gupyToJob(collectedJob.raw, keywords)) + } + + return jobs, nil +} + +func gupyQueryLimit() int { + value := strings.TrimSpace(os.Getenv("GUPY_QUERY_LIMIT")) + if value == "" { + return gupyDefaultQueryLimit + } + parsed, err := strconv.Atoi(value) + if err != nil || parsed <= 0 { + return gupyDefaultQueryLimit + } + return parsed +} + +func (a *GupyAdapter) nextQuerySlot(queries []string, limit int) []string { + if len(queries) == 0 { + return nil + } + if limit <= 0 || limit >= len(queries) { + return queries + } + + a.mu.Lock() + defer a.mu.Unlock() + + offset := a.nextQueryOffset % len(queries) + a.nextQueryOffset = (offset + limit) % len(queries) + + slot := make([]string, 0, limit) + for i := 0; i < limit; i++ { + slot = append(slot, queries[(offset+i)%len(queries)]) + } + + slog.Info("gupy: usando slot rotativo de queries", + "selected", len(slot), + "total", len(queries), + ) + + return slot +} + +func (a *GupyAdapter) fetchAll(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]gupyJob, error) { + limit := a.pageLimit + if limit <= 0 { + limit = gupyDefaultPageLimit + } + + maxOffset := a.maxOffset + if maxOffset <= 0 { + maxOffset = gupyDefaultMaxOffset + } + + batchSize := a.batchSize + if batchSize <= 0 { + batchSize = gupyDefaultBatchSize + } + + all := make([]gupyJob, 0, limit) + for base := 0; base <= maxOffset; base += limit * batchSize { + offsets := make([]int, 0, batchSize) + for i := 0; i < batchSize; i++ { + offset := base + i*limit + if offset > maxOffset { + break + } + offsets = append(offsets, offset) + } + if len(offsets) == 0 { + break + } + + pages := make([][]gupyJob, len(offsets)) + errs := make([]error, len(offsets)) + + var wg sync.WaitGroup + for i, offset := range offsets { + wg.Add(1) + go func(i, offset int) { + defer wg.Done() + page, err := a.fetchPageWithRetry(ctx, keyword, offset, limit, req) + pages[i] = page + errs[i] = err + }(i, offset) + } + wg.Wait() + + stop := false + for i, page := range pages { + if errs[i] != nil { + if gupyIsFullSweepEnd(errs[i], keyword, offsets[i]) { + stop = true + break + } + return nil, errs[i] + } + all = append(all, page...) + if len(page) < limit { + stop = true + break + } + } + + if stop { + break + } + } + + return all, nil +} + +func (a *GupyAdapter) fetchPageWithRetry(ctx context.Context, keyword string, offset, limit int, req domain.ScrapeRequest) ([]gupyJob, error) { + var lastErr error + for attempt := 0; attempt < 3; attempt++ { + page, err := a.fetchPage(ctx, keyword, offset, limit, req) + if err == nil { + return page, nil + } + lastErr = err + + if !gupyIsRetryable(err) { + return nil, err + } + + timer := time.NewTimer(time.Duration(250*(attempt+1)) * time.Millisecond) + select { + case <-ctx.Done(): + timer.Stop() + return nil, ctx.Err() + case <-timer.C: + } + } + + return nil, lastErr +} + +func gupyIsRetryable(err error) bool { + if err == nil { + return false + } + + var httpErr *gupyHTTPError + if errors.As(err, &httpErr) { + return httpErr.StatusCode() >= 500 + } + + if errors.Is(err, context.DeadlineExceeded) || errors.Is(err, context.Canceled) { + return true + } + + var netErr net.Error + return errors.As(err, &netErr) +} + +func gupyIsFullSweepEnd(err error, keyword string, offset int) bool { + if strings.TrimSpace(keyword) != "" || offset == 0 { + return false + } + + var httpErr *gupyHTTPError + if !errors.As(err, &httpErr) { + return false + } + + return httpErr.StatusCode() == http.StatusBadRequest +} + +func (a *GupyAdapter) fetchPage(ctx context.Context, keyword string, offset, limit int, req domain.ScrapeRequest) ([]gupyJob, error) { + pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond + if pageTimeout <= 0 { + pageTimeout = 15 * time.Second + } + + pageCtx, cancel := context.WithTimeout(ctx, pageTimeout) + defer cancel() + + endpoint, err := gupySearchURL(a.baseURL, keyword, offset, limit, req) + if err != nil { + return nil, err + } + + httpReq, err := http.NewRequestWithContext(pageCtx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, fmt.Errorf("gupy: build request: %w", err) + } + httpReq.Header.Set("Accept", "application/json") + httpReq.Header.Set("User-Agent", "JobsScraper/1.0") + + resp, err := a.client.Do(httpReq) + if err != nil { + return nil, fmt.Errorf("gupy: http do: %w", err) + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return nil, &gupyHTTPError{statusCode: resp.StatusCode, keyword: keyword, offset: offset} + } + + var data gupyResponse + if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { + return nil, fmt.Errorf("gupy: decode json: %w", err) + } + + return data.Data, nil +} + +func gupySearchURL(baseURL, keyword string, offset, limit int, req domain.ScrapeRequest) (string, error) { + u, err := url.Parse(baseURL) + if err != nil { + return "", fmt.Errorf("gupy: parse base url: %w", err) + } + + q := u.Query() + if strings.TrimSpace(keyword) != "" { + q.Set("jobName", keyword) + } + if req.RemoteOnly { + q.Set("workplaceType", "remote") + } + q.Set("offset", fmt.Sprint(offset)) + q.Set("limit", fmt.Sprint(limit)) + u.RawQuery = q.Encode() + + return u.String(), nil +} + +func gupyMatchesRequest(job gupyJob, req domain.ScrapeRequest) bool { + if req.RemoteOnly && !gupyIsRemote(job) { + return false + } + + if location := strings.TrimSpace(req.SearchLocation); location != "" && !gupyIsRemote(job) { + if !adapterutil.ContainsNormalized(gupyLocation(job), location) { + return false + } + } + + return true +} + +func gupyToJob(job gupyJob, keywords []string) domain.Job { + source := "Gupy" + keyword := "" + if len(keywords) > 0 { + keyword = keywords[0] + } + + return domain.Job{ + ID: gupyJobID(job), + Title: strings.TrimSpace(job.Name), + Company: strings.TrimSpace(job.CareerPageName), + Location: gupyLocation(job), + URL: gupyURL(job), + Modality: gupyModality(job), + Description: strings.TrimSpace(job.Description), + PostedAt: strings.TrimSpace(job.PublishedDate), + Source: source, + Sources: []string{source}, + Keyword: keyword, + Keywords: keywords, + } +} + +func gupyJobID(job gupyJob) string { + if job.ID != nil { + if id := strings.TrimSpace(fmt.Sprint(job.ID)); id != "" { + return "gupy:" + id + } + } + if u := gupyURL(job); u != "" { + return u + } + return strings.TrimSpace(job.Name) +} + +func gupyDedupeKey(job gupyJob) string { + if job.ID != nil { + if id := strings.TrimSpace(fmt.Sprint(job.ID)); id != "" { + return id + } + } + return gupyURL(job) +} + +func gupyURL(job gupyJob) string { + if job.JobURL != "" { + return strings.TrimSpace(job.JobURL) + } + return strings.TrimSpace(job.CareerPageURL) +} + +func gupyLocation(job gupyJob) string { + return strings.Join(adapterutil.NonEmptyStrings([]string{ + job.City, + job.State, + job.Country, + }), " / ") +} + +func gupyModality(job gupyJob) string { + workplaceType := strings.TrimSpace(job.WorkplaceType) + normalized := adapterutil.NormalizeText(workplaceType) + + switch { + case job.IsRemoteWork || strings.Contains(normalized, "remote") || strings.Contains(normalized, "remoto"): + return "Remoto" + case strings.Contains(normalized, "hybrid") || strings.Contains(normalized, "hibrido"): + return "Híbrido" + case strings.Contains(normalized, "onsite") || strings.Contains(normalized, "on site"): + return "Presencial" + default: + return workplaceType + } +} + +func gupyIsRemote(job gupyJob) bool { + return gupyModality(job) == "Remoto" +} + +func gupyExpandedQueries(keywords []string, req domain.ScrapeRequest) map[string][]string { + out := make(map[string][]string) + for _, keyword := range keywords { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + + for _, query := range gupyQueryVariants(keyword) { + query = strings.TrimSpace(query) + if query == "" { + continue + } + out[query] = adapterutil.UniqueTrimmedStrings(append(out[query], keyword)) + } + } + + if gupyRawDiscoveryEnabled() && gupyLooksLikeTechnologySearch(keywords) { + for _, query := range gupyRawDiscoveryQueries() { + out[query] = adapterutil.UniqueTrimmedStrings(append(out[query], query)) + } + } + + if gupyFullSweepEnabled() && gupyLooksLikeTechnologySearch(keywords) { + fullSweepKeyword := "gupy:full-sweep" + if req.RemoteOnly { + fullSweepKeyword = "gupy:remote-full-sweep" + } + out[""] = adapterutil.UniqueTrimmedStrings(append(out[""], fullSweepKeyword)) + } + + return out +} + +func gupyRawDiscoveryEnabled() bool { + value := strings.TrimSpace(os.Getenv("GUPY_RAW_DISCOVERY_ENABLED")) + if value == "" { + return true + } + + return !strings.EqualFold(value, "false") +} + +func gupyFullSweepEnabled() bool { + value := strings.TrimSpace(os.Getenv("GUPY_FULL_SWEEP_ENABLED")) + if value != "" { + return !strings.EqualFold(value, "false") + } + + value = strings.TrimSpace(os.Getenv("GUPY_FULL_REMOTE_SWEEP_ENABLED")) + if value == "" { + return true + } + + return !strings.EqualFold(value, "false") +} + +func gupyLooksLikeTechnologySearch(keywords []string) bool { + text := adapterutil.NormalizeText(strings.Join(keywords, " ")) + signals := []string{ + "software", "developer", "desenvolvedor", "backend", "frontend", "full stack", + "mobile", "engineer", "engenheiro", "devops", "data", "dados", "qa", "tech", + "java", "python", "php", "javascript", "typescript", "node", "react", "angular", + "vue", "golang", "kotlin", "swift", "flutter", "cloud", "aws", "azure", "gcp", + } + + for _, signal := range signals { + if strings.Contains(text, signal) { + return true + } + } + + return false +} + +func gupyRawDiscoveryQueries() []string { + return []string{ + "software", + "desenvolvedor", + "desenvolvedora", + "engenheiro de software", + "engenheira de software", + "programador", + "programadora", + "backend", + "back end", + "frontend", + "front end", + "full stack", + "fullstack", + "mobile", + "dados", + "data", + "business intelligence", + "analytics", + "devops", + "sre", + "cloud", + "qa", + "quality assurance", + "automação", + "segurança da informação", + "cybersecurity", + "java", + "spring boot", + "python", + "django", + "fastapi", + "php", + "laravel", + "javascript", + "typescript", + "node", + "node.js", + "nestjs", + "react", + "next.js", + "react native", + "angular", + "vue", + "golang", + "kotlin", + "swift", + "flutter", + "ruby", + "rails", + ".net", + "dotnet", + "c#", + "kubernetes", + "aws", + "azure", + "gcp", + } +} + +func gupyLooksLikeTechnologyJob(job domain.Job, _ []string) bool { + title := adapterutil.NormalizeText(job.Title) + description := adapterutil.NormalizeText(job.Description) + + if title == "" { + return false + } + + if gupyIsNonConcreteOpening(title) { + return false + } + + if gupyHasHardNegativeTitle(title) { + return false + } + + if gupyHasStrongTechnologyRole(title) { + return true + } + + if gupyHasTechnologySignal(title) && gupyHasRoleSignal(title) { + return true + } + + if gupyHasGenericTechnologyTitle(title) { + combined := strings.Join(adapterutil.NonEmptyStrings([]string{title, description}), " ") + if gupyHasStrongTechnologyRole(combined) || gupyHasTechnologySignal(combined) { + return true + } + } + + return false +} + +func gupyIsNonConcreteOpening(title string) bool { + terms := []string{ + "banco de talentos", + "talent pool", + "cadastro reserva", + } + + for _, term := range terms { + if adapterutil.ContainsNormalized(title, term) { + return true + } + } + + return false +} + +func gupyHasHardNegativeTitle(title string) bool { + if gupyHasStrongTechnologyRole(title) || (gupyHasTechnologySignal(title) && gupyHasRoleSignal(title)) { + return false + } + + hardNegativeTerms := []string{ + "motorista", "motorista de van", "atendente", "operador", "operadora", + "promotor", "promotora", "vendedor", "vendedora", "recepcionista", + "auxiliar", "assistente", "estagiario administrativo", + "analista qualidade", "analista de qualidade", + } + + for _, term := range hardNegativeTerms { + if adapterutil.ContainsNormalized(title, term) { + return true + } + } + + return false +} + +func gupyHasStrongTechnologyRole(text string) bool { + strongTerms := []string{ + "software engineer", "software developer", "engenheiro de software", + "engenheira de software", "arquiteto de software", "arquiteta de software", + "software architect", "desenvolvedor", "desenvolvedora", + "programador", "programadora", "backend", "back end", "frontend", + "front end", "full stack", "fullstack", "mobile developer", + "desenvolvedor mobile", "devops", "site reliability", "sre", + "data engineer", "data analyst", "analytics engineer", + "engenheiro de dados", "cientista de dados", "analista de dados", + "qa engineer", "analista qa", "quality assurance", "sdet", + "security engineer", "engenheiro de seguranca", "cloud engineer", + "platform engineer", "go engineer", "tech lead", "technical lead", + } + + for _, term := range strongTerms { + if adapterutil.ContainsNormalized(text, term) { + return true + } + } + + return false +} + +func gupyHasGenericTechnologyTitle(title string) bool { + terms := []string{ + "analista de sistemas", + "arquiteto de software", + "arquiteta de software", + "software", + "tecnologia da informacao", + } + + for _, term := range terms { + if adapterutil.ContainsNormalized(title, term) { + return true + } + } + + return false +} + +func gupyHasTechnologySignal(text string) bool { + terms := []string{ + "java", "spring", "python", "django", "fastapi", "php", "laravel", + "symfony", "javascript", "typescript", "node", "nodejs", "node js", + "nestjs", "react", "nextjs", "next js", "react native", "angular", + "vue", "golang", "kotlin", "swift", "flutter", "ruby", "rails", + "dotnet", "csharp", "kubernetes", "docker", "aws", "azure", "gcp", + "terraform", "postgresql", "mysql", "mongodb", "redis", "kafka", + } + + for _, term := range terms { + if adapterutil.ContainsNormalized(text, term) { + return true + } + } + + return false +} + +func gupyHasRoleSignal(text string) bool { + terms := []string{ + "developer", "desenvolvedor", "desenvolvedora", "engineer", + "engenheiro", "engenheira", "programador", "programadora", + "analista de sistemas", "analista desenvolvedor", + } + + for _, term := range terms { + if adapterutil.ContainsNormalized(text, term) { + return true + } + } + + return false +} + +func gupyQueryVariants(keyword string) []string { + normalized := adapterutil.NormalizeText(keyword) + var variants []string + add := func(values ...string) { + for _, value := range values { + value = strings.TrimSpace(value) + if value != "" { + variants = append(variants, value) + } + } + } + + add(keyword) + + switch normalized { + case "software engineer", "software developer", "desenvolvedor de software": + add("software developer", "software engineer", "desenvolvedor de software", "engenheiro de software") + case "developer", "desenvolvedor": + add("developer", "desenvolvedor") + case "backend developer", "backend engineer", "desenvolvedor backend": + add("backend developer", "backend engineer", "desenvolvedor backend", "desenvolvedor back end") + case "frontend developer", "frontend engineer", "front end developer", "desenvolvedor frontend": + add("frontend developer", "front-end developer", "desenvolvedor frontend", "desenvolvedor front end") + case "full stack developer": + add("full stack developer", "fullstack developer", "desenvolvedor full stack", "desenvolvedor fullstack") + case "mobile developer": + add("mobile developer", "desenvolvedor mobile") + case "platform engineer": + add("platform engineer", "engenheiro de plataforma") + case "devops engineer": + add("devops engineer", "devops", "engenheiro devops") + case "data engineer": + add("data engineer", "engenheiro de dados") + case "qa engineer": + add("qa engineer", "quality assurance", "analista de qa", "analista qa", "tester") + case "tech lead": + add("tech lead", "technical lead", "lider tecnico", "líder técnico") + case "engineering manager": + add("engineering manager", "gerente de engenharia") + case "site reliability engineer": + add("site reliability engineer", "sre", "engenheiro de confiabilidade") + case "security engineer": + add("security engineer", "engenheiro de segurança", "seguranca da informacao", "segurança da informação") + case "automation engineer": + add("automation engineer", "engenheiro de automação", "automacao") + case "sdet": + add("sdet", "software development engineer in test") + } + + addTechnologyVariants(normalized, add) + + return adapterutil.UniqueTrimmedStrings(variants) +} + +func addTechnologyVariants(normalized string, add func(...string)) { + technologies := map[string][]string{ + "java developer": {"java developer", "desenvolvedor java", "java"}, + "spring boot developer": {"spring boot developer", "desenvolvedor spring boot", "spring boot"}, + "python developer": {"python developer", "desenvolvedor python", "python"}, + "django developer": {"django developer", "desenvolvedor django", "django"}, + "fastapi developer": {"fastapi developer", "desenvolvedor fastapi", "fastapi"}, + "php developer": {"php developer", "desenvolvedor php", "php"}, + "laravel developer": {"laravel developer", "desenvolvedor laravel", "laravel"}, + "symfony developer": {"symfony developer", "desenvolvedor symfony", "symfony"}, + "javascript developer": {"javascript developer", "desenvolvedor javascript", "javascript"}, + "typescript developer": {"typescript developer", "desenvolvedor typescript", "typescript"}, + "node developer": {"node developer", "node.js developer", "desenvolvedor node", "node.js", "nodejs"}, + "node.js developer": {"node.js developer", "desenvolvedor node", "node.js", "nodejs"}, + "nestjs developer": {"nestjs developer", "desenvolvedor nestjs", "nest.js", "nestjs"}, + "react developer": {"react developer", "desenvolvedor react", "react"}, + "next.js developer": {"next.js developer", "desenvolvedor next", "next.js", "nextjs"}, + "react native developer": {"react native developer", "desenvolvedor react native", "react native"}, + "angular developer": {"angular developer", "desenvolvedor angular", "angular"}, + "vue developer": {"vue developer", "desenvolvedor vue", "vue.js", "vuejs"}, + "golang developer": {"golang developer", "desenvolvedor golang", "golang"}, + "go developer": {"go developer", "desenvolvedor go", "golang"}, + "c# developer": {"c# developer", "desenvolvedor c#", "c#"}, + ".net developer": {".net developer", "desenvolvedor .net", ".net", "dotnet"}, + "ruby developer": {"ruby developer", "desenvolvedor ruby", "ruby"}, + "rails developer": {"rails developer", "ruby on rails", "desenvolvedor rails"}, + "kotlin developer": {"kotlin developer", "desenvolvedor kotlin", "kotlin"}, + "swift developer": {"swift developer", "desenvolvedor swift", "swift"}, + "flutter developer": {"flutter developer", "desenvolvedor flutter", "flutter"}, + "rust developer": {"rust developer", "desenvolvedor rust", "rust"}, + "kubernetes engineer": {"kubernetes engineer", "kubernetes", "k8s"}, + "cloud engineer": {"cloud engineer", "engenheiro cloud", "cloud"}, + "aws engineer": {"aws engineer", "engenheiro aws", "aws"}, + "azure engineer": {"azure engineer", "engenheiro azure", "azure"}, + "gcp engineer": {"gcp engineer", "engenheiro gcp", "google cloud"}, + } + + if variants, ok := technologies[normalized]; ok { + add(variants...) + } +} diff --git a/scraper-go/internal/adapters/gupy/adapter_test.go b/scraper-go/internal/adapters/gupy/adapter_test.go new file mode 100644 index 0000000..c0fc84e --- /dev/null +++ b/scraper-go/internal/adapters/gupy/adapter_test.go @@ -0,0 +1,526 @@ +package gupy + +import ( + "context" + "net/http" + "strconv" + "strings" + "sync" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestGupySearchMapsRemoteJobs(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "false") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.String() != "https://example.test/jobs?jobName=Data+Analyst&limit=100&offset=0&workplaceType=remote" { + t.Fatalf("unexpected endpoint: %s", req.URL.String()) + } + + return testResponse(`{ + "data": [ + { + "id": 123, + "name": "Data Analyst", + "careerPageName": "Acme Dados", + "jobUrl": "https://acme.gupy.io/jobs/123", + "city": "Sao Paulo", + "state": "SP", + "country": "Brasil", + "workplaceType": "remote", + "isRemoteWork": false, + "publishedDate": "2026-07-27T00:00:00Z", + "description": "Build dashboards and metrics." + }, + { + "id": 456, + "name": "Finance Analyst", + "careerPageName": "Acme Dados", + "jobUrl": "https://acme.gupy.io/jobs/456", + "city": "Rio de Janeiro", + "state": "RJ", + "country": "Brasil", + "workplaceType": "on-site" + } + ] + }`), nil + }) + + jobs, err := adapter.Search(context.Background(), "Data Analyst", domain.ScrapeRequest{RemoteOnly: true}) + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected one remote job, got %d", len(jobs)) + } + + job := jobs[0] + if job.ID != "gupy:123" { + t.Fatalf("unexpected ID: %q", job.ID) + } + if job.Title != "Data Analyst" { + t.Fatalf("unexpected title: %q", job.Title) + } + if job.Company != "Acme Dados" { + t.Fatalf("unexpected company: %q", job.Company) + } + if job.Location != "Sao Paulo / SP / Brasil" { + t.Fatalf("unexpected location: %q", job.Location) + } + if job.Modality != "Remoto" { + t.Fatalf("expected remote modality, got %q", job.Modality) + } + if job.URL != "https://acme.gupy.io/jobs/123" { + t.Fatalf("unexpected URL: %q", job.URL) + } + if job.Source != "Gupy" || strings.Join(job.Sources, ",") != "Gupy" { + t.Fatalf("unexpected source metadata: %q %#v", job.Source, job.Sources) + } + if job.Keyword != "Data Analyst" || strings.Join(job.Keywords, ",") != "Data Analyst" { + t.Fatalf("unexpected keyword metadata: %q %#v", job.Keyword, job.Keywords) + } +} + +func TestGupySearchPaginatesUntilShortPage(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "false") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.pageLimit = 2 + adapter.maxOffset = 10 + adapter.batchSize = 1 + + var ( + mu sync.Mutex + offsets []string + ) + + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + mu.Lock() + offsets = append(offsets, req.URL.Query().Get("offset")) + mu.Unlock() + + switch req.URL.Query().Get("offset") { + case "0": + return testResponse(`{"data":[ + {"id":"1","name":"Go Engineer","careerPageName":"Acme","jobUrl":"https://acme.gupy.io/jobs/1","workplaceType":"remote"}, + {"id":"2","name":"Backend Engineer","careerPageName":"Acme","jobUrl":"https://acme.gupy.io/jobs/2","workplaceType":"remote"} + ]}`), nil + case "2": + return testResponse(`{"data":[ + {"id":"3","name":"Platform Engineer","careerPageName":"Acme","jobUrl":"https://acme.gupy.io/jobs/3","workplaceType":"remote"} + ]}`), nil + default: + t.Fatalf("unexpected offset: %s", req.URL.Query().Get("offset")) + return testResponse(`{"data":[]}`), nil + } + }) + + jobs, err := adapter.Search(context.Background(), "engineer", domain.ScrapeRequest{}) + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 3 { + t.Fatalf("expected three jobs across pages, got %d", len(jobs)) + } + + mu.Lock() + defer mu.Unlock() + if strings.Join(offsets, ",") != "0,2" { + t.Fatalf("expected offsets 0,2, got %#v", offsets) + } +} + +func TestGupySearchAcceptsIsRemoteWorkFlagAndCareerPageURLFallback(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "false") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + return testResponse(`{"data":[{ + "id": "abc", + "name": "Software Engineer", + "careerPageName": "Acme", + "careerPageUrl": "https://acme.gupy.io", + "city": "Campinas", + "state": "SP", + "country": "Brasil", + "workplaceType": "hybrid", + "isRemoteWork": true + }]}`), nil + }) + + jobs, err := adapter.Search(context.Background(), "product", domain.ScrapeRequest{RemoteOnly: true}) + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected isRemoteWork job to pass remote filter, got %d", len(jobs)) + } + if jobs[0].URL != "https://acme.gupy.io" { + t.Fatalf("expected career page URL fallback, got %q", jobs[0].URL) + } + if jobs[0].Modality != "Remoto" { + t.Fatalf("expected isRemoteWork to normalize modality to Remoto, got %q", jobs[0].Modality) + } +} + +func TestGupySearchHandlesNonOKStatus(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "false") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + return &http.Response{ + StatusCode: http.StatusTooManyRequests, + Body: http.NoBody, + Header: make(http.Header), + }, nil + }) + + _, err := adapter.fetchAll(context.Background(), "go", domain.ScrapeRequest{}) + if err == nil { + t.Fatal("expected error for non-OK response") + } + if !strings.Contains(err.Error(), "status inesperado 429") { + t.Fatalf("unexpected error: %v", err) + } +} + +func TestGupySearchBatchExpandsPortugueseTechnologyQueries(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "false") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + + seenQueries := make(map[string]bool) + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + query := req.URL.Query().Get("jobName") + seenQueries[query] = true + + if req.URL.Query().Get("offset") != "0" { + return testResponse(`{"data":[]}`), nil + } + + switch query { + case "desenvolvedor java": + return testResponse(`{"data":[{ + "id": "java-1", + "name": "Pessoa Desenvolvedora Java", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/java-1", + "workplaceType": "remote" + }]}`), nil + case "react": + return testResponse(`{"data":[{ + "id": "react-1", + "name": "Desenvolvedor React", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/react-1", + "workplaceType": "remote" + }]}`), nil + default: + return testResponse(`{"data":[]}`), nil + } + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"java developer", "react developer"}, domain.ScrapeRequest{RemoteOnly: true}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 2 { + t.Fatalf("expected two expanded-query jobs, got %d", len(jobs)) + } + if !seenQueries["desenvolvedor java"] || !seenQueries["react"] { + t.Fatalf("expected expanded queries to be searched, got %#v", seenQueries) + } + for _, job := range jobs { + if len(job.Keywords) != 1 { + t.Fatalf("expected original keyword metadata, got %#v", job.Keywords) + } + } +} + +func TestGupySearchBatchAddsRawDiscoveryQueries(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "true") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "false") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + + seenQueries := make(map[string]bool) + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + query := req.URL.Query().Get("jobName") + seenQueries[query] = true + + if query != "software" || req.URL.Query().Get("offset") != "0" { + return testResponse(`{"data":[]}`), nil + } + + return testResponse(`{"data":[{ + "id": "raw-1", + "name": "Software Engineer", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/raw-1", + "workplaceType": "remote" + }]}`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"java developer"}, domain.ScrapeRequest{RemoteOnly: true}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected raw discovery job, got %d", len(jobs)) + } + if !seenQueries["software"] { + t.Fatalf("expected software raw discovery query, got %#v", seenQueries) + } + if jobs[0].Keyword != "software" { + t.Fatalf("expected raw query keyword metadata, got %q", jobs[0].Keyword) + } +} + +func TestGupySearchBatchAddsFullRemoteSweep(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "true") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "true") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + + var sawFullSweep bool + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.Query().Get("jobName") != "" { + return testResponse(`{"data":[]}`), nil + } + if req.URL.Query().Get("workplaceType") != "remote" { + t.Fatalf("expected remote full sweep to request workplaceType=remote, got %q", req.URL.String()) + } + sawFullSweep = true + + return testResponse(`{"data":[{ + "id": "remote-1", + "name": "Pessoa Engenheira de Software", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/remote-1", + "workplaceType": "remote" + }]}`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"software engineer"}, domain.ScrapeRequest{RemoteOnly: true}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected one full-sweep job, got %d", len(jobs)) + } + if !sawFullSweep { + t.Fatal("expected full remote sweep request") + } + if jobs[0].Keyword != "gupy:remote-full-sweep" { + t.Fatalf("expected full-sweep keyword metadata, got %q", jobs[0].Keyword) + } +} + +func TestGupySearchBatchRejectsAdministrativeJobsFromFullSweep(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "true") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "true") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.Query().Get("jobName") != "" { + return testResponse(`{"data":[]}`), nil + } + + return testResponse(`{"data":[ + { + "id": "admin-0", + "name": "MOTORISTA DE VAN - AEROPORTO_FLORIANOPOLIS (SC)", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/admin-0", + "workplaceType": "remote", + "description": "Transporte de clientes e atendimento operacional." + }, + { + "id": "admin-1", + "name": "Assistente de Sistemas - Operações Comerciais", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/admin-1", + "workplaceType": "remote", + "description": "Atendimento, operação comercial e suporte administrativo." + }, + { + "id": "admin-2", + "name": "Assistente de Negócios - Central de Relacionamento", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/admin-2", + "workplaceType": "remote" + }, + { + "id": "admin-3", + "name": "ASSISTENTE CENTRAL DE RESERVAS", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/admin-3", + "workplaceType": "remote" + }, + { + "id": "admin-4", + "name": "ANALISTA QUALIDADE III", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/admin-4", + "workplaceType": "remote", + "description": "Processos de qualidade operacional e auditoria." + }, + { + "id": "admin-5", + "name": "Atendente", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/admin-5", + "workplaceType": "remote" + }, + { + "id": "talent-1", + "name": "[Banco de Talentos] Pessoa Desenvolvedora Frontend Junior", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/talent-1", + "workplaceType": "remote", + "description": "React e TypeScript." + }, + { + "id": "tech-1", + "name": "Pessoa Desenvolvedora Backend", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/tech-1", + "workplaceType": "remote" + } + ]}`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"backend developer"}, domain.ScrapeRequest{RemoteOnly: true}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected only the technical job to pass, got %d", len(jobs)) + } + if jobs[0].ID != "gupy:tech-1" { + t.Fatalf("expected technical job to remain, got %q", jobs[0].ID) + } +} + +func TestGupySearchBatchAddsFullSweepForAllModalities(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "true") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.batchSize = 1 + + var sawFullSweep bool + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.Query().Get("jobName") != "" { + return testResponse(`{"data":[]}`), nil + } + if req.URL.Query().Get("workplaceType") != "" { + t.Fatalf("expected all-modality full sweep without workplaceType, got %q", req.URL.String()) + } + sawFullSweep = true + + return testResponse(`{"data":[{ + "id": "hybrid-1", + "name": "Pessoa Desenvolvedora Backend", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/hybrid-1", + "workplaceType": "hybrid" + }]}`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"backend developer"}, domain.ScrapeRequest{RemoteOnly: false}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected one all-modality full-sweep job, got %d", len(jobs)) + } + if !sawFullSweep { + t.Fatal("expected all-modality full sweep request") + } + if jobs[0].Keyword != "gupy:full-sweep" { + t.Fatalf("expected full-sweep keyword metadata, got %q", jobs[0].Keyword) + } + if jobs[0].Modality != "Híbrido" { + t.Fatalf("expected hybrid modality, got %q", jobs[0].Modality) + } +} + +func TestGupyFullSweepKeepsCollectedJobsWhenHighOffsetReturnsBadRequest(t *testing.T) { + t.Setenv("GUPY_RAW_DISCOVERY_ENABLED", "false") + t.Setenv("GUPY_FULL_SWEEP_ENABLED", "true") + t.Setenv("GUPY_FULL_REMOTE_SWEEP_ENABLED", "false") + + adapter := NewGupy() + adapter.baseURL = "https://example.test/jobs" + adapter.pageLimit = 100 + adapter.batchSize = 1 + adapter.maxOffset = 10100 + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.Query().Get("jobName") != "" { + return testResponse(`{"data":[]}`), nil + } + + offset, err := strconv.Atoi(req.URL.Query().Get("offset")) + if err != nil { + t.Fatalf("unexpected offset: %q", req.URL.Query().Get("offset")) + } + if offset >= 10000 { + return &http.Response{ + StatusCode: http.StatusBadRequest, + Body: http.NoBody, + Header: make(http.Header), + }, nil + } + + return testResponse(`{"data":[{ + "id": "full-1", + "name": "Pessoa Desenvolvedora Full Stack", + "careerPageName": "Acme", + "jobUrl": "https://acme.gupy.io/jobs/full-1", + "workplaceType": "hybrid" + }]}`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"software engineer"}, domain.ScrapeRequest{RemoteOnly: false}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected collected job to survive full-sweep 400, got %d", len(jobs)) + } +} diff --git a/scraper-go/internal/adapters/gupy/test_helpers_test.go b/scraper-go/internal/adapters/gupy/test_helpers_test.go new file mode 100644 index 0000000..7fd64fd --- /dev/null +++ b/scraper-go/internal/adapters/gupy/test_helpers_test.go @@ -0,0 +1,15 @@ +package gupy + +import ( + "net/http" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" +) + +func testHTTPClient(fn testutil.RoundTripFunc) *http.Client { + return testutil.HTTPClient(fn) +} + +func testResponse(body string) *http.Response { + return testutil.Response(body) +} diff --git a/scraper-go/internal/adapters/inhire/adapter.go b/scraper-go/internal/adapters/inhire/adapter.go new file mode 100644 index 0000000..4b7dd1d --- /dev/null +++ b/scraper-go/internal/adapters/inhire/adapter.go @@ -0,0 +1,773 @@ +package inhire + +import ( + "context" + "encoding/json" + "fmt" + "html" + "io" + "log/slog" + "net/http" + "os" + "regexp" + "sort" + "strconv" + "strings" + "sync" + "time" + "unicode" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "golang.org/x/text/transform" + "golang.org/x/text/unicode/norm" +) + +const ( + inhireDefaultAPIURL = "https://api.inhire.app/job-posts/public/pages" + inhireDefaultTenantsFile = "./internal/interfaces/inhireTenants.json" + inhireDefaultConcurrency = 16 + inhireDefaultDetailConcurrency = 8 + inhireDefaultDetailTimeout = 10 * time.Second + inhireDefaultDetailMaxBytes = int64(768 * 1024) + inhireMaxDescriptionRunes = 12000 +) + +type InHireAdapter struct { + client *http.Client + apiURL string + tenantsFile string + concurrency int +} + +type inhireTenant struct { + Slug string `json:"slug"` + TenantName string `json:"tenantName"` + JobsCount int `json:"jobsCount"` + ListCompany string `json:"listCompany"` + SampleJobs int `json:"sampleJobs"` +} + +type inhirePageResponse struct { + TenantName string `json:"tenantName"` + About string `json:"about"` + JobsPage []inhireJob `json:"jobsPage"` +} + +type inhireJob struct { + DisplayName string `json:"displayName"` + JobID string `json:"jobId"` + Status string `json:"status"` + WorkplaceType string `json:"workplaceType"` + Location string `json:"location"` + CareerPageID string `json:"careerPageId"` + CareerPageIDs []string `json:"careerPageIds"` +} + +func NewInHire() *InHireAdapter { + return &InHireAdapter{ + client: &http.Client{ + Timeout: 45 * time.Second, + Transport: &http.Transport{ + MaxIdleConnsPerHost: 24, + IdleConnTimeout: 90 * time.Second, + }, + }, + apiURL: inhireDefaultAPIURL, + tenantsFile: inhireDefaultTenantsFile, + concurrency: inhireDefaultConcurrency, + } +} + +func (a *InHireAdapter) SourceName() string { + return "InHire" +} + +func (a *InHireAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + return nil, nil + } + + return a.SearchBatch(ctx, []string{keyword}, req) +} + +func (a *InHireAdapter) SearchBatch(ctx context.Context, keywords []string, req domain.ScrapeRequest) ([]domain.Job, error) { + tenants, err := a.loadTenants() + if err != nil { + return nil, err + } + if len(tenants) == 0 { + return nil, nil + } + + rawJobs := a.fetchAllTenants(ctx, tenants, req) + jobs := make([]domain.Job, 0, len(rawJobs)) + var skippedStatus, skippedRemote int + for _, job := range rawJobs { + if status := strings.TrimSpace(job.raw.Status); status != "" && !strings.EqualFold(status, "published") { + skippedStatus++ + continue + } + if req.RemoteOnly && inhireModality(job.raw) != "Remoto" { + skippedRemote++ + continue + } + + matchedKeywords := inhireMatchingKeywords(job.raw, keywords) + mapped := inhireToJob(job.tenant, job.raw, matchedKeywords) + if mapped.Title == "" || mapped.URL == "" { + continue + } + + jobs = append(jobs, mapped) + } + + if inhireDetailEnrichmentEnabled() { + jobs = a.enrichJobsWithDetails(ctx, jobs, keywords) + } + + slog.Info("inhire: funil do adapter", + "tenants_catalog", len(tenants), + "jobs_raw", len(rawJobs), + "skipped_status", skippedStatus, + "skipped_remote", skippedRemote, + "jobs_returned", len(jobs), + ) + + return jobs, nil +} + +func (a *InHireAdapter) enrichJobsWithDetails(ctx context.Context, jobs []domain.Job, keywords []string) []domain.Job { + if len(jobs) == 0 { + return jobs + } + + concurrency := inhireDetailConcurrency() + timeout := inhireDetailTimeout() + eligible := make([]int, 0, len(jobs)) + for index := range jobs { + if inhireShouldEnrichDetails(jobs[index], keywords) { + eligible = append(eligible, index) + } + } + if len(eligible) == 0 { + return jobs + } + + started := time.Now() + var enriched, failed int + var mu sync.Mutex + work := make(chan int) + var wg sync.WaitGroup + workerCount := min(concurrency, len(eligible)) + + for i := 0; i < workerCount; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for index := range work { + detail, err := a.fetchJobDetailText(ctx, jobs[index].URL, timeout) + mu.Lock() + if err != nil { + failed++ + mu.Unlock() + continue + } + if detail != "" { + jobs[index].Description = strings.Join(adapterutil.NonEmptyStrings([]string{ + jobs[index].Description, + detail, + }), "\n") + jobs[index].Keywords = adapterutil.UniqueTrimmedStrings(append(jobs[index].Keywords, inhireMatchingKeywordsInText(detail, keywords)...)) + if jobs[index].Keyword == "" && len(jobs[index].Keywords) > 0 { + jobs[index].Keyword = jobs[index].Keywords[0] + } + enriched++ + } + mu.Unlock() + } + }() + } + +sendWork: + for _, index := range eligible { + select { + case <-ctx.Done(): + break sendWork + case work <- index: + } + } + close(work) + wg.Wait() + + slog.Info("inhire: detalhes enriquecidos", + "jobs_total", len(jobs), + "eligible", len(eligible), + "enriched", enriched, + "failed", failed, + "concurrency", concurrency, + "duration", time.Since(started).Round(time.Millisecond).String(), + ) + + return jobs +} + +func (a *InHireAdapter) fetchJobDetailText(ctx context.Context, url string, timeout time.Duration) (string, error) { + url = strings.TrimSpace(url) + if url == "" { + return "", nil + } + + reqCtx, cancel := context.WithTimeout(ctx, timeout) + defer cancel() + + httpReq, err := http.NewRequestWithContext(reqCtx, http.MethodGet, url, nil) + if err != nil { + return "", fmt.Errorf("inhire: build detail request: %w", err) + } + httpReq.Header.Set("Accept", "text/html,application/xhtml+xml") + httpReq.Header.Set("User-Agent", "JobsScraper/1.0") + + resp, err := a.client.Do(httpReq) + if err != nil { + return "", fmt.Errorf("inhire: detail http: %w", err) + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return "", fmt.Errorf("inhire: detail status inesperado %d", resp.StatusCode) + } + + body, err := io.ReadAll(io.LimitReader(resp.Body, inhireDefaultDetailMaxBytes)) + if err != nil { + return "", fmt.Errorf("inhire: read detail body: %w", err) + } + + return inhireExtractDetailText(string(body)), nil +} + +func inhireDetailEnrichmentEnabled() bool { + value := strings.TrimSpace(os.Getenv("INHIRE_ENRICH_DETAILS")) + if value == "" { + return false + } + return strings.EqualFold(value, "true") || value == "1" || strings.EqualFold(value, "yes") +} + +func inhireDetailConcurrency() int { + value := strings.TrimSpace(os.Getenv("INHIRE_DETAILS_CONCURRENCY")) + if value == "" { + return inhireDefaultDetailConcurrency + } + parsed, err := strconv.Atoi(value) + if err != nil || parsed <= 0 { + return inhireDefaultDetailConcurrency + } + return parsed +} + +func inhireDetailTimeout() time.Duration { + value := strings.TrimSpace(os.Getenv("INHIRE_DETAILS_TIMEOUT_MS")) + if value == "" { + return inhireDefaultDetailTimeout + } + parsed, err := strconv.Atoi(value) + if err != nil || parsed <= 0 { + return inhireDefaultDetailTimeout + } + return time.Duration(parsed) * time.Millisecond +} + +func inhireShouldEnrichDetails(job domain.Job, keywords []string) bool { + if strings.TrimSpace(job.URL) == "" { + return false + } + + mode := strings.ToLower(strings.TrimSpace(os.Getenv("INHIRE_DETAILS_MODE"))) + if mode == "all" { + return true + } + + if inhireTitleLooksTechnical(job.Title) { + return false + } + + if len(job.Keywords) == 0 { + return true + } + + if len(strings.TrimSpace(job.Description)) < 80 { + title := adapterutil.NormalizeText(job.Title) + for _, keyword := range keywords { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + if adapterutil.MatchesKeyword(title, keyword) || strings.Contains(title, adapterutil.NormalizeText(keyword)) { + return false + } + } + return true + } + + return false +} + +func inhireTitleLooksTechnical(title string) bool { + text := " " + adapterutil.NormalizeText(title) + " " + familyHints := []string{ + "backend", "front end", "frontend", "full stack", "fullstack", + "developer", "desenvolvedor", "engineer", "engenheiro", "devops", + "sre", "qa", "sdet", "software", "mobile", "android", "ios", + "dados", "data", "cloud", "security", "seguranca", + } + technologyHints := []string{ + "java", "spring", "python", "django", "fastapi", "php", "laravel", + "javascript", "typescript", "node", "nestjs", "react", "next js", + "angular", "vue", "go", "golang", "csharp", "dotnet", "ruby", + "rails", "kotlin", "swift", "flutter", "rust", "kubernetes", + "docker", "terraform", "aws", "azure", "gcp", "sql", + } + + hasFamily := false + for _, hint := range familyHints { + if strings.Contains(text, " "+hint+" ") { + hasFamily = true + break + } + } + if !hasFamily { + return false + } + + for _, hint := range technologyHints { + if strings.Contains(text, " "+hint+" ") { + return true + } + } + + return false +} + +func (a *InHireAdapter) loadTenants() ([]inhireTenant, error) { + path := strings.TrimSpace(os.Getenv("INHIRE_TENANTS_FILE")) + if path == "" { + path = a.tenantsFile + } + if path == "" { + path = inhireDefaultTenantsFile + } + + data, err := os.ReadFile(path) + if err != nil { + return nil, fmt.Errorf("inhire: leitura do arquivo '%s': %w", path, err) + } + + var tenants []inhireTenant + if err := json.Unmarshal(data, &tenants); err != nil { + return nil, fmt.Errorf("inhire: parse do arquivo '%s': %w", path, err) + } + + out := make([]inhireTenant, 0, len(tenants)) + seen := make(map[string]struct{}, len(tenants)) + for _, tenant := range tenants { + tenant.Slug = strings.TrimSpace(tenant.Slug) + tenant.TenantName = strings.TrimSpace(tenant.TenantName) + if tenant.Slug == "" { + continue + } + if _, ok := seen[tenant.Slug]; ok { + continue + } + seen[tenant.Slug] = struct{}{} + out = append(out, tenant) + } + + sort.Slice(out, func(i, j int) bool { + return out[i].Slug < out[j].Slug + }) + + return out, nil +} + +type inhireTenantJob struct { + tenant inhireTenant + raw inhireJob +} + +func (a *InHireAdapter) fetchAllTenants(ctx context.Context, tenants []inhireTenant, req domain.ScrapeRequest) []inhireTenantJob { + concurrency := a.concurrency + if concurrency <= 0 { + concurrency = inhireDefaultConcurrency + } + if concurrency > len(tenants) { + concurrency = len(tenants) + } + if concurrency <= 0 { + return nil + } + + jobs := make([]inhireTenantJob, 0) + var fetched, failed, withJobs int + var mu sync.Mutex + work := make(chan inhireTenant) + var wg sync.WaitGroup + + for i := 0; i < concurrency; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for tenant := range work { + page, err := a.fetchTenant(ctx, tenant, req) + if err != nil { + mu.Lock() + failed++ + mu.Unlock() + continue + } + if page.TenantName != "" { + tenant.TenantName = strings.TrimSpace(page.TenantName) + } + mu.Lock() + fetched++ + if len(page.JobsPage) > 0 { + withJobs++ + } + for _, job := range page.JobsPage { + jobs = append(jobs, inhireTenantJob{tenant: tenant, raw: job}) + } + mu.Unlock() + } + }() + } + + for _, tenant := range tenants { + select { + case <-ctx.Done(): + close(work) + wg.Wait() + return jobs + case work <- tenant: + } + } + close(work) + wg.Wait() + + slog.Info("inhire: tenants consultados", + "tenants_catalog", len(tenants), + "tenants_ok", fetched, + "tenants_failed", failed, + "tenants_with_jobs", withJobs, + "jobs_raw", len(jobs), + ) + + return jobs +} + +func (a *InHireAdapter) fetchTenant(ctx context.Context, tenant inhireTenant, req domain.ScrapeRequest) (inhirePageResponse, error) { + pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond + if pageTimeout <= 0 { + pageTimeout = 20 * time.Second + } + + reqCtx, cancel := context.WithTimeout(ctx, pageTimeout) + defer cancel() + + httpReq, err := http.NewRequestWithContext(reqCtx, http.MethodGet, a.apiURL, nil) + if err != nil { + return inhirePageResponse{}, fmt.Errorf("inhire: build request: %w", err) + } + httpReq.Header.Set("Accept", "application/json") + httpReq.Header.Set("Content-Type", "application/json") + httpReq.Header.Set("X-Inhire-Client", "web-inhire") + httpReq.Header.Set("X-Tenant", tenant.Slug) + httpReq.Header.Set("User-Agent", "JobsScraper/1.0") + + resp, err := a.client.Do(httpReq) + if err != nil { + return inhirePageResponse{}, fmt.Errorf("inhire: http do tenant %q: %w", tenant.Slug, err) + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return inhirePageResponse{}, fmt.Errorf("inhire: status inesperado %d para tenant %q", resp.StatusCode, tenant.Slug) + } + + var page inhirePageResponse + if err := json.NewDecoder(resp.Body).Decode(&page); err != nil { + return inhirePageResponse{}, fmt.Errorf("inhire: decode tenant %q: %w", tenant.Slug, err) + } + + return page, nil +} + +func inhireMatchesFilters(job inhireJob, req domain.ScrapeRequest) bool { + if status := strings.TrimSpace(job.Status); status != "" && !strings.EqualFold(status, "published") { + return false + } + + if req.RemoteOnly && inhireModality(job) != "Remoto" { + return false + } + + if location := strings.TrimSpace(req.SearchLocation); location != "" && inhireModality(job) != "Remoto" { + if !adapterutil.ContainsNormalized(inhireLocation(job), location) { + return false + } + } + + return true +} + +func inhireToJob(tenant inhireTenant, job inhireJob, keywords []string) domain.Job { + source := "InHire" + keyword := "" + if len(keywords) > 0 { + keyword = keywords[0] + } + + company := strings.TrimSpace(tenant.TenantName) + if company == "" { + company = strings.TrimSpace(tenant.Slug) + } + + return domain.Job{ + ID: inhireJobID(tenant, job), + Title: strings.TrimSpace(job.DisplayName), + Company: company, + Location: inhireLocation(job), + URL: inhireURL(tenant, job), + Modality: inhireModality(job), + Description: strings.Join(adapterutil.NonEmptyStrings([]string{ + strings.TrimSpace(job.Status), + strings.TrimSpace(job.WorkplaceType), + strings.TrimSpace(job.CareerPageID), + }), "\n"), + Source: source, + Sources: []string{source}, + Keyword: keyword, + Keywords: keywords, + } +} + +func inhireJobID(tenant inhireTenant, job inhireJob) string { + if job.JobID != "" { + return "inhire:" + strings.TrimSpace(tenant.Slug) + ":" + strings.TrimSpace(job.JobID) + } + return inhireURL(tenant, job) +} + +func inhireURL(tenant inhireTenant, job inhireJob) string { + if tenant.Slug == "" || job.JobID == "" { + return "" + } + + return fmt.Sprintf( + "https://%s.inhire.app/vagas/%s/%s", + strings.TrimSpace(tenant.Slug), + strings.TrimSpace(job.JobID), + inhireSlugify(job.DisplayName), + ) +} + +func inhireLocation(job inhireJob) string { + return strings.TrimSpace(job.Location) +} + +func inhireModality(job inhireJob) string { + normalized := adapterutil.NormalizeText(job.WorkplaceType) + switch { + case strings.Contains(normalized, "remote") || strings.Contains(normalized, "remoto"): + return "Remoto" + case strings.Contains(normalized, "hybrid") || strings.Contains(normalized, "hibrido"): + return "Híbrido" + case strings.Contains(normalized, "on site") || strings.Contains(normalized, "onsite") || strings.Contains(normalized, "presencial"): + return "Presencial" + default: + return strings.TrimSpace(job.WorkplaceType) + } +} + +func inhireMatchingKeywords(job inhireJob, keywords []string) []string { + text := adapterutil.NormalizeText(strings.Join([]string{ + job.DisplayName, + job.WorkplaceType, + job.Location, + }, " ")) + + matched := make([]string, 0, len(keywords)) + for _, keyword := range keywords { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + if adapterutil.MatchesKeyword(text, keyword) || strings.Contains(text, adapterutil.NormalizeText(keyword)) { + matched = append(matched, keyword) + } + } + + return adapterutil.UniqueTrimmedStrings(matched) +} + +func inhireMatchingKeywordsInText(text string, keywords []string) []string { + normalizedText := inhireNormalizeKeywordText(text) + matched := make([]string, 0, len(keywords)) + for _, keyword := range keywords { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + if inhireTextMatchesKeyword(normalizedText, keyword) { + matched = append(matched, keyword) + } + } + return adapterutil.UniqueTrimmedStrings(matched) +} + +func inhireTextMatchesKeyword(normalizedText, keyword string) bool { + terms := strings.Fields(inhireNormalizeKeywordText(keyword)) + if len(terms) == 0 { + return true + } + searchText := " " + normalizedText + " " + for _, term := range terms { + if term == "go" { + if strings.Contains(searchText, " go ") || strings.Contains(searchText, " golang ") { + continue + } + return false + } + if !strings.Contains(searchText, " "+term+" ") { + return false + } + } + return true +} + +func inhireNormalizeKeywordText(value string) string { + value = strings.ToLower(html.UnescapeString(value)) + replacer := strings.NewReplacer( + "/", " ", + "-", " ", + ".", " ", + ",", " ", + ":", " ", + ";", " ", + "(", " ", + ")", " ", + "[", " ", + "]", " ", + "{", " ", + "}", " ", + ) + return strings.Join(strings.Fields(replacer.Replace(value)), " ") +} + +var ( + inhireJSONScriptPattern = regexp.MustCompile(`(?is)]*(?:id=["']__NEXT_DATA__["']|type=["']application/json["'])[^>]*>(.*?)`) + inhireScriptStylePattern = regexp.MustCompile(`(?is)]*>.*?|]*>.*?`) + inhireTagPattern = regexp.MustCompile(`(?is)<[^>]+>`) + inhireWhitespacePattern = regexp.MustCompile(`\s+`) +) + +func inhireExtractDetailText(rawHTML string) string { + rawHTML = strings.TrimSpace(rawHTML) + if rawHTML == "" { + return "" + } + + parts := make([]string, 0, 4) + for _, match := range inhireJSONScriptPattern.FindAllStringSubmatch(rawHTML, -1) { + if len(match) < 2 { + continue + } + values := inhireExtractJSONStrings(html.UnescapeString(match[1])) + if len(values) > 0 { + parts = append(parts, strings.Join(values, " ")) + } + } + + visibleHTML := inhireScriptStylePattern.ReplaceAllString(rawHTML, " ") + visibleText := inhireTagPattern.ReplaceAllString(visibleHTML, " ") + parts = append(parts, visibleText) + + return inhireNormalizeDetailText(strings.Join(parts, " ")) +} + +func inhireExtractJSONStrings(rawJSON string) []string { + var payload any + decoder := json.NewDecoder(strings.NewReader(rawJSON)) + decoder.UseNumber() + if err := decoder.Decode(&payload); err != nil { + return nil + } + + values := make([]string, 0, 64) + inhireCollectJSONStrings(payload, &values) + return values +} + +func inhireCollectJSONStrings(value any, out *[]string) { + if len(*out) >= 300 { + return + } + + switch typed := value.(type) { + case string: + text := inhireNormalizeDetailText(typed) + if len([]rune(text)) >= 3 { + *out = append(*out, text) + } + case []any: + for _, item := range typed { + inhireCollectJSONStrings(item, out) + } + case map[string]any: + for _, item := range typed { + inhireCollectJSONStrings(item, out) + } + } +} + +func inhireNormalizeDetailText(text string) string { + text = html.UnescapeString(text) + text = strings.ReplaceAll(text, `\u003c`, "<") + text = strings.ReplaceAll(text, `\u003e`, ">") + text = strings.ReplaceAll(text, `\u0026`, "&") + text = inhireTagPattern.ReplaceAllString(text, " ") + text = inhireWhitespacePattern.ReplaceAllString(text, " ") + text = strings.TrimSpace(text) + + runes := []rune(text) + if len(runes) > inhireMaxDescriptionRunes { + text = string(runes[:inhireMaxDescriptionRunes]) + } + + return text +} + +func inhireSlugify(value string) string { + value = strings.ToLower(html.UnescapeString(value)) + value = strings.ReplaceAll(value, "&", " and ") + t := transform.Chain(norm.NFD, transform.RemoveFunc(func(r rune) bool { + return unicode.Is(unicode.Mn, r) + }), norm.NFC) + value, _, _ = transform.String(t, value) + + var b strings.Builder + lastDash := false + for _, r := range value { + if (r >= 'a' && r <= 'z') || (r >= '0' && r <= '9') { + b.WriteRune(r) + lastDash = false + continue + } + if !lastDash { + b.WriteRune('-') + lastDash = true + } + } + + return strings.Trim(b.String(), "-") +} diff --git a/scraper-go/internal/adapters/inhire/adapter_test.go b/scraper-go/internal/adapters/inhire/adapter_test.go new file mode 100644 index 0000000..448e447 --- /dev/null +++ b/scraper-go/internal/adapters/inhire/adapter_test.go @@ -0,0 +1,270 @@ +package inhire + +import ( + "context" + "net/http" + "os" + "path/filepath" + "strings" + "sync" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestInHireSearchBatchFetchesTenantsAndMapsJobs(t *testing.T) { + tenantsFile := writeInHireTenantsFile(t, `[ + {"slug":"acme","tenantName":"Acme"}, + {"slug":"brq","tenantName":"BRQ"} + ]`) + t.Setenv("INHIRE_TENANTS_FILE", tenantsFile) + t.Setenv("INHIRE_ENRICH_DETAILS", "false") + + var ( + mu sync.Mutex + tenants []string + ) + + adapter := NewInHire() + adapter.apiURL = "https://example.test/inhire" + adapter.concurrency = 1 + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.String() != "https://example.test/inhire" { + t.Fatalf("unexpected endpoint: %s", req.URL.String()) + } + if req.Header.Get("X-Inhire-Client") != "web-inhire" { + t.Fatalf("expected X-Inhire-Client header") + } + + tenant := req.Header.Get("X-Tenant") + mu.Lock() + tenants = append(tenants, tenant) + mu.Unlock() + + switch tenant { + case "acme": + return testResponse(`{ + "tenantName": "Acme Tecnologia", + "jobsPage": [ + { + "displayName": "Pessoa Desenvolvedora Java Sênior", + "jobId": "job-1", + "status": "published", + "workplaceType": "Hybrid", + "location": "São Paulo, SP, BR" + }, + { + "displayName": "Pessoa Analista Financeira", + "jobId": "job-2", + "status": "closed", + "workplaceType": "Remote", + "location": "BR" + } + ] + }`), nil + case "brq": + return testResponse(`{ + "tenantName": "BRQ", + "jobsPage": [ + { + "displayName": "Engenheiro de Dados", + "jobId": "job-3", + "status": "published", + "workplaceType": "Remote", + "location": "BR" + } + ] + }`), nil + default: + t.Fatalf("unexpected tenant: %s", tenant) + return testResponse(`{"jobsPage":[]}`), nil + } + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"java developer", "dados"}, domain.ScrapeRequest{}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 2 { + t.Fatalf("expected two published jobs, got %d", len(jobs)) + } + if tenantsJoined := strings.Join(tenants, ","); tenantsJoined != "acme,brq" { + t.Fatalf("unexpected tenants order: %s", tenantsJoined) + } + + first := jobs[0] + if first.Company != "Acme Tecnologia" { + t.Fatalf("unexpected company: %q", first.Company) + } + if first.Modality != "Híbrido" { + t.Fatalf("expected hybrid modality, got %q", first.Modality) + } + if first.URL != "https://acme.inhire.app/vagas/job-1/pessoa-desenvolvedora-java-senior" { + t.Fatalf("unexpected URL: %q", first.URL) + } + if first.Source != "InHire" || strings.Join(first.Sources, ",") != "InHire" { + t.Fatalf("unexpected source metadata: %q %#v", first.Source, first.Sources) + } + + second := jobs[1] + if second.Modality != "Remoto" { + t.Fatalf("expected remote modality, got %q", second.Modality) + } + if second.Keyword != "dados" { + t.Fatalf("expected matched keyword, got %q", second.Keyword) + } +} + +func TestInHireSearchBatchRespectsRemoteOnly(t *testing.T) { + tenantsFile := writeInHireTenantsFile(t, `[{"slug":"acme","tenantName":"Acme"}]`) + t.Setenv("INHIRE_TENANTS_FILE", tenantsFile) + t.Setenv("INHIRE_ENRICH_DETAILS", "false") + + adapter := NewInHire() + adapter.apiURL = "https://example.test/inhire" + adapter.concurrency = 1 + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + return testResponse(`{ + "tenantName": "Acme", + "jobsPage": [ + {"displayName":"Backend Java","jobId":"remote","status":"published","workplaceType":"Remote","location":"BR"}, + {"displayName":"Backend Java","jobId":"hybrid","status":"published","workplaceType":"Hybrid","location":"São Paulo, SP, BR"} + ] + }`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"java"}, domain.ScrapeRequest{RemoteOnly: true}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected only remote job, got %d", len(jobs)) + } + if jobs[0].ID != "inhire:acme:remote" { + t.Fatalf("unexpected job ID: %q", jobs[0].ID) + } +} + +func TestInHireLoadTenantsHandlesMissingFile(t *testing.T) { + t.Setenv("INHIRE_TENANTS_FILE", filepath.Join(t.TempDir(), "missing.json")) + t.Setenv("INHIRE_ENRICH_DETAILS", "false") + + adapter := NewInHire() + _, err := adapter.SearchBatch(context.Background(), []string{"java"}, domain.ScrapeRequest{}) + if err == nil { + t.Fatal("expected missing tenants file error") + } + if !strings.Contains(err.Error(), "inhire: leitura") { + t.Fatalf("unexpected error: %v", err) + } +} + +func TestInHireSearchBatchEnrichesAmbiguousJobsWithDetails(t *testing.T) { + tenantsFile := writeInHireTenantsFile(t, `[{"slug":"acme","tenantName":"Acme"}]`) + t.Setenv("INHIRE_TENANTS_FILE", tenantsFile) + t.Setenv("INHIRE_ENRICH_DETAILS", "true") + t.Setenv("INHIRE_DETAILS_MODE", "ambiguous") + t.Setenv("INHIRE_DETAILS_CONCURRENCY", "1") + t.Setenv("INHIRE_DETAILS_TIMEOUT_MS", "1000") + + var detailCalls int + + adapter := NewInHire() + adapter.apiURL = "https://example.test/inhire" + adapter.concurrency = 1 + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + switch req.URL.String() { + case "https://example.test/inhire": + return testResponse(`{ + "tenantName": "Acme", + "jobsPage": [ + { + "displayName": "Analista de Sistemas Pleno", + "jobId": "job-1", + "status": "published", + "workplaceType": "Hybrid", + "location": "São Paulo, SP, BR" + }, + { + "displayName": "Backend Java Developer", + "jobId": "job-2", + "status": "published", + "workplaceType": "Remote", + "location": "BR" + } + ] + }`), nil + case "https://acme.inhire.app/vagas/job-1/analista-de-sistemas-pleno": + detailCalls++ + return testResponse(` + + + + +
Responsabilidades de engenharia de software.
+ `), nil + default: + t.Fatalf("unexpected endpoint: %s", req.URL.String()) + return testResponse(`{}`), nil + } + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"java developer", "spring boot developer"}, domain.ScrapeRequest{}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 2 { + t.Fatalf("expected two jobs, got %d", len(jobs)) + } + if detailCalls != 1 { + t.Fatalf("expected one detail call for ambiguous job, got %d", detailCalls) + } + if !strings.Contains(jobs[0].Description, "Spring Boot") { + t.Fatalf("expected enriched description, got %q", jobs[0].Description) + } + if strings.Join(jobs[0].Keywords, ",") != "java developer,spring boot developer" { + t.Fatalf("expected keywords from detail, got %#v", jobs[0].Keywords) + } +} + +func TestInHireExtractDetailTextReadsJSONAndVisibleHTML(t *testing.T) { + text := inhireExtractDetailText(` + + +

Vaga

APIs e testes automatizados

`) + + if !strings.Contains(text, "React, TypeScript e Node.js") { + t.Fatalf("expected JSON text, got %q", text) + } + if !strings.Contains(text, "APIs e testes automatizados") { + t.Fatalf("expected visible HTML text, got %q", text) + } + if strings.Contains(text, "display:none") { + t.Fatalf("expected style content to be removed, got %q", text) + } +} + +func TestInHireMatchingKeywordsInText(t *testing.T) { + matched := inhireMatchingKeywordsInText( + "Developer backend com Java, Spring Boot, PostgreSQL e APIs.", + []string{"java developer", "spring boot developer"}, + ) + + if strings.Join(matched, ",") != "java developer,spring boot developer" { + t.Fatalf("unexpected matched keywords: %#v", matched) + } +} + +func writeInHireTenantsFile(t *testing.T, payload string) string { + t.Helper() + + path := filepath.Join(t.TempDir(), "tenants.json") + if err := os.WriteFile(path, []byte(payload), 0o600); err != nil { + t.Fatalf("write tenants file: %v", err) + } + return path +} diff --git a/scraper-go/internal/adapters/inhire/test_helpers_test.go b/scraper-go/internal/adapters/inhire/test_helpers_test.go new file mode 100644 index 0000000..314467a --- /dev/null +++ b/scraper-go/internal/adapters/inhire/test_helpers_test.go @@ -0,0 +1,15 @@ +package inhire + +import ( + "net/http" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" +) + +func testHTTPClient(fn testutil.RoundTripFunc) *http.Client { + return testutil.HTTPClient(fn) +} + +func testResponse(body string) *http.Response { + return testutil.Response(body) +} diff --git a/scraper-go/internal/adapters/jooble.go b/scraper-go/internal/adapters/jooble.go deleted file mode 100644 index 0d0c575..0000000 --- a/scraper-go/internal/adapters/jooble.go +++ /dev/null @@ -1,200 +0,0 @@ -package adapters - -import ( - "bytes" - "context" - "encoding/json" - "fmt" - "net/http" - "time" - - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" - "github.com/redis/go-redis/v9" -) - -const ( - joobleQuotaKey = "jooble:quota:daily" // contador no Valkey - joobleQuotaLimit = 300 // chamadas máximas por dia - joobleSlotSize = 37 // keywords por execução - joobleRotationKey = "jooble:rotation:offset" // offset rotativo no Valkey -) - -type JoobleAdapter struct { - apiKey string - client *http.Client - rdb *redis.Client -} - -func NewJooble(apiKey string, rdb *redis.Client) *JoobleAdapter { - return &JoobleAdapter{ - apiKey: apiKey, - client: &http.Client{Timeout: 15 * time.Second}, - rdb: rdb, - } -} - -func (a *JoobleAdapter) SourceName() string { return "Jooble" } - -// quotaRemaining retorna quantas chamadas ainda cabem hoje. -// Retorna 0 se o Valkey falhar (fail-safe: não estoura a cota). -func (a *JoobleAdapter) quotaRemaining(ctx context.Context) int { - used, err := a.rdb.Get(ctx, joobleQuotaKey).Int() - if err == redis.Nil { - return joobleQuotaLimit // chave não existe ainda → dia novo - } - if err != nil { - return 0 // Valkey indisponível → para silenciosamente - } - remaining := joobleQuotaLimit - used - if remaining < 0 { - return 0 - } - return remaining -} - -// incrementQuota registra 1 chamada usada. -// Na primeira chamada do dia, define TTL até meia-noite. -func (a *JoobleAdapter) incrementQuota(ctx context.Context) { - pipe := a.rdb.Pipeline() - pipe.Incr(ctx, joobleQuotaKey) - - // TTL dinâmico: segundos restantes até meia-noite - now := time.Now() - midnight := time.Date(now.Year(), now.Month(), now.Day()+1, 0, 0, 0, 0, now.Location()) - ttl := time.Until(midnight) - pipe.Expire(ctx, joobleQuotaKey, ttl) - - pipe.Exec(ctx) //nolint:errcheck — falha aqui não deve parar o scraper -} - -// nextSlot retorna o subconjunto rotacionado de keywords para essa execução. -func (a *JoobleAdapter) nextSlot(ctx context.Context, keywords []string, slotSize int) []string { - if len(keywords) == 0 { - return nil - } - - // Lê offset atual; se não existir começa do 0 - offset, err := a.rdb.Get(ctx, joobleRotationKey).Int() - if err != nil { - offset = 0 - } - - // Salva próximo offset para a execução seguinte - nextOffset := (offset + slotSize) % len(keywords) - a.rdb.Set(ctx, joobleRotationKey, nextOffset, 0) //nolint:errcheck - - // Monta o slice rotacionado (wrap-around circular) - result := make([]string, 0, slotSize) - for i := 0; i < slotSize && i < len(keywords); i++ { - result = append(result, keywords[(offset+i)%len(keywords)]) - } - return result -} - -// SearchBatch é o ponto de entrada — recebe todas as keywords mas roda -// apenas o slot rotacionado, respeitando a cota diária. -// Retorna os jobs encontrados e para silenciosamente se a cota acabar. -func (a *JoobleAdapter) SearchBatch(ctx context.Context, allKeywords []string, req models.ScrapeRequest) ([]models.Job, error) { - remaining := a.quotaRemaining(ctx) - if remaining <= 0 { - return nil, nil // cota esgotada → para silenciosamente - } - - slotSize := joobleSlotSize - if slotSize > remaining { - slotSize = remaining // não ultrapassa o que sobrou - } - - slot := a.nextSlot(ctx, allKeywords, slotSize) - if len(slot) == 0 { - return nil, nil - } - - var allJobs []models.Job - for _, keyword := range slot { - jobs, err := a.search(ctx, keyword, req) - if err != nil { - // 429 ou erro de rede → para silenciosamente, não aborta os outros adapters - break - } - a.incrementQuota(ctx) - allJobs = append(allJobs, jobs...) - } - - return allJobs, nil -} - -// search faz 1 chamada para 1 keyword (lógica original preservada). -func (a *JoobleAdapter) search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) { - endpoint := "https://br.jooble.org/api/" + a.apiKey - - payload := map[string]string{ - "keywords": keyword, - "location": req.SearchLocation, - } - if payload["location"] == "" { - payload["location"] = "Brasil" - } - - body, _ := json.Marshal(payload) - - httpReq, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, bytes.NewBuffer(body)) - if err != nil { - return nil, err - } - httpReq.Header.Set("Content-Type", "application/json") - - resp, err := a.client.Do(httpReq) - if err != nil { - return nil, err - } - defer resp.Body.Close() - - if resp.StatusCode == http.StatusTooManyRequests { - return nil, fmt.Errorf("jooble: 429") - } - if resp.StatusCode != http.StatusOK { - return nil, fmt.Errorf("jooble: status %d", resp.StatusCode) - } - - var joobleRes struct { - Jobs []struct { - Title string `json:"title"` - Location string `json:"location"` - Snippet string `json:"snippet"` - Source string `json:"source"` - Type string `json:"type"` - Link string `json:"link"` - Company string `json:"company"` - Updated string `json:"updated"` - Salary string `json:"salary"` - ID int64 `json:"id"` - } `json:"jobs"` - } - - if err := json.NewDecoder(resp.Body).Decode(&joobleRes); err != nil { - return nil, err - } - - jobs := make([]models.Job, 0, len(joobleRes.Jobs)) - for _, j := range joobleRes.Jobs { - jobs = append(jobs, models.Job{ - ID: j.Link, - Title: j.Title, - Company: j.Company, - Location: j.Location, - URL: j.Link, - Salary: j.Salary, - Source: "Jooble", - Sources: []string{"Jooble"}, - Keyword: keyword, - Keywords: []string{keyword}, - }) - } - - return jobs, nil -} - -func (a *JoobleAdapter) Search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) { - return a.SearchBatch(ctx, []string{keyword}, req) -} diff --git a/scraper-go/internal/adapters/jooble/adapter.go b/scraper-go/internal/adapters/jooble/adapter.go new file mode 100644 index 0000000..6527e92 --- /dev/null +++ b/scraper-go/internal/adapters/jooble/adapter.go @@ -0,0 +1,338 @@ +package jooble + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "log/slog" + "net/http" + "os" + "strconv" + "strings" + "sync" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/redis/go-redis/v9" +) + +const ( + joobleQuotaKey = "jooble:quota:daily" // contador no Valkey + joobleQuotaLimit = 300 // chamadas máximas por dia + joobleSlotSize = 37 // keywords por execução + joobleRotationKey = "jooble:rotation:offset" // offset rotativo no Valkey + joobleDefaultPagesPerKeyword = 3 + joobleDefaultResultsPerPage = 20 + joobleMaxResultsPerPage = 50 +) + +type JoobleAdapter struct { + apiKey string + client *http.Client + rdb *redis.Client + mu sync.Mutex +} + +func NewJooble(apiKey string, rdb *redis.Client) *JoobleAdapter { + return &JoobleAdapter{ + apiKey: apiKey, + client: &http.Client{Timeout: 15 * time.Second}, + rdb: rdb, + } +} + +func (a *JoobleAdapter) SourceName() string { return "Jooble" } + +// quotaRemaining retorna quantas chamadas ainda cabem hoje. +// Se Valkey estiver indisponível, segue com o limite local para não zerar o adapter silenciosamente. +func (a *JoobleAdapter) quotaRemaining(ctx context.Context) int { + if a.rdb == nil { + slog.Warn("jooble: Valkey indisponível, seguindo sem contador distribuído de cota") + return joobleQuotaLimit + } + + used, err := a.rdb.Get(ctx, joobleQuotaKey).Int() + if err == redis.Nil { + return joobleQuotaLimit // chave não existe ainda → dia novo + } + if err != nil { + slog.Warn("jooble: falha ao ler cota no Valkey, seguindo sem bloquear adapter", "error", err) + return joobleQuotaLimit + } + remaining := joobleQuotaLimit - used + if remaining < 0 { + return 0 + } + return remaining +} + +// incrementQuota registra 1 chamada usada. +// Na primeira chamada do dia, define TTL até meia-noite. +func (a *JoobleAdapter) incrementQuota(ctx context.Context) { + if a.rdb == nil { + return + } + + pipe := a.rdb.Pipeline() + pipe.Incr(ctx, joobleQuotaKey) + + // TTL dinâmico: segundos restantes até meia-noite + now := time.Now() + midnight := time.Date(now.Year(), now.Month(), now.Day()+1, 0, 0, 0, 0, now.Location()) + ttl := time.Until(midnight) + pipe.Expire(ctx, joobleQuotaKey, ttl) + + pipe.Exec(ctx) //nolint:errcheck — falha aqui não deve parar o scraper +} + +// nextSlot retorna o subconjunto rotacionado de keywords para essa execução. +func (a *JoobleAdapter) nextSlot(ctx context.Context, keywords []string, slotSize int) []string { + if len(keywords) == 0 { + return nil + } + if a.rdb == nil { + if slotSize > len(keywords) { + slotSize = len(keywords) + } + return append([]string(nil), keywords[:slotSize]...) + } + + // Lê offset atual; se não existir começa do 0 + offset, err := a.rdb.Get(ctx, joobleRotationKey).Int() + if err != nil { + offset = 0 + } + + // Salva próximo offset para a execução seguinte + nextOffset := (offset + slotSize) % len(keywords) + a.rdb.Set(ctx, joobleRotationKey, nextOffset, 0) //nolint:errcheck + + // Monta o slice rotacionado (wrap-around circular) + result := make([]string, 0, slotSize) + for i := 0; i < slotSize && i < len(keywords); i++ { + result = append(result, keywords[(offset+i)%len(keywords)]) + } + return result +} + +// SearchBatch é o ponto de entrada — recebe todas as keywords mas roda +// apenas o slot rotacionado, respeitando a cota diária. +// Retorna os jobs encontrados e para silenciosamente se a cota acabar. +func (a *JoobleAdapter) SearchBatch(ctx context.Context, allKeywords []string, req domain.ScrapeRequest) ([]domain.Job, error) { + a.mu.Lock() + defer a.mu.Unlock() + + remaining := a.quotaRemaining(ctx) + if remaining <= 0 { + slog.Warn("jooble: cota diária esgotada, adapter não fará novas chamadas", "limit", joobleQuotaLimit) + return nil, nil // cota esgotada → para silenciosamente + } + + slotSize := joobleSlotSize + if slotSize > remaining { + slotSize = remaining // não ultrapassa o que sobrou + } + + slot := a.nextSlot(ctx, allKeywords, slotSize) + if len(slot) == 0 { + return nil, nil + } + + var allJobs []domain.Job + for _, keyword := range slot { + jobs, err := a.search(ctx, keyword, req) + if err != nil { + if len(allJobs) == 0 { + return nil, err + } + slog.Warn("jooble: busca interrompida após resultados parciais", "keyword", keyword, "error", err) + break + } + allJobs = append(allJobs, jobs...) + } + + return allJobs, nil +} + +// search faz 1 chamada para 1 keyword (lógica original preservada). +func (a *JoobleAdapter) search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + maxPages := req.MaxPagesPerKeyword + if maxPages <= 0 { + maxPages = joobleDefaultPagesPerKeyword + } + + resultsPerPage := req.ResultsPerPage + if resultsPerPage <= 0 { + resultsPerPage = joobleDefaultResultsPerPage + } + if resultsPerPage > joobleMaxResultsPerPage { + resultsPerPage = joobleMaxResultsPerPage + } + + pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond + if pageTimeout <= 0 { + pageTimeout = 15 * time.Second + } + + var allJobs []domain.Job + seenPages := make(map[string]struct{}) + + for page := 1; page <= maxPages; page++ { + pageCtx, cancel := context.WithTimeout(ctx, pageTimeout) + jobs, total, err := a.fetchPage(pageCtx, keyword, req, page, resultsPerPage) + cancel() + + if err != nil { + if len(allJobs) > 0 { + slog.Warn("jooble: página falhou após resultados parciais", "keyword", keyword, "page", page, "error", err) + break + } + return nil, err + } + + a.incrementQuota(ctx) + + if len(jobs) == 0 { + break + } + if adapterutil.RepeatedJobPage(seenPages, jobs) { + break + } + + allJobs = append(allJobs, jobs...) + + if total > 0 && len(allJobs) >= total { + break + } + if len(jobs) < resultsPerPage { + break + } + } + + return allJobs, nil +} + +func (a *JoobleAdapter) fetchPage( + ctx context.Context, + keyword string, + req domain.ScrapeRequest, + page int, + resultsPerPage int, +) ([]domain.Job, int, error) { + endpoint := strings.TrimRight(os.Getenv("JOOBLE_API_BASE"), "/") + if endpoint == "" { + endpoint = "https://br.jooble.org/api" + } + endpoint = endpoint + "/" + a.apiKey + + location := strings.TrimSpace(req.SearchLocation) + if location == "" { + location = "Brasil" + } + + payload := map[string]any{ + "keywords": keyword, + "location": location, + "page": strconv.Itoa(page), + "ResultOnPage": resultsPerPage, + "SearchMode": 0, + "companysearch": "false", + } + if location != "" { + payload["radius"] = "80" + } + + body, _ := json.Marshal(payload) + + httpReq, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, bytes.NewBuffer(body)) + if err != nil { + return nil, 0, err + } + httpReq.Header.Set("Content-Type", "application/json") + httpReq.Header.Set("Accept", "application/json") + httpReq.Header.Set("User-Agent", "JobsScraper/1.0") + + resp, err := a.client.Do(httpReq) + if err != nil { + return nil, 0, err + } + defer resp.Body.Close() + + if resp.StatusCode == http.StatusTooManyRequests { + return nil, 0, fmt.Errorf("jooble: 429") + } + if resp.StatusCode == http.StatusForbidden { + return nil, 0, fmt.Errorf("jooble: 403 acesso negado, verifique JOOBLE_API_KEY") + } + if resp.StatusCode != http.StatusOK { + return nil, 0, fmt.Errorf("jooble: status %d", resp.StatusCode) + } + + var joobleRes joobleSearchResponse + if err := json.NewDecoder(resp.Body).Decode(&joobleRes); err != nil { + return nil, 0, err + } + + rawJobs := joobleRes.Jobs + if len(rawJobs) == 0 { + rawJobs = joobleRes.Results + } + + jobs := make([]domain.Job, 0, len(rawJobs)) + for _, j := range rawJobs { + id := strings.TrimSpace(strconv.FormatInt(j.ID, 10)) + if j.ID == 0 { + id = strings.TrimSpace(j.Link) + } + if id == "" { + id = strings.Join([]string{ + strings.TrimSpace(j.Title), + strings.TrimSpace(j.Company), + strings.TrimSpace(j.Location), + }, "|") + } + + jobs = append(jobs, domain.Job{ + ID: id, + Title: strings.TrimSpace(j.Title), + Company: strings.TrimSpace(j.Company), + Location: strings.TrimSpace(j.Location), + URL: strings.TrimSpace(j.Link), + Salary: strings.TrimSpace(j.Salary), + Modality: strings.TrimSpace(j.Type), + Description: strings.TrimSpace(j.Snippet), + PostedAt: strings.TrimSpace(j.Updated), + Source: "Jooble", + Sources: []string{"Jooble"}, + Keyword: keyword, + Keywords: []string{keyword}, + }) + } + + return jobs, joobleRes.TotalCount, nil +} + +func (a *JoobleAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + return a.SearchBatch(ctx, []string{keyword}, req) +} + +type joobleSearchResponse struct { + TotalCount int `json:"totalCount"` + Jobs []joobleJob `json:"jobs"` + Results []joobleJob `json:"results"` +} + +type joobleJob struct { + Title string `json:"title"` + Location string `json:"location"` + Snippet string `json:"snippet"` + Source string `json:"source"` + Type string `json:"type"` + Link string `json:"link"` + Company string `json:"company"` + Updated string `json:"updated"` + Salary string `json:"salary"` + ID int64 `json:"id"` +} diff --git a/scraper-go/internal/adapters/jooble/pagination_test.go b/scraper-go/internal/adapters/jooble/pagination_test.go new file mode 100644 index 0000000..9473375 --- /dev/null +++ b/scraper-go/internal/adapters/jooble/pagination_test.go @@ -0,0 +1,107 @@ +package jooble + +import ( + "context" + "encoding/json" + "net/http" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestJoobleSearchUsesRestPayloadAndPagination(t *testing.T) { + calls := 0 + adapter := NewJooble("test-key", nil) + adapter.client = testutil.HTTPClient(func(req *http.Request) (*http.Response, error) { + calls++ + + if req.Method != http.MethodPost { + t.Fatalf("expected POST request, got %s", req.Method) + } + if req.URL.Host != "br.jooble.org" || req.URL.Path != "/api/test-key" { + t.Fatalf("unexpected Jooble endpoint: %s", req.URL.String()) + } + + var payload map[string]any + if err := json.NewDecoder(req.Body).Decode(&payload); err != nil { + t.Fatalf("invalid request payload: %v", err) + } + + page, ok := payload["page"].(string) + if !ok { + t.Fatalf("expected page as string, got %#v", payload["page"]) + } + if payload["keywords"] != "go developer" { + t.Fatalf("unexpected keywords payload: %#v", payload["keywords"]) + } + if payload["location"] != "Brasil" { + t.Fatalf("unexpected location payload: %#v", payload["location"]) + } + if payload["ResultOnPage"] != float64(2) { + t.Fatalf("unexpected ResultOnPage payload: %#v", payload["ResultOnPage"]) + } + + if page == "1" { + return testutil.Response(`{ + "totalCount":3, + "jobs":[ + { + "title":"Go Developer", + "company":"Acme", + "location":"Brasil", + "snippet":"Backend with Go", + "type":"Full-time", + "link":"https://example.com/jooble/123", + "salary":"100", + "updated":"2026-07-29T00:00:00Z", + "id":123 + }, + { + "title":"Platform Engineer", + "company":"Beta", + "location":"Remote", + "snippet":"Golang platform", + "link":"https://example.com/jooble/456", + "id":456 + } + ] + }`), nil + } + + return testutil.Response(`{ + "totalCount":3, + "jobs":[ + { + "title":"Software Engineer", + "company":"Gamma", + "location":"Remote", + "snippet":"Go services", + "link":"https://example.com/jooble/789", + "id":789 + } + ] + }`), nil + }) + + jobs, err := adapter.Search(context.Background(), "go developer", domain.ScrapeRequest{ + MaxPagesPerKeyword: 2, + ResultsPerPage: 2, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 3 { + t.Fatalf("expected 3 jobs across pages, got %d", len(jobs)) + } + if calls != 2 { + t.Fatalf("expected 2 paginated calls, got %d", calls) + } + if jobs[0].ID != "123" { + t.Fatalf("expected id from Jooble payload, got %q", jobs[0].ID) + } + if jobs[0].Description != "Backend with Go" { + t.Fatalf("expected snippet as description, got %q", jobs[0].Description) + } +} diff --git a/scraper-go/internal/adapters/lever.go b/scraper-go/internal/adapters/lever.go deleted file mode 100644 index 6251e33..0000000 --- a/scraper-go/internal/adapters/lever.go +++ /dev/null @@ -1,157 +0,0 @@ -package adapters - -import ( - "context" - "encoding/json" - "fmt" - "net/http" - "os" - "strings" - "time" - - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" -) - -type LeverAdapter struct { - client *http.Client - companySlug string - companyName string -} - -type leverCompany struct { - Slug string `json:"slug"` - Name string `json:"name"` -} - -type leverPosting struct { - ID string `json:"id"` - Text string `json:"text"` - HostedURL string `json:"hostedUrl"` - CreatedAt int64 `json:"createdAt"` - - Categories struct { - Team string `json:"team"` - Department string `json:"department"` - Location string `json:"location"` - Commitment string `json:"commitment"` - Level string `json:"level"` - } `json:"categories"` - - Description string `json:"description"` - State string `json:"state"` -} - -func NewLever(companySlug, companyName string) *LeverAdapter { - return &LeverAdapter{ - client: &http.Client{Timeout: 60 * time.Second}, - companySlug: companySlug, - companyName: companyName, - } -} - -func FetchLeverSlugs(_ context.Context) ([]leverCompany, error) { - path := os.Getenv("LEVER_COMPANIES_FILE") - if path == "" { - path = "./internal/interfaces/leverCompanies.json" - } - - data, err := os.ReadFile(path) - if err != nil { - return nil, fmt.Errorf("lever: leitura do arquivo '%s': %w", path, err) - } - - var companies []leverCompany - if err := json.Unmarshal(data, &companies); err != nil { - return nil, fmt.Errorf("lever: parse do arquivo '%s': %w", path, err) - } - - if len(companies) == 0 { - return nil, fmt.Errorf("lever: nenhuma empresa encontrada em '%s'", path) - } - - return companies, nil -} - -func (a *LeverAdapter) SourceName() string { - return fmt.Sprintf("Lever:%s", a.companyName) -} - -func (a *LeverAdapter) Search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) { - pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond - if pageTimeout <= 0 { - pageTimeout = 15 * time.Second - } - - endpoint := fmt.Sprintf( - "https://api.lever.co/v0/postings/%s?mode=json", - a.companySlug, - ) - - reqCtx, cancel := context.WithTimeout(ctx, pageTimeout) - defer cancel() - - httpReq, err := http.NewRequestWithContext(reqCtx, http.MethodGet, endpoint, nil) - if err != nil { - return nil, fmt.Errorf("lever: build request: %w", err) - } - - httpReq.Header.Set("User-Agent", "JobsScraper/1.0") - httpReq.Header.Set("Accept", "application/json") - - resp, err := a.client.Do(httpReq) - if err != nil { - return nil, fmt.Errorf("lever: http do: %w", err) - } - defer resp.Body.Close() - - switch resp.StatusCode { - case http.StatusOK: - // ok - case http.StatusTooManyRequests: - return nil, fmt.Errorf("lever: rate limit atingido (429)") - case http.StatusNotFound: - return nil, fmt.Errorf("lever: empresa '%s' não encontrada (404)", a.companySlug) - default: - return nil, fmt.Errorf("lever: status inesperado %d", resp.StatusCode) - } - - var raw []leverPosting - if err := json.NewDecoder(resp.Body).Decode(&raw); err != nil { - return nil, fmt.Errorf("lever: decode json: %w", err) - } - - kwLower := strings.ToLower(keyword) - - var jobs []models.Job - for _, j := range raw { - if keyword != "" && !strings.Contains(strings.ToLower(j.Text), kwLower) { - continue - } - - dataPublicacao := "" - if j.CreatedAt != 0 { - dataPublicacao = time.UnixMilli(j.CreatedAt).UTC().Format(time.RFC3339) - } - - local := strings.TrimSpace(j.Categories.Location) - if local == "" { - local = strings.TrimSpace(j.Categories.Department) - } - - jobs = append(jobs, models.Job{ - ID: strings.TrimSpace(j.HostedURL), - Title: strings.TrimSpace(j.Text), - Company: a.companyName, - Location: local, - URL: strings.TrimSpace(j.HostedURL), - Modality: strings.TrimSpace(j.Categories.Commitment), - PostedAt: dataPublicacao, - Source: "Lever", - Sources: []string{"Lever"}, - Keyword: keyword, - Keywords: []string{keyword}, - }) - } - - return jobs, nil -} diff --git a/scraper-go/internal/adapters/lever/adapter.go b/scraper-go/internal/adapters/lever/adapter.go new file mode 100644 index 0000000..b5916eb --- /dev/null +++ b/scraper-go/internal/adapters/lever/adapter.go @@ -0,0 +1,454 @@ +package lever + +import ( + "context" + "encoding/json" + "fmt" + "html" + "net/http" + "os" + "regexp" + "strings" + "sync" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" +) + +type LeverAdapter struct { + client *http.Client + companySlug string + companyName string + apiHost string + apiURL string + + mu sync.Mutex + cachedJobs []leverPosting + cacheErr error + loaded bool +} + +type leverCompany struct { + Slug string `json:"slug"` + Site string `json:"site"` + Name string `json:"name"` + Company string `json:"companyName"` + Region string `json:"region"` + APIURL string `json:"apiUrl"` + JobsURL string `json:"jobsUrl"` +} + +type leverPosting struct { + ID string `json:"id"` + Text string `json:"text"` + HostedURL string `json:"hostedUrl"` + ApplyURL string `json:"applyUrl"` + CreatedAt int64 `json:"createdAt"` + Country string `json:"country"` + + Categories struct { + Team string `json:"team"` + Department string `json:"department"` + Location string `json:"location"` + Commitment string `json:"commitment"` + Level string `json:"level"` + AllLocations []string `json:"allLocations"` + } `json:"categories"` + + OpeningPlain string `json:"openingPlain"` + Description string `json:"description"` + DescriptionPlain string `json:"descriptionPlain"` + DescriptionBodyPlain string `json:"descriptionBodyPlain"` + Lists []leverList `json:"lists"` + AdditionalPlain string `json:"additionalPlain"` + SalaryDescription string `json:"salaryDescriptionPlain"` + SalaryRange *struct { + Min float64 `json:"min"` + Max float64 `json:"max"` + Currency string `json:"currency"` + Interval string `json:"interval"` + } `json:"salaryRange"` + WorkplaceType string `json:"workplaceType"` + State string `json:"state"` +} + +type leverList struct { + Text string `json:"text"` + Content string `json:"content"` +} + +func NewLever(companySlug, companyName string) *LeverAdapter { + return &LeverAdapter{ + client: &http.Client{Timeout: 60 * time.Second}, + companySlug: companySlug, + companyName: companyName, + apiHost: "api.lever.co", + } +} + +func NewLeverWithRegion(companySlug, companyName, region string) *LeverAdapter { + adapter := NewLever(companySlug, companyName) + if strings.EqualFold(strings.TrimSpace(region), "eu") { + adapter.apiHost = "api.eu.lever.co" + } + return adapter +} + +func NewLeverWithEndpoint(companySlug, companyName, region, apiURL string) *LeverAdapter { + adapter := NewLeverWithRegion(companySlug, companyName, region) + adapter.apiURL = strings.TrimSpace(apiURL) + return adapter +} + +func FetchLeverSlugs(_ context.Context) ([]leverCompany, error) { + path := os.Getenv("LEVER_COMPANIES_FILE") + if path == "" { + path = "./internal/interfaces/leverCompanies.json" + } + + data, err := os.ReadFile(path) + if err != nil { + return nil, fmt.Errorf("lever: leitura do arquivo '%s': %w", path, err) + } + + var companies []leverCompany + if err := json.Unmarshal(data, &companies); err != nil { + return nil, fmt.Errorf("lever: parse do arquivo '%s': %w", path, err) + } + + if len(companies) == 0 { + return nil, fmt.Errorf("lever: nenhuma empresa encontrada em '%s'", path) + } + + return companies, nil +} + +func (a *LeverAdapter) SourceName() string { + return fmt.Sprintf("Lever:%s", a.companyName) +} + +func (a *LeverAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + return a.SearchBatch(ctx, []string{keyword}, req) +} + +func (a *LeverAdapter) SearchBatch(ctx context.Context, keywords []string, req domain.ScrapeRequest) ([]domain.Job, error) { + raw, err := a.fetchPostings(ctx, req) + if err != nil { + return nil, err + } + + includeAll := leverIncludeAllJobs() + var jobs []domain.Job + for _, j := range raw { + if !leverMatchesFilters(j, req) { + continue + } + + matchedKeywords := leverMatchingKeywords(j, keywords) + if len(matchedKeywords) == 0 && !includeAll { + continue + } + + dataPublicacao := "" + if j.CreatedAt != 0 { + dataPublicacao = time.UnixMilli(j.CreatedAt).UTC().Format(time.RFC3339) + } + + keyword := "" + if len(matchedKeywords) > 0 { + keyword = matchedKeywords[0] + } + + jobs = append(jobs, domain.Job{ + ID: leverJobID(j), + Title: strings.TrimSpace(j.Text), + Company: a.companyName, + Location: leverLocation(j), + URL: leverURL(j), + Salary: leverSalary(j), + Modality: leverModality(j), + Description: leverDescription(j), + PostedAt: dataPublicacao, + Source: "Lever", + Sources: []string{"Lever"}, + Keyword: keyword, + Keywords: matchedKeywords, + }) + } + + return jobs, nil +} + +func (a *LeverAdapter) fetchPostings(ctx context.Context, req domain.ScrapeRequest) ([]leverPosting, error) { + a.mu.Lock() + if a.loaded { + defer a.mu.Unlock() + return a.cachedJobs, a.cacheErr + } + defer a.mu.Unlock() + + pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond + if pageTimeout <= 0 { + pageTimeout = 15 * time.Second + } + + endpoint := a.postingsEndpoint() + + reqCtx, cancel := context.WithTimeout(ctx, pageTimeout) + defer cancel() + + httpReq, err := http.NewRequestWithContext(reqCtx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, fmt.Errorf("lever: build request: %w", err) + } + + httpReq.Header.Set("User-Agent", "JobsScraper/1.0") + httpReq.Header.Set("Accept", "application/json") + + resp, err := a.client.Do(httpReq) + if err != nil { + return nil, fmt.Errorf("lever: http do: %w", err) + } + defer resp.Body.Close() + + switch resp.StatusCode { + case http.StatusOK: + // ok + case http.StatusTooManyRequests: + return nil, fmt.Errorf("lever: rate limit atingido (429)") + case http.StatusNotFound: + return nil, fmt.Errorf("lever: empresa '%s' não encontrada (404)", a.companySlug) + default: + return nil, fmt.Errorf("lever: status inesperado %d", resp.StatusCode) + } + + if err := json.NewDecoder(resp.Body).Decode(&a.cachedJobs); err != nil { + a.cacheErr = err + a.loaded = true + return nil, fmt.Errorf("lever: decode json: %w", err) + } + + a.loaded = true + + return a.cachedJobs, nil +} + +func (a *LeverAdapter) postingsEndpoint() string { + if a.apiURL != "" { + return a.apiURL + } + + return fmt.Sprintf( + "https://%s/v0/postings/%s?mode=json", + a.apiHost, + a.companySlug, + ) +} + +// func leverMatchesRequest(job leverPosting, keyword string, req domain.ScrapeRequest) bool { +// if keyword != "" && !strings.Contains(leverNormalize(leverSearchText(job)), leverNormalize(keyword)) { +// return false +// } + +// return leverMatchesFilters(job, req) +// } + +func leverMatchesFilters(job leverPosting, req domain.ScrapeRequest) bool { + if req.RemoteOnly && leverModality(job) != "Remoto" { + return false + } + + if location := strings.TrimSpace(req.SearchLocation); location != "" && leverModality(job) != "Remoto" { + if !strings.Contains(leverNormalize(leverLocation(job)), leverNormalize(location)) { + return false + } + } + + return true +} + +func leverMatchingKeywords(job leverPosting, keywords []string) []string { + text := leverNormalize(leverSearchText(job)) + matched := make([]string, 0, len(keywords)) + + for _, keyword := range keywords { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + + if strings.Contains(text, leverNormalize(keyword)) { + matched = append(matched, keyword) + } + } + + return adapterutil.UniqueTrimmedStrings(matched) +} + +func leverIncludeAllJobs() bool { + value := strings.TrimSpace(os.Getenv("LEVER_INCLUDE_ALL_JOBS")) + if value == "" { + return true + } + + return !strings.EqualFold(value, "false") +} + +func leverSearchText(job leverPosting) string { + parts := []string{ + job.Text, + job.Country, + job.Categories.Team, + job.Categories.Department, + job.Categories.Location, + job.Categories.Commitment, + job.Categories.Level, + job.OpeningPlain, + job.DescriptionPlain, + job.DescriptionBodyPlain, + job.AdditionalPlain, + job.SalaryDescription, + job.WorkplaceType, + } + parts = append(parts, job.Categories.AllLocations...) + for _, list := range job.Lists { + parts = append(parts, list.Text, list.Content) + } + return strings.Join(parts, " ") +} + +func leverJobID(job leverPosting) string { + if job.HostedURL != "" { + return strings.TrimSpace(job.HostedURL) + } + if job.ID != "" { + return "lever:" + strings.TrimSpace(job.ID) + } + return strings.TrimSpace(job.Text) +} + +func leverURL(job leverPosting) string { + if job.HostedURL != "" { + return strings.TrimSpace(job.HostedURL) + } + return strings.TrimSpace(job.ApplyURL) +} + +func leverLocation(job leverPosting) string { + locations := []string{job.Categories.Location} + locations = append(locations, job.Categories.AllLocations...) + if job.Country != "" { + locations = append(locations, job.Country) + } + if len(adapterutil.UniqueTrimmedStrings(locations)) == 0 { + locations = append(locations, job.Categories.Department) + } + return strings.Join(adapterutil.UniqueTrimmedStrings(locations), " | ") +} + +func leverModality(job leverPosting) string { + text := leverNormalize(strings.Join([]string{ + job.WorkplaceType, + job.Categories.Location, + strings.Join(job.Categories.AllLocations, " "), + }, " ")) + + switch { + case strings.Contains(text, "hybrid") || strings.Contains(text, "hibrido"): + return "Híbrido" + case strings.Contains(text, "remote") || strings.Contains(text, "remoto"): + return "Remoto" + case strings.Contains(text, "onsite") || strings.Contains(text, "on site") || strings.Contains(text, "on-site"): + return "Presencial" + default: + return strings.TrimSpace(job.Categories.Commitment) + } +} + +func leverDescription(job leverPosting) string { + sections := []string{ + job.OpeningPlain, + job.DescriptionPlain, + job.DescriptionBodyPlain, + } + if len(adapterutil.NonEmptyStrings(sections)) == 0 { + sections = append(sections, stripLeverHTML(job.Description)) + } + for _, list := range job.Lists { + title := strings.TrimSpace(list.Text) + content := stripLeverHTML(list.Content) + if title != "" && content != "" { + sections = append(sections, title+": "+content) + } else { + sections = append(sections, title, content) + } + } + sections = append(sections, job.AdditionalPlain, job.SalaryDescription) + return strings.Join(adapterutil.NonEmptyStrings(sections), "\n\n") +} + +func leverSalary(job leverPosting) string { + if job.SalaryRange == nil { + return strings.TrimSpace(job.SalaryDescription) + } + + min := job.SalaryRange.Min + max := job.SalaryRange.Max + currency := strings.TrimSpace(job.SalaryRange.Currency) + interval := strings.TrimSpace(job.SalaryRange.Interval) + + switch { + case min > 0 && max > 0: + return strings.TrimSpace(fmt.Sprintf("%.0f - %.0f %s %s", min, max, currency, interval)) + case min > 0: + return strings.TrimSpace(fmt.Sprintf("A partir de %.0f %s %s", min, currency, interval)) + case max > 0: + return strings.TrimSpace(fmt.Sprintf("Até %.0f %s %s", max, currency, interval)) + default: + return strings.TrimSpace(job.SalaryDescription) + } +} + +func stripLeverHTML(value string) string { + value = html.UnescapeString(value) + value = regexp.MustCompile(`(?s)<[^>]+>`).ReplaceAllString(value, " ") + value = html.UnescapeString(value) + return strings.Join(strings.Fields(value), " ") +} + +func leverNormalize(value string) string { + value = strings.ToLower(html.UnescapeString(value)) + value = strings.ReplaceAll(value, "-", " ") + value = strings.ReplaceAll(value, "/", " ") + return strings.Join(strings.Fields(value), " ") +} + +func BuildLeverAdapters(ctx context.Context) ([]ports.JobSource, error) { + companies, err := FetchLeverSlugs(ctx) + if err != nil { + return nil, err + } + + result := make([]ports.JobSource, 0, len(companies)) + for _, company := range companies { + slug := strings.TrimSpace(company.Slug) + if slug == "" { + slug = strings.TrimSpace(company.Site) + } + name := strings.TrimSpace(company.Name) + if name == "" { + name = strings.TrimSpace(company.Company) + } + if slug == "" { + continue + } + if name == "" { + name = slug + } + result = append(result, NewLeverWithEndpoint(slug, name, company.Region, company.APIURL)) + } + + return result, nil +} diff --git a/scraper-go/internal/adapters/lever/adapter_test.go b/scraper-go/internal/adapters/lever/adapter_test.go new file mode 100644 index 0000000..ba2c82c --- /dev/null +++ b/scraper-go/internal/adapters/lever/adapter_test.go @@ -0,0 +1,184 @@ +package lever + +import ( + "context" + "net/http" + "strings" + "sync" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestLeverSearchMatchesRichPostingFields(t *testing.T) { + t.Setenv("LEVER_INCLUDE_ALL_JOBS", "false") + + adapter := NewLever("acme", "Acme") + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.String() != "https://api.lever.co/v0/postings/acme?mode=json" { + t.Fatalf("unexpected endpoint: %s", req.URL.String()) + } + return testResponse(`[ + { + "id": "123", + "text": "Backend Engineer", + "hostedUrl": "https://jobs.lever.co/acme/123", + "createdAt": 1785110400000, + "country": "BR", + "categories": { + "team": "Platform", + "department": "Engineering", + "location": "Remote - Brazil", + "commitment": "Full-time", + "level": "Senior", + "allLocations": ["Brazil", "Remote"] + }, + "descriptionPlain": "Build payment services with Java.", + "lists": [{"text": "Stack", "content": "
  • Kafka
  • PostgreSQL
  • "}], + "additionalPlain": "Distributed systems.", + "salaryRange": {"min": 100000, "max": 140000, "currency": "USD", "interval": "per-year-salary"}, + "workplaceType": "remote" + }, + { + "id": "456", + "text": "Finance Analyst", + "hostedUrl": "https://jobs.lever.co/acme/456", + "categories": {"location": "New York", "commitment": "Full-time"}, + "descriptionPlain": "Budget planning.", + "workplaceType": "onsite" + } + ]`), nil + }) + + jobs, err := adapter.Search(context.Background(), "kafka", domain.ScrapeRequest{ + SearchLocation: "Brazil", + RemoteOnly: true, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected one rich field match, got %d", len(jobs)) + } + + job := jobs[0] + if job.ID != "https://jobs.lever.co/acme/123" { + t.Fatalf("unexpected ID: %q", job.ID) + } + if job.Location != "Remote - Brazil | Brazil | Remote | BR" { + t.Fatalf("unexpected location: %q", job.Location) + } + if job.Modality != "Remoto" { + t.Fatalf("expected remote modality, got %q", job.Modality) + } + if !strings.Contains(job.Description, "Stack: Kafka PostgreSQL") { + t.Fatalf("expected list content in description: %q", job.Description) + } + if job.Salary != "100000 - 140000 USD per-year-salary" { + t.Fatalf("unexpected salary: %q", job.Salary) + } +} + +func TestLeverSearchBatchCachesPostingsAcrossKeywords(t *testing.T) { + t.Setenv("LEVER_INCLUDE_ALL_JOBS", "false") + + var ( + mu sync.Mutex + calls int + ) + + adapter := NewLever("acme", "Acme") + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + mu.Lock() + calls++ + mu.Unlock() + + return testResponse(`[ + { + "id": "123", + "text": "Go Engineer", + "hostedUrl": "https://jobs.lever.co/acme/123", + "categories": {"location": "Remote"}, + "descriptionPlain": "Backend APIs", + "workplaceType": "remote" + } + ]`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"go", "backend", "engineer"}, domain.ScrapeRequest{}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 1 { + t.Fatalf("expected one job with merged keywords, got %d", len(jobs)) + } + if strings.Join(jobs[0].Keywords, ",") != "go,backend,engineer" { + t.Fatalf("unexpected keywords: %#v", jobs[0].Keywords) + } + + mu.Lock() + defer mu.Unlock() + if calls != 1 { + t.Fatalf("expected one Lever postings call, got %d", calls) + } +} + +func TestLeverSearchBatchIncludesAllBoardJobsByDefault(t *testing.T) { + adapter := NewLever("acme", "Acme") + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + return testResponse(`[ + { + "id": "123", + "text": "Go Engineer", + "hostedUrl": "https://jobs.lever.co/acme/123", + "categories": {"location": "Remote"}, + "descriptionPlain": "Backend APIs", + "workplaceType": "remote" + }, + { + "id": "456", + "text": "Customer Success Manager", + "hostedUrl": "https://jobs.lever.co/acme/456", + "categories": {"location": "Remote"}, + "descriptionPlain": "Enterprise customer onboarding.", + "workplaceType": "remote" + } + ]`), nil + }) + + jobs, err := adapter.SearchBatch(context.Background(), []string{"go"}, domain.ScrapeRequest{}) + if err != nil { + t.Fatalf("SearchBatch returned error: %v", err) + } + if len(jobs) != 2 { + t.Fatalf("expected all board jobs by default, got %d", len(jobs)) + } + if len(jobs[0].Keywords) != 1 || jobs[0].Keywords[0] != "go" { + t.Fatalf("expected keyword match on first job, got %#v", jobs[0].Keywords) + } + if len(jobs[1].Keywords) != 0 || jobs[1].Keyword != "" { + t.Fatalf("expected unmatched job without keyword metadata, got %q %#v", jobs[1].Keyword, jobs[1].Keywords) + } +} + +func TestLeverUsesExplicitAPIEndpoint(t *testing.T) { + adapter := NewLeverWithEndpoint( + "acme", + "Acme", + "future-region", + "https://api.future.lever.co/v0/postings/acme?mode=json", + ) + adapter.client = testHTTPClient(func(req *http.Request) (*http.Response, error) { + if req.URL.String() != "https://api.future.lever.co/v0/postings/acme?mode=json" { + t.Fatalf("unexpected endpoint: %s", req.URL.String()) + } + + return testResponse(`[]`), nil + }) + + _, err := adapter.Search(context.Background(), "engineer", domain.ScrapeRequest{}) + if err != nil { + t.Fatalf("Search returned error: %v", err) + } +} diff --git a/scraper-go/internal/adapters/lever/test_helpers_test.go b/scraper-go/internal/adapters/lever/test_helpers_test.go new file mode 100644 index 0000000..441f0a7 --- /dev/null +++ b/scraper-go/internal/adapters/lever/test_helpers_test.go @@ -0,0 +1,15 @@ +package lever + +import ( + "net/http" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" +) + +func testHTTPClient(fn testutil.RoundTripFunc) *http.Client { + return testutil.HTTPClient(fn) +} + +func testResponse(body string) *http.Response { + return testutil.Response(body) +} diff --git a/scraper-go/internal/adapters/linkedin.go b/scraper-go/internal/adapters/linkedin/adapter.go similarity index 64% rename from scraper-go/internal/adapters/linkedin.go rename to scraper-go/internal/adapters/linkedin/adapter.go index 0446776..37c30e9 100644 --- a/scraper-go/internal/adapters/linkedin.go +++ b/scraper-go/internal/adapters/linkedin/adapter.go @@ -1,25 +1,34 @@ -package adapters +package linkedin import ( "context" "fmt" + "log/slog" "net/http" "net/url" + "os" + "strconv" "strings" + "sync" "time" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/dedup" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" "github.com/PuerkitoBio/goquery" ) const linkedinSearchURL = "https://www.linkedin.com/jobs-guest/jobs/api/seeMoreJobPostings/search" const linkedinPageStep = 25 +const defaultLinkedInMaxPages = 5 +const defaultLinkedInKeywordSlotSize = 30 type LinkedInAdapter struct { - client *http.Client - semaphore chan struct{} + client *http.Client + semaphore chan struct{} + mu sync.Mutex + nextOffset int } func NewLinkedIn() *LinkedInAdapter { @@ -31,7 +40,19 @@ func NewLinkedIn() *LinkedInAdapter { func (a *LinkedInAdapter) SourceName() string { return "linkedin" } -func buildLinkedInURL(keyword string, req models.ScrapeRequest, start int) string { +func linkedinKeywordSlotSize() int { + value := strings.TrimSpace(os.Getenv("LINKEDIN_KEYWORD_SLOT_SIZE")) + if value == "" { + return defaultLinkedInKeywordSlotSize + } + parsed, err := strconv.Atoi(value) + if err != nil || parsed <= 0 { + return defaultLinkedInKeywordSlotSize + } + return parsed +} + +func buildLinkedInURL(keyword string, req domain.ScrapeRequest, start int) string { u, _ := url.Parse(linkedinSearchURL) q := u.Query() q.Set("keywords", keyword) @@ -58,7 +79,7 @@ func buildLinkedInURL(keyword string, req models.ScrapeRequest, start int) strin return u.String() } -func (a *LinkedInAdapter) fetchJobsChunk(ctx context.Context, keyword string, req models.ScrapeRequest, start int) ([]models.Job, error) { +func (a *LinkedInAdapter) fetchJobsChunk(ctx context.Context, keyword string, req domain.ScrapeRequest, start int) ([]domain.Job, error) { pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond if pageTimeout <= 0 { pageTimeout = 15 * time.Second @@ -95,7 +116,7 @@ func (a *LinkedInAdapter) fetchJobsChunk(ctx context.Context, keyword string, re return nil, err } - var jobs []models.Job + var jobs []domain.Job doc.Find(".base-card, .job-search-card").Each(func(_ int, card *goquery.Selection) { titulo := strings.TrimSpace(card.Find(".base-search-card__title").Text()) if titulo == "" { @@ -110,7 +131,7 @@ func (a *LinkedInAdapter) fetchJobsChunk(ctx context.Context, keyword string, re if link == "" { link, _ = card.Find("a[href*='/jobs/view/']").Attr("href") } - jobs = append(jobs, models.Job{ + jobs = append(jobs, domain.Job{ Title: titulo, Company: empresa, Location: local, @@ -121,7 +142,7 @@ func (a *LinkedInAdapter) fetchJobsChunk(ctx context.Context, keyword string, re return jobs, nil } -func normalizeLinkedInLocation(location string, req models.ScrapeRequest) string { +func normalizeLinkedInLocation(location string, req domain.ScrapeRequest) string { location = strings.TrimSpace(location) searchLocation := strings.TrimSpace(req.SearchLocation) if searchLocation == "" { @@ -196,10 +217,10 @@ func normalizeLinkedInComparable(value string) string { return replacer.Replace(strings.ToLower(strings.TrimSpace(value))) } -func normalizeLinkedInJob(keyword string, req models.ScrapeRequest, job models.Job) models.Job { +func normalizeLinkedInJob(keyword string, req domain.ScrapeRequest, job domain.Job) domain.Job { u := dedup.NormalizeURL(strings.TrimSpace(job.URL)) - normalized := models.Job{ + normalized := domain.Job{ Title: strings.TrimSpace(job.Title), Company: strings.TrimSpace(job.Company), Location: normalizeLinkedInLocation(job.Location, req), @@ -216,8 +237,8 @@ func normalizeLinkedInJob(keyword string, req models.ScrapeRequest, job models.J return normalized } -func dedupeLinkedIn(jobs []models.Job) []models.Job { - unique := make(map[string]models.Job, len(jobs)) +func dedupeLinkedIn(jobs []domain.Job) []domain.Job { + unique := make(map[string]domain.Job, len(jobs)) order := make([]string, 0, len(jobs)) for _, job := range jobs { key := job.URL @@ -232,21 +253,115 @@ func dedupeLinkedIn(jobs []models.Job) []models.Job { order = append(order, key) } } - result := make([]models.Job, 0, len(order)) + result := make([]domain.Job, 0, len(order)) for _, key := range order { result = append(result, unique[key]) } return result } -func (a *LinkedInAdapter) Search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) { - // Semáforo: no máximo 2 keywords rodando ao mesmo tempo no LinkedIn +func (a *LinkedInAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { a.semaphore <- struct{}{} defer func() { <-a.semaphore }() + return a.searchKeyword(ctx, keyword, req) +} + +func (a *LinkedInAdapter) SearchBatch(ctx context.Context, keywords []string, req domain.ScrapeRequest) ([]domain.Job, error) { + slot := a.nextKeywordSlot(keywords, linkedinKeywordSlotSize()) + if len(slot) == 0 { + return nil, nil + } + + if len(slot) < len(keywords) { + slog.Info("linkedin: usando slot rotativo de keywords", + "selected", len(slot), + "total", len(keywords), + ) + } + + type searchResult struct { + jobs []domain.Job + err error + } + + results := make(chan searchResult, len(slot)) + var wg sync.WaitGroup + + for _, keyword := range slot { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + continue + } + + wg.Add(1) + go func(keyword string) { + defer wg.Done() + + select { + case a.semaphore <- struct{}{}: + defer func() { <-a.semaphore }() + case <-ctx.Done(): + results <- searchResult{err: ctx.Err()} + return + } + + jobs, err := a.searchKeyword(ctx, keyword, req) + results <- searchResult{jobs: jobs, err: err} + }(keyword) + } + + wg.Wait() + close(results) + + var allJobs []domain.Job + var firstErr error + for result := range results { + if result.err != nil { + if firstErr == nil { + firstErr = result.err + } + continue + } + allJobs = append(allJobs, result.jobs...) + } + if len(allJobs) > 0 { + return dedupeLinkedIn(allJobs), nil + } + + return nil, firstErr +} + +func (a *LinkedInAdapter) nextKeywordSlot(keywords []string, slotSize int) []string { + if len(keywords) == 0 { + return nil + } + if slotSize <= 0 || slotSize >= len(keywords) { + return append([]string(nil), keywords...) + } + + a.mu.Lock() + defer a.mu.Unlock() + + offset := a.nextOffset % len(keywords) + a.nextOffset = (offset + slotSize) % len(keywords) + + slot := make([]string, 0, slotSize) + for i := 0; i < slotSize; i++ { + slot = append(slot, keywords[(offset+i)%len(keywords)]) + } + return slot +} + +func (a *LinkedInAdapter) searchKeyword(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + keyword = strings.TrimSpace(keyword) + if keyword == "" { + return nil, nil + } + maxPages := req.MaxPagesPerKeyword if maxPages <= 0 { - maxPages = 5 + maxPages = defaultLinkedInMaxPages } waitBetween := time.Duration(req.WaitBetweenSearchesMs) * time.Millisecond @@ -254,7 +369,8 @@ func (a *LinkedInAdapter) Search(ctx context.Context, keyword string, req models waitBetween = 3000 * time.Millisecond } - var allJobs []models.Job + var allJobs []domain.Job + seenPages := make(map[string]struct{}) for pageIndex := 0; pageIndex < maxPages; pageIndex++ { start := pageIndex * linkedinPageStep @@ -286,9 +402,15 @@ func (a *LinkedInAdapter) Search(ctx context.Context, keyword string, req models break } + normalizedJobs := make([]domain.Job, 0, len(jobs)) for _, job := range jobs { - allJobs = append(allJobs, normalizeLinkedInJob(keyword, req, job)) + normalizedJobs = append(normalizedJobs, normalizeLinkedInJob(keyword, req, job)) } + if adapterutil.RepeatedJobPage(seenPages, normalizedJobs) { + break + } + + allJobs = append(allJobs, normalizedJobs...) if pageIndex < maxPages-1 { select { diff --git a/scraper-go/internal/adapters/linkedin_test.go b/scraper-go/internal/adapters/linkedin/adapter_test.go similarity index 79% rename from scraper-go/internal/adapters/linkedin_test.go rename to scraper-go/internal/adapters/linkedin/adapter_test.go index 514beff..b608c6f 100644 --- a/scraper-go/internal/adapters/linkedin_test.go +++ b/scraper-go/internal/adapters/linkedin/adapter_test.go @@ -1,14 +1,14 @@ -package adapters +package linkedin import ( "testing" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/stretchr/testify/assert" ) func TestNormalizeLinkedInLocation_UsaPaisDaBusca(t *testing.T) { - req := models.ScrapeRequest{SearchLocation: "Brasil"} + req := domain.ScrapeRequest{SearchLocation: "Brasil"} assert.Equal( t, diff --git a/scraper-go/internal/adapters/linkedin/pagination_test.go b/scraper-go/internal/adapters/linkedin/pagination_test.go new file mode 100644 index 0000000..167c74f --- /dev/null +++ b/scraper-go/internal/adapters/linkedin/pagination_test.go @@ -0,0 +1,51 @@ +package linkedin + +import ( + "context" + "fmt" + "net/http" + "strconv" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestLinkedInSearchContinuesPastFormerDefaultPagesUntilEmpty(t *testing.T) { + calls := 0 + adapter := NewLinkedIn() + adapter.client = testutil.HTTPClient(func(req *http.Request) (*http.Response, error) { + calls++ + start, err := strconv.Atoi(req.URL.Query().Get("start")) + if err != nil { + t.Fatalf("invalid start query: %v", err) + } + page := start/linkedinPageStep + 1 + if page > 6 { + return testutil.Response(""), nil + } + return testutil.Response(fmt.Sprintf(` +
    +

    Dev Go %d

    +

    Acme

    + Brasil + +
    + `, page, page)), nil + }) + + jobs, err := adapter.Search(context.Background(), "go", domain.ScrapeRequest{ + WaitBetweenSearchesMs: 1, + MaxPagesPerKeyword: 10, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 6 { + t.Fatalf("expected 6 jobs from pages past old limit, got %d", len(jobs)) + } + if calls != 7 { + t.Fatalf("expected 7 calls including empty page, got %d calls", calls) + } +} diff --git a/scraper-go/internal/adapters/registry.go b/scraper-go/internal/adapters/registry.go index 9d49fec..484861e 100644 --- a/scraper-go/internal/adapters/registry.go +++ b/scraper-go/internal/adapters/registry.go @@ -4,43 +4,66 @@ import ( "context" "log/slog" "os" + "strings" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adzuna" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/greenhouse" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/gupy" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/inhire" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/jooble" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/lever" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/linkedin" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/themuse" "github.com/redis/go-redis/v9" ) func GetAdapters(rdb *redis.Client) []Adapter { var list []Adapter - list = append(list, NewLinkedIn()) + list = append(list, linkedin.NewLinkedIn()) if appID, appKey := os.Getenv("ADZUNA_APP_ID"), os.Getenv("ADZUNA_APP_KEY"); appID != "" && appKey != "" { - list = append(list, NewAdzuna(appID, appKey, "br")) + list = append(list, adzuna.NewAdzuna(appID, appKey, "br")) } else { slog.Warn("ADZUNA_APP_ID ou ADZUNA_APP_KEY não configurados, adapter ignorado") } - list = append(list, NewTheMuse()) + list = append(list, themuse.NewTheMuse()) + + if strings.EqualFold(os.Getenv("GUPY_ENABLED"), "true") { + list = append(list, gupy.NewGupy()) + slog.Info("Gupy habilitado") + } + + if strings.EqualFold(os.Getenv("INHIRE_ENABLED"), "true") { + list = append(list, inhire.NewInHire()) + slog.Info("InHire habilitado") + } if apiKey := os.Getenv("JOOBLE_API_KEY"); apiKey != "" { - list = append(list, NewJooble(apiKey, rdb)) + list = append(list, jooble.NewJooble(apiKey, rdb)) } else { slog.Warn("JOOBLE_API_KEY não configurada, adapter ignorado") } - greenhouseSlugs, err := FetchGreenhouseSlugs(context.Background()) - if err != nil { - slog.Warn("falha ao carregar slugs do Greenhouse", "error", err) - } - for _, slug := range greenhouseSlugs { - list = append(list, NewGreenhouse(slug, slug)) + if strings.EqualFold(os.Getenv("GREENHOUSE_ENABLED"), "true") { + greenhouseAdapters, err := greenhouse.BuildGreenhouseAdapters(context.Background()) + if err != nil { + slog.Warn("Greenhouse ignorado: falha ao carregar empresas", "error", err) + } else { + list = append(list, greenhouseAdapters...) + slog.Info("Greenhouse habilitado", "adapters", len(greenhouseAdapters)) + } } - leverCompanies, err := FetchLeverSlugs(context.Background()) - if err != nil { - slog.Warn("falha ao carregar empresas do Lever", "error", err) - } - for _, c := range leverCompanies { - list = append(list, NewLever(c.Slug, c.Name)) + if strings.EqualFold(os.Getenv("LEVER_ENABLED"), "true") { + leverAdapters, err := lever.BuildLeverAdapters(context.Background()) + if err != nil { + slog.Warn("Lever ignorado: falha ao carregar empresas", "error", err) + } else { + list = append(list, leverAdapters...) + slog.Info("Lever habilitado", "adapters", len(leverAdapters)) + } } return list diff --git a/scraper-go/internal/adapters/testutil/http.go b/scraper-go/internal/adapters/testutil/http.go new file mode 100644 index 0000000..97d76d0 --- /dev/null +++ b/scraper-go/internal/adapters/testutil/http.go @@ -0,0 +1,29 @@ +package testutil + +import ( + "io" + "net/http" + "strings" +) + +type RoundTripFunc func(*http.Request) (*http.Response, error) + +func (f RoundTripFunc) RoundTrip(req *http.Request) (*http.Response, error) { + return f(req) +} + +func HTTPClient(fn RoundTripFunc) *http.Client { + return &http.Client{Transport: fn} +} + +func Response(body string) *http.Response { + return StatusResponse(http.StatusOK, body) +} + +func StatusResponse(status int, body string) *http.Response { + return &http.Response{ + StatusCode: status, + Body: io.NopCloser(strings.NewReader(body)), + Header: make(http.Header), + } +} diff --git a/scraper-go/internal/adapters/themuse.go b/scraper-go/internal/adapters/themuse.go deleted file mode 100644 index db0a28e..0000000 --- a/scraper-go/internal/adapters/themuse.go +++ /dev/null @@ -1,161 +0,0 @@ -package adapters - -import ( - "context" - "encoding/json" - "fmt" - "net/http" - "net/url" - "strings" - "time" - - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" -) - -// TheMuseAdapter espelha o theMuseAdapter do JS. -type TheMuseAdapter struct { - client *http.Client -} - -func NewTheMuse() *TheMuseAdapter { - return &TheMuseAdapter{ - client: &http.Client{Timeout: 60 * time.Second}, - } -} - -func (a *TheMuseAdapter) SourceName() string { return "The Muse" } - -// buildTheMuseUrl espelha exatamente o buildTheMuseUrl do JS. -func buildTheMuseURL(keyword string, req models.ScrapeRequest, page int) string { - u, _ := url.Parse("https://www.themuse.com/api/public/jobs") - q := u.Query() - - q.Set("page", fmt.Sprintf("%d", page)) - q.Set("descending", "true") - - if keyword != "" { - q.Set("category", keyword) - } - - if req.SearchLocation != "" { - q.Set("location", req.SearchLocation) - } - - u.RawQuery = q.Encode() - return u.String() -} - -type theMuseResponse struct { - Results []struct { - Name string `json:"name"` - Company struct { - Name string `json:"name"` - } `json:"company"` - Refs struct { - LandingPage string `json:"landing_page"` - } `json:"refs"` - Locations []struct { - Name string `json:"name"` - } `json:"locations"` - PublicationDate string `json:"publication_date"` - } `json:"results"` -} - -func (a *TheMuseAdapter) Search(ctx context.Context, keyword string, req models.ScrapeRequest) ([]models.Job, error) { - maxPages := req.MaxPagesPerKeyword - if maxPages <= 0 { - maxPages = 3 - } - - pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond - if pageTimeout <= 0 { - pageTimeout = 15 * time.Second - } - - waitBetween := time.Duration(req.WaitBetweenSearchesMs) * time.Millisecond - if waitBetween <= 0 { - waitBetween = 1000 * time.Millisecond - } - - var allJobs []models.Job - - for page := 1; page <= maxPages; page++ { - endpoint := buildTheMuseURL(keyword, req, page) - - pageCtx, cancel := context.WithTimeout(ctx, pageTimeout) - jobs, err := a.fetchPage(pageCtx, endpoint, keyword) - cancel() - - if err != nil { - return nil, err - } - - if len(jobs) == 0 { - // Espelha o break do JS quando results está vazio. - break - } - - allJobs = append(allJobs, jobs...) - - if page < maxPages { - select { - case <-ctx.Done(): - return allJobs, nil - case <-time.After(waitBetween): - } - } - } - - return allJobs, nil -} - -func (a *TheMuseAdapter) fetchPage(ctx context.Context, endpoint, keyword string) ([]models.Job, error) { - httpReq, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) - if err != nil { - return nil, err - } - - resp, err := a.client.Do(httpReq) - if err != nil { - return nil, err - } - defer resp.Body.Close() - - if resp.StatusCode != http.StatusOK { - return nil, fmt.Errorf("status %d", resp.StatusCode) - } - - var data theMuseResponse - if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { - return nil, err - } - - if len(data.Results) == 0 { - return []models.Job{}, nil - } - - jobs := make([]models.Job, 0, len(data.Results)) - for _, r := range data.Results { - // Espelha: Array.isArray(job.locations) ? job.locations.map(i => i.name).join(", ") : "" - locationParts := make([]string, 0, len(r.Locations)) - for _, loc := range r.Locations { - locationParts = append(locationParts, loc.Name) - } - local := strings.Join(locationParts, ", ") - - jobs = append(jobs, models.Job{ - ID: strings.TrimSpace(r.Refs.LandingPage), - Title: strings.TrimSpace(r.Name), - Company: strings.TrimSpace(r.Company.Name), - Location: local, - URL: strings.TrimSpace(r.Refs.LandingPage), - PostedAt: r.PublicationDate, - Source: "The Muse", - Sources: []string{"The Muse"}, - Keyword: keyword, - Keywords: []string{keyword}, - }) - } - - return jobs, nil -} diff --git a/scraper-go/internal/adapters/themuse/adapter.go b/scraper-go/internal/adapters/themuse/adapter.go new file mode 100644 index 0000000..07f1345 --- /dev/null +++ b/scraper-go/internal/adapters/themuse/adapter.go @@ -0,0 +1,522 @@ +package themuse + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "net/url" + "os" + "regexp" + "strings" + "sync" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/adapterutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +const ( + theMuseFeedTTL = 30 * time.Minute + theMuseDetailConcurrency = 10 + theMuseDetailTimeout = 15 * time.Second + defaultTheMuseMaxPages = 50 +) + +var ( + theMuseHTMLTagPattern = regexp.MustCompile(`<[^>]*>`) + theMuseSpacePattern = regexp.MustCompile(`\s+`) + theMuseTokenPattern = regexp.MustCompile(`[a-z0-9+#.]+`) +) + +// TheMuseAdapter consome o feed público e filtra localmente por keyword. +type TheMuseAdapter struct { + client *http.Client + mu sync.Mutex + fetchMu sync.Mutex + cache []domain.Job + cached time.Time + cacheID string + details map[string]domain.Job +} + +func NewTheMuse() *TheMuseAdapter { + return &TheMuseAdapter{ + client: &http.Client{Timeout: 60 * time.Second}, + details: make(map[string]domain.Job), + } +} + +func (a *TheMuseAdapter) SourceName() string { return "The Muse" } + +func buildTheMuseURL(page int) string { + u, _ := url.Parse("https://www.themuse.com/api/public/jobs") + q := u.Query() + + q.Set("page", fmt.Sprintf("%d", page)) + q.Set("descending", "true") + + if apiKey := os.Getenv("THEMUSE_API_KEY"); apiKey != "" { + q.Set("api_key", apiKey) + } + + u.RawQuery = q.Encode() + return u.String() +} + +func buildTheMuseDetailURL(id string) string { + u, _ := url.Parse("https://www.themuse.com/api/public/jobs/" + url.PathEscape(id)) + q := u.Query() + + if apiKey := os.Getenv("THEMUSE_API_KEY"); apiKey != "" { + q.Set("api_key", apiKey) + } + + u.RawQuery = q.Encode() + return u.String() +} + +type theMuseNamedValue struct { + Name string `json:"name"` +} + +type theMuseResponse struct { + Results []struct { + ID json.Number `json:"id"` + Name string `json:"name"` + Company struct { + Name string `json:"name"` + } `json:"company"` + Refs struct { + LandingPage string `json:"landing_page"` + } `json:"refs"` + Locations []theMuseNamedValue `json:"locations"` + Categories []theMuseNamedValue `json:"categories"` + Levels []theMuseNamedValue `json:"levels"` + Contents string `json:"contents"` + Description string `json:"description"` + PublicationDate string `json:"publication_date"` + } `json:"results"` +} + +type theMuseDetailResponse struct { + ID json.Number `json:"id"` + Name string `json:"name"` + Company struct { + Name string `json:"name"` + } `json:"company"` + Refs struct { + LandingPage string `json:"landing_page"` + } `json:"refs"` + Locations []theMuseNamedValue `json:"locations"` + Categories []theMuseNamedValue `json:"categories"` + Levels []theMuseNamedValue `json:"levels"` + Contents string `json:"contents"` + Description string `json:"description"` + PublicationDate string `json:"publication_date"` +} + +func (a *TheMuseAdapter) Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) { + feed, err := a.fetchFeed(ctx, req) + if err != nil { + return nil, err + } + + matches := make([]domain.Job, 0) + for _, job := range feed { + if !theMuseMatchesKeyword(job, keyword) { + continue + } + job.Keyword = keyword + job.Keywords = []string{keyword} + matches = append(matches, job) + } + + return matches, nil +} + +func (a *TheMuseAdapter) fetchFeed(ctx context.Context, req domain.ScrapeRequest) ([]domain.Job, error) { + maxPages := req.MaxPagesPerKeyword + if maxPages <= 0 { + maxPages = defaultTheMuseMaxPages + } + + pageTimeout := time.Duration(req.PageTimeoutMs) * time.Millisecond + if pageTimeout <= 0 { + pageTimeout = 15 * time.Second + } + + waitBetween := time.Duration(req.WaitBetweenSearchesMs) * time.Millisecond + if waitBetween <= 0 { + waitBetween = 1000 * time.Millisecond + } + + cacheID := fmt.Sprintf("pages:%d", maxPages) + + a.mu.Lock() + if a.cacheID == cacheID && time.Since(a.cached) < theMuseFeedTTL { + defer a.mu.Unlock() + return append([]domain.Job(nil), a.cache...), nil + } + a.mu.Unlock() + + a.fetchMu.Lock() + defer a.fetchMu.Unlock() + + a.mu.Lock() + if a.cacheID == cacheID && time.Since(a.cached) < theMuseFeedTTL { + defer a.mu.Unlock() + return append([]domain.Job(nil), a.cache...), nil + } + a.mu.Unlock() + + var allJobs []domain.Job + seenPages := make(map[string]struct{}) + + for page := 1; page <= maxPages; page++ { + endpoint := buildTheMuseURL(page) + + pageCtx, cancel := context.WithTimeout(ctx, pageTimeout) + jobs, err := a.fetchPage(pageCtx, endpoint) + cancel() + + if err != nil { + return nil, err + } + + if len(jobs) == 0 { + // Espelha o break do JS quando results está vazio. + break + } + if adapterutil.RepeatedJobPage(seenPages, jobs) { + break + } + + allJobs = append(allJobs, jobs...) + + if page < maxPages { + select { + case <-ctx.Done(): + return allJobs, nil + case <-time.After(waitBetween): + } + } + } + + allJobs = a.enrichJobs(ctx, allJobs) + + a.mu.Lock() + a.cacheID = cacheID + a.cached = time.Now() + a.cache = append([]domain.Job(nil), allJobs...) + a.mu.Unlock() + + return append([]domain.Job(nil), allJobs...), nil +} + +func (a *TheMuseAdapter) fetchPage(ctx context.Context, endpoint string) ([]domain.Job, error) { + httpReq, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, err + } + + resp, err := a.client.Do(httpReq) + if err != nil { + return nil, err + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return nil, fmt.Errorf("status %d", resp.StatusCode) + } + + var data theMuseResponse + if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { + return nil, err + } + + if len(data.Results) == 0 { + return []domain.Job{}, nil + } + + jobs := make([]domain.Job, 0, len(data.Results)) + for _, r := range data.Results { + jobs = append(jobs, theMuseJobFromListResult( + strings.TrimSpace(r.ID.String()), + r.Name, + r.Company.Name, + r.Refs.LandingPage, + r.Locations, + r.Categories, + r.Levels, + r.Contents, + r.Description, + r.PublicationDate, + )) + } + + return jobs, nil +} + +func (a *TheMuseAdapter) enrichJobs(ctx context.Context, jobs []domain.Job) []domain.Job { + if len(jobs) == 0 { + return jobs + } + + enriched := make([]domain.Job, len(jobs)) + copy(enriched, jobs) + + sem := make(chan struct{}, theMuseDetailConcurrency) + var wg sync.WaitGroup + + for i := range enriched { + id := strings.TrimSpace(enriched[i].ID) + if id == "" || strings.HasPrefix(id, "http://") || strings.HasPrefix(id, "https://") { + continue + } + + a.mu.Lock() + detail, ok := a.details[id] + a.mu.Unlock() + if ok { + enriched[i] = mergeTheMuseJob(enriched[i], detail) + continue + } + + wg.Add(1) + sem <- struct{}{} + go func(index int, jobID string) { + defer wg.Done() + defer func() { <-sem }() + + detailCtx, cancel := context.WithTimeout(ctx, theMuseDetailTimeout) + defer cancel() + + detail, err := a.fetchDetail(detailCtx, jobID) + if err != nil { + return + } + + a.mu.Lock() + a.details[jobID] = detail + a.mu.Unlock() + + enriched[index] = mergeTheMuseJob(enriched[index], detail) + }(i, id) + } + + wg.Wait() + return enriched +} + +func (a *TheMuseAdapter) fetchDetail(ctx context.Context, id string) (domain.Job, error) { + httpReq, err := http.NewRequestWithContext(ctx, http.MethodGet, buildTheMuseDetailURL(id), nil) + if err != nil { + return domain.Job{}, err + } + + resp, err := a.client.Do(httpReq) + if err != nil { + return domain.Job{}, err + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return domain.Job{}, fmt.Errorf("detail status %d", resp.StatusCode) + } + + var data theMuseDetailResponse + if err := json.NewDecoder(resp.Body).Decode(&data); err != nil { + return domain.Job{}, err + } + + return theMuseJobFromListResult( + strings.TrimSpace(data.ID.String()), + data.Name, + data.Company.Name, + data.Refs.LandingPage, + data.Locations, + data.Categories, + data.Levels, + data.Contents, + data.Description, + data.PublicationDate, + ), nil +} + +func theMuseJobFromListResult( + id string, + title string, + company string, + landingPage string, + locations []theMuseNamedValue, + categories []theMuseNamedValue, + levels []theMuseNamedValue, + contents string, + description string, + publicationDate string, +) domain.Job { + urlValue := strings.TrimSpace(landingPage) + jobID := strings.TrimSpace(id) + if jobID == "" { + jobID = urlValue + } + + descParts := []string{ + stripTheMuseHTML(contents), + stripTheMuseHTML(description), + strings.Join(theMuseNames(categories), " "), + strings.Join(theMuseNames(levels), " "), + } + + return domain.Job{ + ID: jobID, + Title: strings.TrimSpace(title), + Company: strings.TrimSpace(company), + Location: strings.Join(theMuseNames(locations), ", "), + URL: urlValue, + Description: strings.TrimSpace(strings.Join(nonEmptyStrings(descParts), " ")), + PostedAt: publicationDate, + Source: "The Muse", + Sources: []string{"The Muse"}, + } +} + +func mergeTheMuseJob(base, detail domain.Job) domain.Job { + if detail.ID != "" { + base.ID = detail.ID + } + if detail.Title != "" { + base.Title = detail.Title + } + if detail.Company != "" { + base.Company = detail.Company + } + if detail.Location != "" { + base.Location = detail.Location + } + if detail.URL != "" { + base.URL = detail.URL + } + if detail.Description != "" { + base.Description = detail.Description + } + if detail.PostedAt != "" { + base.PostedAt = detail.PostedAt + } + return base +} + +func theMuseMatchesKeyword(job domain.Job, keyword string) bool { + keyword = strings.TrimSpace(strings.ToLower(keyword)) + if keyword == "" { + return true + } + + searchText := strings.ToLower(strings.Join([]string{ + job.Title, + job.Company, + job.Location, + job.Description, + }, " ")) + + groups := theMuseKeywordGroups(keyword) + if len(groups) == 0 { + return false + } + + for _, group := range groups { + if !theMuseTextHasAnyToken(searchText, group) { + return false + } + } + + return true +} + +func theMuseKeywordGroups(keyword string) [][]string { + terms := theMuseTokenPattern.FindAllString(keyword, -1) + groups := make([][]string, 0, len(terms)) + + for _, term := range terms { + switch term { + case "developer", "desenvolvedor": + groups = append(groups, []string{"developer", "engineer", "dev", "software"}) + case "engineer": + groups = append(groups, []string{"engineer", "developer"}) + case "frontend", "front-end": + groups = append(groups, []string{"frontend", "front-end", "front", "ui"}) + case "backend": + groups = append(groups, []string{"backend", "back-end", "server"}) + case "node.js", "node": + groups = append(groups, []string{"node.js", "nodejs", "node"}) + case "golang", "go": + groups = append(groups, []string{"golang", "go"}) + case "javascript": + groups = append(groups, []string{"javascript", "js"}) + case "typescript": + groups = append(groups, []string{"typescript", "ts"}) + case "c#": + groups = append(groups, []string{"c#", "csharp"}) + case ".net": + groups = append(groups, []string{".net", "dotnet"}) + default: + groups = append(groups, []string{term}) + } + } + + return groups +} + +func theMuseTextHasAnyToken(text string, terms []string) bool { + for _, term := range terms { + if theMuseTextHasToken(text, term) { + return true + } + } + return false +} + +func theMuseTextHasToken(text, term string) bool { + if term == "" { + return true + } + + if strings.Contains(term, ".") || strings.Contains(term, "#") || strings.Contains(term, "+") { + return strings.Contains(text, term) + } + + for _, token := range theMuseTokenPattern.FindAllString(text, -1) { + if token == term { + return true + } + } + + return false +} + +func theMuseNames(values []theMuseNamedValue) []string { + names := make([]string, 0, len(values)) + for _, value := range values { + name := strings.TrimSpace(value.Name) + if name != "" { + names = append(names, name) + } + } + return names +} + +func stripTheMuseHTML(value string) string { + withoutTags := theMuseHTMLTagPattern.ReplaceAllString(value, " ") + return strings.TrimSpace(theMuseSpacePattern.ReplaceAllString(withoutTags, " ")) +} + +func nonEmptyStrings(values []string) []string { + result := make([]string, 0, len(values)) + for _, value := range values { + if trimmed := strings.TrimSpace(value); trimmed != "" { + result = append(result, trimmed) + } + } + return result +} diff --git a/scraper-go/internal/adapters/themuse/pagination_test.go b/scraper-go/internal/adapters/themuse/pagination_test.go new file mode 100644 index 0000000..7fd6c10 --- /dev/null +++ b/scraper-go/internal/adapters/themuse/pagination_test.go @@ -0,0 +1,50 @@ +package themuse + +import ( + "context" + "fmt" + "net/http" + "strconv" + "testing" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters/testutil" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestTheMuseSearchContinuesPastFormerDefaultPagesUntilEmpty(t *testing.T) { + calls := 0 + adapter := NewTheMuse() + adapter.client = testutil.HTTPClient(func(req *http.Request) (*http.Response, error) { + calls++ + page, err := strconv.Atoi(req.URL.Query().Get("page")) + if err != nil { + t.Fatalf("invalid page query: %v", err) + } + if page > 4 { + return testutil.Response(`{"results":[]}`), nil + } + return testutil.Response(fmt.Sprintf(`{ + "results":[{ + "name":"Dev Go %d", + "company":{"name":"Acme"}, + "refs":{"landing_page":"https://example.com/themuse/%d"}, + "locations":[{"name":"Brasil"}], + "publication_date":"2026-07-27T00:00:00Z" + }] + }`, page, page)), nil + }) + + jobs, err := adapter.Search(context.Background(), "go", domain.ScrapeRequest{ + WaitBetweenSearchesMs: 1, + }) + + if err != nil { + t.Fatalf("Search returned error: %v", err) + } + if len(jobs) != 4 { + t.Fatalf("expected 4 jobs from pages past old limit, got %d", len(jobs)) + } + if calls != 5 { + t.Fatalf("expected 5 calls including empty page, got %d", calls) + } +} diff --git a/scraper-go/internal/cache/factory.go b/scraper-go/internal/cache/factory.go index 1ff17eb..06d589b 100644 --- a/scraper-go/internal/cache/factory.go +++ b/scraper-go/internal/cache/factory.go @@ -24,13 +24,33 @@ func NewCache() (Cache, error) { client := redis.NewClient(opts) - ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) defer cancel() - if err := client.Ping(ctx).Err(); err != nil { + if err := waitForRedisPing(ctx, client); err != nil { _ = client.Close() return nil, fmt.Errorf("cache: could not reach Valkey at %q: %w", url, err) } return NewRedisCache(client), nil } + +func waitForRedisPing(ctx context.Context, client *redis.Client) error { + var lastErr error + ticker := time.NewTicker(time.Second) + defer ticker.Stop() + + for { + if err := client.Ping(ctx).Err(); err == nil { + return nil + } else { + lastErr = err + } + + select { + case <-ctx.Done(): + return lastErr + case <-ticker.C: + } + } +} diff --git a/scraper-go/internal/classifier/classifier.go b/scraper-go/internal/classifier/classifier.go new file mode 100644 index 0000000..01a4777 --- /dev/null +++ b/scraper-go/internal/classifier/classifier.go @@ -0,0 +1,388 @@ +package classifier + +import ( + "log/slog" + "math" + "sort" + "strconv" + "strings" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +type familyScore struct { + family string + score int +} + +type sourceClassificationStats struct { + source string + input int + approved int + rejected int + rejectedWithDescription int + rejectedWithKeywords int + rejectedWithTechnologies int + rejectionReasons map[string]int + rejectedFamilies map[string]int + rejectedTechnologies map[string]int + rejectedTitles map[string]int + rejectedCompanies map[string]int +} + +func Classify(job domain.Job) domain.Classification { + text := normalizeText(strings.Join([]string{ + job.Title, + job.Company, + job.Location, + job.Modality, + job.Description, + }, " ")) + + scores := scoreFamilies(text) + technologies := detectTechnologies(text) + seniority := detectSeniority(text) + + if blocked, reason := blockedOpeningReason(job); blocked { + return domain.Classification{ + PrimaryFamily: "other", + Technologies: technologies, + Seniority: seniority, + InScope: false, + Confidence: 0, + Reasons: []string{reason}, + } + } + + if len(scores) == 0 { + return domain.Classification{ + PrimaryFamily: "other", + Technologies: technologies, + Seniority: seniority, + InScope: false, + Confidence: 0, + Reasons: []string{"nenhuma familia reconhecida"}, + } + } + + sort.Slice(scores, func(i, j int) bool { + if scores[i].score == scores[j].score { + return scores[i].family < scores[j].family + } + return scores[i].score > scores[j].score + }) + + primary := scores[0] + related := make([]string, 0, len(scores)-1) + for _, candidate := range scores[1:] { + if candidate.score >= 3 { + related = append(related, candidate.family) + } + } + if primary.family == "mobile" && containsString(technologies, "react-native") { + related = append(related, "frontend") + } + + confidence := math.Min(0.99, 0.35+(float64(primary.score)*0.08)) + + reason := "classificacao local por titulo descricao tecnologias" + if primary.score < 2 { + reason = "score abaixo do minimo" + } + + return domain.Classification{ + PrimaryFamily: primary.family, + RelatedFamilies: unique(related), + Technologies: technologies, + Seniority: seniority, + InScope: primary.score >= 2, + Confidence: math.Round(confidence*100) / 100, + Reasons: []string{reason}, + } +} + +func blockedOpeningReason(job domain.Job) (bool, string) { + title := normalizeText(job.Title) + if title == "" { + return false, "" + } + + nonConcreteTerms := []string{ + "banco de talentos", + "talent pool", + "cadastro reserva", + } + for _, term := range nonConcreteTerms { + if containsTokenOrPhrase(title, term) { + return true, "vaga nao concreta: " + term + } + } + + administrativeTerms := []string{ + "assistente central de reservas", + "central de reservas", + "assistente de negocios", + "central de relacionamento", + "assistente contabil", + "assistente financeiro", + "assistente de autorizacao", + "motorista", + "motorista de van", + "atendente", + "analista qualidade", + "analista de qualidade", + } + for _, term := range administrativeTerms { + if containsTokenOrPhrase(title, term) { + return true, "vaga administrativa: " + term + } + } + + return false, "" +} + +func containsString(values []string, target string) bool { + for _, value := range values { + if value == target { + return true + } + } + return false +} + +func ClassifyJobs(jobs []domain.Job) []domain.Job { + classified := make([]domain.Job, 0, len(jobs)) + statsBySource := make(map[string]*sourceClassificationStats) + + for _, job := range jobs { + stats := classificationStatsForSource(statsBySource, jobSource(job)) + stats.input++ + + classification := Classify(job) + if !classification.InScope { + stats.rejected++ + if strings.TrimSpace(job.Description) != "" { + stats.rejectedWithDescription++ + } + if len(job.Keywords) > 0 || strings.TrimSpace(job.Keyword) != "" { + stats.rejectedWithKeywords++ + } + if len(classification.Technologies) > 0 { + stats.rejectedWithTechnologies++ + } + stats.rejectionReasons[classificationReason(classification)]++ + if classification.PrimaryFamily != "" { + stats.rejectedFamilies[classification.PrimaryFamily]++ + } + for _, technology := range classification.Technologies { + if technology = strings.TrimSpace(technology); technology != "" { + stats.rejectedTechnologies[technology]++ + } + } + stats.rejectedTitles[compactStatLabel(job.Title, "")]++ + stats.rejectedCompanies[compactStatLabel(job.Company, "")]++ + continue + } + + stats.approved++ + job.Classification = &classification + classified = append(classified, job) + } + + logClassificationStats(statsBySource) + + return classified +} + +func classificationStatsForSource(statsBySource map[string]*sourceClassificationStats, source string) *sourceClassificationStats { + stats, ok := statsBySource[source] + if ok { + return stats + } + + stats = &sourceClassificationStats{ + source: source, + rejectionReasons: make(map[string]int), + rejectedFamilies: make(map[string]int), + rejectedTechnologies: make(map[string]int), + rejectedTitles: make(map[string]int), + rejectedCompanies: make(map[string]int), + } + statsBySource[source] = stats + return stats +} + +func jobSource(job domain.Job) string { + if source := strings.TrimSpace(job.Source); source != "" { + return source + } + for _, source := range job.Sources { + if source = strings.TrimSpace(source); source != "" { + return source + } + } + return "unknown" +} + +func classificationReason(classification domain.Classification) string { + for _, reason := range classification.Reasons { + if reason = strings.TrimSpace(reason); reason != "" { + return reason + } + } + if classification.PrimaryFamily != "" && !classification.InScope { + return "fora do escopo: " + classification.PrimaryFamily + } + return "fora do escopo" +} + +func compactStatLabel(value, fallback string) string { + value = strings.Join(strings.Fields(strings.TrimSpace(value)), " ") + if value == "" { + return fallback + } + + runes := []rune(value) + if len(runes) > 90 { + return string(runes[:90]) + "..." + } + + return value +} + +func logClassificationStats(statsBySource map[string]*sourceClassificationStats) { + if len(statsBySource) == 0 { + return + } + + sources := make([]string, 0, len(statsBySource)) + for source := range statsBySource { + sources = append(sources, source) + } + sort.Strings(sources) + + for _, source := range sources { + stats := statsBySource[source] + if stats.input == 0 { + continue + } + + slog.Info("classifier: funil por source", + "source", stats.source, + "input", stats.input, + "approved", stats.approved, + "rejected", stats.rejected, + "rejected_with_description", stats.rejectedWithDescription, + "rejected_with_keywords", stats.rejectedWithKeywords, + "rejected_with_technologies", stats.rejectedWithTechnologies, + "rejection_reasons", topCountLabels(stats.rejectionReasons, 5), + "rejected_families", topCountLabels(stats.rejectedFamilies, 8), + "rejected_technologies", topCountLabels(stats.rejectedTechnologies, 15), + "top_rejected_titles", topCountLabels(stats.rejectedTitles, 12), + "top_rejected_companies", topCountLabels(stats.rejectedCompanies, 12), + ) + } +} + +func topCountLabels(values map[string]int, limit int) []string { + if len(values) == 0 || limit <= 0 { + return nil + } + + type countLabel struct { + label string + count int + } + + items := make([]countLabel, 0, len(values)) + for label, count := range values { + items = append(items, countLabel{label: label, count: count}) + } + + sort.Slice(items, func(i, j int) bool { + if items[i].count == items[j].count { + return items[i].label < items[j].label + } + return items[i].count > items[j].count + }) + + if len(items) > limit { + items = items[:limit] + } + + out := make([]string, 0, len(items)) + for _, item := range items { + out = append(out, item.label+"="+strconv.Itoa(item.count)) + } + + return out +} + +func scoreFamilies(text string) []familyScore { + scores := make([]familyScore, 0, len(familyRules)) + + for _, rule := range familyRules { + score := 0 + + for _, term := range rule.StrongTerms { + if containsTokenOrPhrase(text, term) { + score += 4 + } + } + for _, term := range rule.TechnologyTerms { + if containsTokenOrPhrase(text, term) { + score++ + } + } + for _, term := range rule.NegativeTerms { + if containsTokenOrPhrase(text, term) { + score -= 3 + } + } + + if score > 0 { + scores = append(scores, familyScore{family: rule.Family, score: score}) + } + } + + return scores +} + +func detectTechnologies(text string) []string { + technologies := make([]string, 0) + + for technology, aliases := range technologyAliases { + for _, alias := range aliases { + if containsTokenOrPhrase(text, alias) { + technologies = append(technologies, technology) + break + } + } + } + + technologies = unique(technologies) + sort.Strings(technologies) + return technologies +} + +func detectSeniority(text string) string { + if containsAny(text, "estagio", "estagiario", "intern", "internship", "trainee") { + return "estagio" + } + if containsAny(text, "senior", "sr", "especialista", "lead", "principal", "staff") { + return "senior" + } + if containsAny(text, "junior", "jr", "entry level", "assistente") { + return "junior" + } + return "pleno" +} + +func containsAny(text string, needles ...string) bool { + for _, needle := range needles { + if containsTokenOrPhrase(text, needle) { + return true + } + } + return false +} diff --git a/scraper-go/internal/classifier/classifier_test.go b/scraper-go/internal/classifier/classifier_test.go new file mode 100644 index 0000000..c76e134 --- /dev/null +++ b/scraper-go/internal/classifier/classifier_test.go @@ -0,0 +1,113 @@ +package classifier + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +func TestClassifyFrontendComSinonimosEmPortuguesEIngles(t *testing.T) { + cases := []domain.Job{ + { + Title: "Front-end Software Developer", + Description: "React, Next.js e TypeScript", + }, + { + Title: "Desenvolvedor Front-end", + Description: "Produto web com React", + }, + } + + for _, job := range cases { + classification := Classify(job) + + assert.True(t, classification.InScope) + assert.Equal(t, "frontend", classification.PrimaryFamily) + assert.Contains(t, classification.Technologies, "react") + } +} + +func TestClassifyMobileComFamiliaRelacionadaFrontend(t *testing.T) { + classification := Classify(domain.Job{ + Title: "Senior Software Engineer - React Native", + Description: "Mobile app com TypeScript", + }) + + assert.True(t, classification.InScope) + assert.Equal(t, "mobile", classification.PrimaryFamily) + assert.Contains(t, classification.RelatedFamilies, "frontend") + assert.Contains(t, classification.Technologies, "react-native") + assert.Equal(t, "senior", classification.Seniority) +} + +func TestClassifyForaDoEscopo(t *testing.T) { + classification := Classify(domain.Job{ + Title: "Graphic Designer", + Description: "Branding e campanhas visuais", + }) + + assert.False(t, classification.InScope) + assert.Equal(t, "other", classification.PrimaryFamily) +} + +func TestClassifyRejeitaBancoDeTalentosMesmoComCargoTecnico(t *testing.T) { + classification := Classify(domain.Job{ + Title: "[Banco de Talentos] Pessoa Desenvolvedora Frontend Junior", + Description: "React, TypeScript e desenvolvimento web.", + }) + + assert.False(t, classification.InScope) + assert.Equal(t, "other", classification.PrimaryFamily) + assert.Contains(t, classification.Reasons[0], "vaga nao concreta") +} + +func TestClassifyRejeitaCargoAdministrativoConhecido(t *testing.T) { + cases := []domain.Job{ + { + Title: "Assistente Central de Reservas", + Description: "Atendimento, relacionamento e processos administrativos.", + }, + { + Title: "MOTORISTA DE VAN - AEROPORTO_FLORIANOPOLIS (SC)", + Description: "Transporte de clientes e atendimento operacional.", + }, + { + Title: "ANALISTA QUALIDADE III", + Description: "Processos de qualidade operacional e auditoria.", + }, + { + Title: "Atendente", + }, + } + + for _, job := range cases { + classification := Classify(job) + + assert.False(t, classification.InScope) + assert.Equal(t, "other", classification.PrimaryFamily) + assert.Contains(t, classification.Reasons[0], "vaga administrativa") + } +} + +func TestJobSourceFallback(t *testing.T) { + assert.Equal(t, "InHire", jobSource(domain.Job{Source: " InHire "})) + assert.Equal(t, "Gupy", jobSource(domain.Job{Sources: []string{"", " Gupy "}})) + assert.Equal(t, "unknown", jobSource(domain.Job{})) +} + +func TestTopCountLabelsOrdenaPorContagemENome(t *testing.T) { + labels := topCountLabels(map[string]int{ + "Analista": 3, + "Designer": 1, + "Comercial": 3, + "Administrativo": 2, + }, 3) + + assert.Equal(t, []string{ + "Analista=3", + "Comercial=3", + "Administrativo=2", + }, labels) +} diff --git a/scraper-go/internal/classifier/normalize.go b/scraper-go/internal/classifier/normalize.go new file mode 100644 index 0000000..2fb90c2 --- /dev/null +++ b/scraper-go/internal/classifier/normalize.go @@ -0,0 +1,58 @@ +package classifier + +import ( + "strings" + "unicode" + + "golang.org/x/text/transform" + "golang.org/x/text/unicode/norm" +) + +func normalizeText(value string) string { + t := transform.Chain(norm.NFD, transform.RemoveFunc(func(r rune) bool { + return unicode.Is(unicode.Mn, r) + }), norm.NFC) + + result, _, _ := transform.String(t, value) + + var b strings.Builder + for _, r := range strings.ToLower(result) { + if unicode.IsLetter(r) || unicode.IsNumber(r) { + b.WriteRune(r) + } else { + b.WriteRune(' ') + } + } + + return strings.Join(strings.Fields(b.String()), " ") +} + +func containsTokenOrPhrase(text, needle string) bool { + needle = normalizeText(needle) + if needle == "" { + return false + } + if strings.Contains(needle, " ") { + return strings.Contains(text, needle) + } + return strings.Contains(" "+text+" ", " "+needle+" ") +} + +func unique(values []string) []string { + seen := make(map[string]struct{}, len(values)) + result := make([]string, 0, len(values)) + + for _, value := range values { + value = strings.TrimSpace(value) + if value == "" { + continue + } + if _, exists := seen[value]; exists { + continue + } + seen[value] = struct{}{} + result = append(result, value) + } + + return result +} diff --git a/scraper-go/internal/classifier/taxonomy.go b/scraper-go/internal/classifier/taxonomy.go new file mode 100644 index 0000000..52c843f --- /dev/null +++ b/scraper-go/internal/classifier/taxonomy.go @@ -0,0 +1,162 @@ +package classifier + +type familyRule struct { + Family string + StrongTerms []string + TechnologyTerms []string + NegativeTerms []string +} + +var familyRules = []familyRule{ + { + Family: "backend", + StrongTerms: []string{ + "backend", "back end", "back-end", "server side", "api", "apis", + "microservices", "distributed systems", "sistemas distribuidos", + "desenvolvedor backend", "engenheiro backend", + }, + TechnologyTerms: []string{ + "go", "golang", "java", "spring", "spring boot", "node", "node js", + "nestjs", "python", "django", "flask", "ruby", "rails", "php", + "laravel", "csharp", "dotnet", "postgresql", "mysql", "mongodb", + "redis", "kafka", "rabbitmq", "grpc", "rest", + }, + }, + { + Family: "frontend", + StrongTerms: []string{ + "frontend", "front end", "front-end", "web developer", "ui engineer", + "desenvolvedor frontend", "desenvolvedor front end", "engenheiro frontend", + }, + TechnologyTerms: []string{ + "react", "next js", "vue", "angular", "svelte", "typescript", + "javascript", "html", "css", "tailwind", "graphql", + }, + NegativeTerms: []string{"designer", "ux", "ui designer", "product designer"}, + }, + { + Family: "mobile", + StrongTerms: []string{ + "mobile", "ios", "android", "react native", "flutter", + "desenvolvedor mobile", "engenheiro mobile", + }, + TechnologyTerms: []string{ + "swift", "kotlin", "react native", "flutter", "dart", "android", "ios", + }, + }, + { + Family: "fullstack", + StrongTerms: []string{ + "fullstack", "full stack", "full-stack", "desenvolvedor fullstack", + "desenvolvedor full stack", + }, + TechnologyTerms: []string{ + "react", "node", "node js", "typescript", "javascript", "next js", + "postgresql", "mongodb", + }, + }, + { + Family: "platform", + StrongTerms: []string{ + "platform engineer", "platform software", "plataforma", + "developer platform", "engenheiro de plataforma", + }, + TechnologyTerms: []string{ + "kubernetes", "docker", "terraform", "aws", "azure", "gcp", + "google cloud", "ci cd", "github actions", "gitlab ci", + }, + }, + { + Family: "devops", + StrongTerms: []string{ + "devops", "site reliability", "sre", "cloud engineer", + "infrastructure engineer", "engenheiro devops", "infraestrutura", + }, + TechnologyTerms: []string{ + "kubernetes", "docker", "terraform", "ansible", "jenkins", "aws", + "azure", "gcp", "prometheus", "grafana", + }, + }, + { + Family: "data", + StrongTerms: []string{ + "data engineer", "data analyst", "data scientist", "analytics engineer", + "machine learning", "ml engineer", "ai engineer", "engenheiro de dados", + "cientista de dados", "analista de dados", + }, + TechnologyTerms: []string{ + "python", "sql", "spark", "airflow", "dbt", "bigquery", "snowflake", + "pandas", "tensorflow", "pytorch", + }, + }, + { + Family: "qa", + StrongTerms: []string{ + "qa engineer", "test engineer", "automation engineer", "sdet", + "quality assurance", "analista qa", "engenheiro qa", "tester", + }, + TechnologyTerms: []string{ + "selenium", "cypress", "playwright", "jest", "vitest", "junit", + }, + }, + { + Family: "security", + StrongTerms: []string{ + "security engineer", "cybersecurity", "application security", + "devsecops", "information security", "seguranca da informacao", + }, + TechnologyTerms: []string{ + "owasp", "iam", "soc", "siem", "vulnerability", "pentest", + }, + }, + { + Family: "leadership", + StrongTerms: []string{ + "tech lead", "technical lead", "engineering lead", "engineering manager", + "head of engineering", "cto", "lider tecnico", "gerente de engenharia", + }, + }, + { + Family: "software", + StrongTerms: []string{ + "software engineer", "software developer", "engenheiro de software", + "desenvolvedor de software", "application developer", + }, + }, +} + +var technologyAliases = map[string][]string{ + "angular": {"angular"}, + "aws": {"aws", "amazon web services"}, + "azure": {"azure"}, + "csharp": {"c#", "c sharp", "csharp"}, + "docker": {"docker"}, + "dotnet": {".net", "dotnet", "asp net"}, + "flutter": {"flutter"}, + "gcp": {"gcp", "google cloud"}, + "go": {"go", "golang"}, + "graphql": {"graphql"}, + "java": {"java"}, + "javascript": {"javascript", "js"}, + "kafka": {"kafka"}, + "kotlin": {"kotlin"}, + "kubernetes": {"kubernetes", "k8s"}, + "mongodb": {"mongodb", "mongo db"}, + "mysql": {"mysql"}, + "nestjs": {"nestjs", "nest js"}, + "next.js": {"next js", "nextjs", "next.js"}, + "node.js": {"node js", "nodejs", "node.js"}, + "php": {"php"}, + "postgresql": {"postgresql", "postgres", "postgres sql"}, + "python": {"python"}, + "rabbitmq": {"rabbitmq", "rabbit mq"}, + "react": {"react", "reactjs", "react js"}, + "react-native": {"react native", "react-native"}, + "redis": {"redis"}, + "ruby": {"ruby"}, + "spring": {"spring", "spring boot"}, + "swift": {"swift"}, + "terraform": {"terraform"}, + "typescript": {"typescript", "ts"}, + "vue": {"vue", "vue js", "vuejs"}, +} diff --git a/scraper-go/internal/cronjob/cronjob.go b/scraper-go/internal/cronjob/cronjob.go index 45595e7..a6ef559 100644 --- a/scraper-go/internal/cronjob/cronjob.go +++ b/scraper-go/internal/cronjob/cronjob.go @@ -11,6 +11,7 @@ import ( "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/pipeline" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" ) type Config struct { @@ -30,30 +31,34 @@ func DefaultConfig() Config { SearchLocation: "Brasil", JobTypes: "C,F", TimeFilter: "r604800", - RemoteOnly: true, - MaxConcurrency: 150, + RemoteOnly: false, + MaxConcurrency: 40, } } type Scheduler struct { - cfg Config - kwStore *keywords.Store - jobStore *jobstore.Store - rdb *redis.Client - OnComplete func(keywords []string, scraped, saved int, duration time.Duration) - mu sync.Mutex - running bool - stop chan struct{} + cfg Config + kwStore *keywords.Store + jobStore *jobstore.Store + adapterList []ports.JobSource + rdb *redis.Client + OnComplete func(keywords []string, scraped, saved int, duration time.Duration) + mu sync.Mutex + running bool + lastRunAt time.Time + lastJobs int + stop chan struct{} } -func New(cfg Config, kwStore *keywords.Store, jobStore *jobstore.Store, rdb *redis.Client) *Scheduler { +func New(cfg Config, kwStore *keywords.Store, jobStore *jobstore.Store, adapterList []ports.JobSource, rdb *redis.Client) *Scheduler { return &Scheduler{ - cfg: cfg, - kwStore: kwStore, - jobStore: jobStore, - rdb: rdb, - OnComplete: nil, - stop: make(chan struct{}), + cfg: cfg, + kwStore: kwStore, + jobStore: jobStore, + adapterList: adapterList, + rdb: rdb, + OnComplete: nil, + stop: make(chan struct{}), } } @@ -103,6 +108,12 @@ func (s *Scheduler) IsRunning() bool { return s.running } +func (s *Scheduler) Snapshot() (running bool, lastRunAt time.Time, jobsCollected int) { + s.mu.Lock() + defer s.mu.Unlock() + return s.running, s.lastRunAt, s.lastJobs +} + func (s *Scheduler) run(ctx context.Context) { s.mu.Lock() if s.running { @@ -140,7 +151,7 @@ func (s *Scheduler) run(ctx context.Context) { MaxConcurrency: s.cfg.MaxConcurrency, } - jobs, err := pipeline.ScrapeAllSources(scrapeCtx, config, s.rdb) + jobs, err := pipeline.ScrapeAllSources(scrapeCtx, config, s.adapterList, s.rdb) if err != nil { slog.Error("cronjob: scrape falhou", "error", err) return @@ -156,6 +167,11 @@ func (s *Scheduler) run(ctx context.Context) { // ✅ Constrói o índice invertido para buscas por keyword pipeline.IndexJobsInValkey(scrapeCtx, s.rdb, jobs, kws) + s.mu.Lock() + s.lastRunAt = time.Now() + s.lastJobs = len(jobs) + s.mu.Unlock() + slog.Info("cronjob: execução concluída", "duration", time.Since(start).Round(time.Second), "scraped", len(jobs), diff --git a/scraper-go/internal/dedup/dedup.go b/scraper-go/internal/dedup/dedup.go index 76baf07..3b5c6fd 100644 --- a/scraper-go/internal/dedup/dedup.go +++ b/scraper-go/internal/dedup/dedup.go @@ -8,11 +8,11 @@ import ( "golang.org/x/text/transform" "golang.org/x/text/unicode/norm" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" ) -func DedupeJobs(jobs []models.Job) []models.Job { - unique := make(map[string]*models.Job, len(jobs)) +func DedupeJobs(jobs []domain.Job) []domain.Job { + unique := make(map[string]*domain.Job, len(jobs)) for i := range jobs { job := jobs[i] @@ -27,14 +27,14 @@ func DedupeJobs(jobs []models.Job) []models.Job { if len(job.Sources) == 0 { job.Sources = []string{job.Source} } - if len(job.Keywords) == 0 { + if len(job.Keywords) == 0 && strings.TrimSpace(job.Keyword) != "" { job.Keywords = []string{job.Keyword} } unique[key] = &job } - result := make([]models.Job, 0, len(unique)) + result := make([]domain.Job, 0, len(unique)) for _, j := range unique { result = append(result, *j) } @@ -42,7 +42,7 @@ func DedupeJobs(jobs []models.Job) []models.Job { return result } -func buildKey(j *models.Job) string { +func buildKey(j *domain.Job) string { title := normalizeText(j.Title) company := normalizeText(j.Company) location := normalizeText(j.Location) @@ -66,7 +66,7 @@ func buildKey(j *models.Job) string { return "fallback:" + title + "|" + company + "|" + location + "|" + normalizeText(j.Source) } -func merge(existing, incoming *models.Job) *models.Job { +func merge(existing, incoming *domain.Job) *domain.Job { merged := *existing merged.Sources = uniqueStrings(append(existing.Sources, incoming.Sources...)) diff --git a/scraper-go/internal/domain/job.go b/scraper-go/internal/domain/job.go new file mode 100644 index 0000000..6e30e16 --- /dev/null +++ b/scraper-go/internal/domain/job.go @@ -0,0 +1,52 @@ +package domain + +type Job struct { + ID string `json:"id"` + Title string `json:"title"` + Company string `json:"company"` + Location string `json:"location"` + URL string `json:"url"` + Salary string `json:"salary,omitempty"` + Modality string `json:"modality,omitempty"` + Description string `json:"description,omitempty"` + PostedAt string `json:"postedAt,omitempty"` + Source string `json:"source"` + Sources []string `json:"sources"` + Keyword string `json:"keyword"` + Keywords []string `json:"keywords"` + + Classification *Classification `json:"classification,omitempty"` +} + +type Classification struct { + PrimaryFamily string `json:"primaryFamily"` + RelatedFamilies []string `json:"relatedFamilies,omitempty"` + Technologies []string `json:"technologies,omitempty"` + Seniority string `json:"seniority,omitempty"` + InScope bool `json:"inScope"` + Confidence float64 `json:"confidence"` + Reasons []string `json:"reasons,omitempty"` +} + +type ScrapeRequest struct { + Keywords []string `json:"keywords"` + SearchLocation string `json:"searchLocation"` + SearchGeoID string `json:"searchGeoId"` + SearchLanguage string `json:"searchLanguage"` + JobTypes string `json:"jobTypes"` + TimeFilter string `json:"timeFilter"` + RemoteOnly bool `json:"remoteOnly"` + Sources []string `json:"sources"` + ResultsPerPage int `json:"resultsPerPage"` + MaxPagesPerKeyword int `json:"maxPagesPerKeyword"` + WaitBetweenSearchesMs int `json:"waitBetweenSearchesMs"` + PageTimeoutMs int `json:"pageTimeoutMs"` + MaxConcurrency int `json:"maxConcurrency"` +} + +type ScrapeResponse struct { + Jobs []Job `json:"jobs"` + Total int `json:"total"` + CachedAt string `json:"cachedAt"` + FromCache bool `json:"fromCache"` +} diff --git a/scraper-go/internal/interfaces/greenhouseCompanies.json b/scraper-go/internal/interfaces/greenhouseCompanies.json index 265707e..f77008d 100644 --- a/scraper-go/internal/interfaces/greenhouseCompanies.json +++ b/scraper-go/internal/interfaces/greenhouseCompanies.json @@ -1,34 +1,3341 @@ [ - "ernstandyoung", - "ernstyouth", - "ey", - "varsitytutors", - "bostonconsultinggroup", - "bcg", - "disney", - "riotgames", - "reddit", - "sony", - "twitch", - "robinhood", - "snapchat", - "paloaltonetworks", - "nerdwallet", - "coursera", - "doordash", + "103644278", + "10alabs", + "10beauty", + "10xgenomics", + "12jlkfsk", + "12twenty", + "1456754456yhgbhfg", + "1800contacts", + "1800gotjunk", + "1pyra)mid_health&care", + "204951305985924", + "21shares", + "2k", + "2kearlycareers", + "2kmadrid", + "2u", + "2uinlinepromotions", + "31stunion", + "3cloud", + "3dayblindscorporate", + "3mgroofing", + "3redpartners", + "5364856uhdfnvbkldfnbhrpkdfgbdvtyhro", + "540", + "5minlab", + "5wpr", + "60decibelsinc", + "66degrees", + "6sense", + "7shifts", + "8451", + "8thlightrebuild", + "a3c41b8b71eff8c4", + "aacpnw", + "abacusfinancegroup", + "abacusinsights", + "abbyy", + "abcellera", + "abilitypath", + "abinbev", + "able", + "absherconstruction", + "absolutesecurityintl", + "absurdventures", + "accela", + "accelerationpartners", + "acceleronfusion", + "accelschools", + "accenturefederalservices", + "accesshealthca", + "accessholdingsmanagementfirm", + "accesso", + "accordion", + "accreditedlabs", + "accuweather", + "aceable", + "acentria", + "acilearning", + "aclu", + "acluinternships", + "aclunc", + "acog", + "acommerce", + "acornhealth", + "acquia", + "acrisureinnovation", + "acryldata", + "action", + "actpowerservices", + "acuitymd", + "acumen", + "acutechgroupinc", + "ada18", + "adamsclinical", + "adapter", + "addepar1", + "adelphigrouplimited", + "adelphiresearch", + "adfinternational", + "aditumbio", + "adjustjobs", + "admios", + "adswerveinc", + "advancedtechnologyservices", + "advertisingspecialtyinstitute", + "advocateconstruction", + "advocatesforchildrenofnewyork", + "adyen", + "aechelontechnology", + "aegisventures", + "aegworldwide", + "aerospike", + "aestudio", + "aevexaerospace", + "affinidi", + "affinitiv", "affirm", - "cloudera", - "automattic", - "gitlab", - "elastic", - "confluent", - "databricks", - "instacart", + "affirmedrxpbc", + "afresh", + "ag1", + "agebold", + "agecareers", + "agency", + "agencywithin", + "ageoflearninginc", + "agibank", + "agilesix", + "agoda", + "ahrefsjobs", + "aift", + "air", + "aircapture", + "aircompany", + "airnorth", + "airsculpt", + "airship", + "airspace", + "airtable", + "airtamejobs", + "airtrunk", + "airwave", + "aizerhealth", + "ajboggs", + "akidolabs", + "akko", + "akuity", + "alabamatitleloansinc", + "alamarbiosciences", + "alarmcom", + "albertmackenziellp", + "alertmedia", + "algolia", + "align", + "align46", + "alkujobs", + "allbooked", + "allcareers", + "allencontrolsystems", + "alliancedefendingfreedom", + "alliedmaker", + "allinc", + "alltrna", + "allwebleads", + "alma31", + "aloyoga", + "alpaca", + "alpha9oncology", + "alphaalternatives", + "alphafmcroles", + "alphagrepsecurities", + "alphapublicschools", + "alphasense", + "alphasensehelsinki", + "alphasenseindia", + "alpineinternships", + "alpineinternships-private", + "alt", + "altanaai", + "altentechnologyusa", + "altium", + "altoslabs", + "altruist", + "altscore", + "alu", + "alumis", + "alumniventures", + "alveole", + "alxafrica", + "amaehealth", + "ambercharterschools", + "ambiqmicroinc", + "ameelio", + "amenitiz", + "americanfloodcoalition", + "americaninstitutesforresearch", + "amiralearning", + "amity", + "amount", + "amperesand", + "amperon", + "amplemarket", + "amplitude", + "ampm", + "ampsortation", + "amtechsoftware", + "amwell", + "amylyx", + "amyrisinc", + "analyst1", + "analyticservicesinc", + "anaplan", + "anchanto", + "andesite", + "andurilindustries", + "animalmedicalcenter", + "aninebing", + "annexonbioscience", + "anodize", + "antenna", + "anteriad", + "anteristech", + "anthropic", + "antora", + "aoti", + "apartmentlife", + "aperaaiinc", + "aperiasolutions", + "aperiatechnologies", + "apexcompanies", + "apexcompaniescsw", + "apexit", + "apiiro", + "apiphani", + "apogeetherapeutics", + "apolloio", + "apothecom", + "apparatus", + "appdirect", + "appfire", + "appian", + "appier", + "applovin", + "applytobambi", + "applytogreenspark", + "applytoslabstack", + "applytosuno", + "applytowhoosh", + "appnovation", + "appodeal", + "appomni", + "appspace", + "apptronik", + "aptoslabs", + "aputah", + "aquaticcapitalmanagement", + "aquia", + "arborenergy", + "arcadiacareers", + "arcaea", + "arcanaanalytics", + "arcboatcompany", + "arceeai", + "arcesiumllc", + "archer56", + "archera", + "arcinstitute", + "arcoeducacao", + "ardentmc", + "arenaclub", + "arenaim", + "arevonenergyimpltest", + "arielinvestments", + "arizeai", + "arkestroinc", + "arkoselabs", + "arlosolutionsllc", + "armada", + "armamentsresearchcompany", + "armissecurity", + "armorcode", + "array", + "arrayeducation", + "arspharmaceuticalsoperationsinc", + "artefact", + "artefactlinkedin", + "articlegroup", + "arvaintelligence", + "arvinas", + "arxroboticsgmbh", "asana", - "zoom", - "peloton", - "box", - "okta", + "ascend21", + "asgjobs", + "ashfieldadvisory", + "ashfieldmedcomms", + "asigovernment", + "aspectbiosystems", + "aspirehealthalliance", + "assemblyai", + "assetliving", + "assetwatch", + "assystinc", + "asteralabs", + "astoundcommercesandbox", + "astranis", + "astspacemobile", + "atalantatherapeutics", + "atariinc", + "atbayjobs", + "atek", + "athleticsbaseballops", + "athleticsbusinessops", + "atlassand", + "atlasxhm", + "atomai", + "atomiccartoons", + "atomicmachines", + "atriumcampus", + "attain", + "attainpartners", + "attainsports", + "attaintalent", + "attentionarc", + "attentive", + "attivopartners", + "attn", + "attotrading", + "attune", + "atwellgroup", + "auctane", + "audaxgroup", + "audibenehearcom", + "audioeye", + "augmentcomputing", + "auldwhitetalentcommunity", + "aura798", + "aurosglobal", + "authenticbrandsgroup", + "authenticinsurance", + "autods", + "autogenai", + "automatticcareers", + "automox", + "autoproff", + "autoscout24", + "autotradercanada", + "avantium", + "avantus", + "avebykormancommunities", + "aviationinstituteofmaintenance", + "avidhealth", + "avidxchangeinc", + "avomdincdbaavo", + "avride", + "awin", + "axiad", + "axicorpfinancialservicesptyltd", + "axios", + "axle", + "axon", + "axonag", + "axontalentcommunity", + "axq", + "axs", + "axsometherapeutics", + "axuall", + "aypapower", + "azragames", + "azuritypharmaceuticals", + "azuritypharmaceuticalsindia", + "b12", + "babylist", + "backblaze", + "baidu", + "bamboohr17", + "bancopan", + "bandwidth", + "banyancanopygroup", + "banyaninfrastructure", + "banyansoftware", + "barbarian", + "barbaricum", + "baringa", + "barkleyokrp", + "barrfoundation", + "basejobs", + "bathworksmichigan", + "baton", + "baublebar", + "bauerhockeycascademaveriklacrosse", + "bayada", + "bayasystems", + "bbyo", + "bdainc", + "beaconbiosignals", + "beam", + "beamtherapeutics", + "beamup", + "bearing", + "beatboxbeveragesllc", + "beautifulai", + "beelinemedicines", + "behavox", + "bellcabinetry", + "bellroy", + "benchprep", + "bennie", + "berkadiatalentpool", + "berkshiregroupllc", + "berlinrosen", + "bertramcapitalmanagement", + "berylls", + "bestpass", + "besxar", + "bethesdahealthgroup", + "betsson", + "betterhelp", + "betterhelpcom", + "bettinghero", + "bettygamingca", + "bettyjobboard", + "bevicareers", + "bevlabvet", + "beyondfinance", + "beyondtrust", + "bgbx", + "bgbxconsulting", + "bgeinc", + "bgeinccampus", + "bidease", + "billiontoone", + "biohub", + "biolumina", + "biomechanicsconsultingandresearchllc", + "biomedrealty", + "bird", + "birgo", + "bitcoindepot", + "bitgo", + "bitly", + "bitmex", + "bitpanda", + "blab", + "blackbirdhealth", + "blackcanyonconsulting", + "blackduck", + "blackedgecapital", + "blackforestlabs", + "blacklane", + "blackthorn", + "blankstreet", + "blastpoint", + "blend", + "blenheimchalcot", + "blenheimchalcotindia", + "blinkhealth", + "blip-global", + "blockchain", + "blockrenovation", + "bloombergorg", + "bloomerang", + "bloomreach", + "bluecherry", + "blueconic", + "bluecrestcapitalmanagement", + "bluecubeservices", + "bluefishai", + "bluehole", + "bluelabsanalyticsinc", + "bluemoonmetals", + "blueprintmedicines", + "blueroseresearch", + "bluestarfamilies", + "bluevineindia", + "bluevineus", + "bluewaterthinking", + "blumira", + "blytheco", + "bmnt", + "bobbie", + "bobtail", + "boingo", + "boku", + "boldmetrics", + "boloai", + "bombas", + "bondora", + "bondvet", + "boomentertainment", + "boomilp", + "boostedai", + "boostlingo", + "bosapropertiesinc", + "bottomlinetechnologies", + "bouldercare", + "boulevard", + "boxinc", + "bpcs", + "bpd", + "bracebridgecapital", + "braeburn", + "brainlabs", + "brainpop", + "brainstation", + "braintrusttutors", + "branch", + "brandtechplus", + "brandwatch", + "braskem", + "brave", + "braveheartbio", + "bravo", + "braze", + "breakwatercorp", + "breezeairways", + "breezecash", + "breezeway", + "brevium", + "bridgebio", + "bridgewater89", + "bridgewaterassociatescampusrecruitingreferral", + "brightai", + "brightai1", + "brightcoreenergy", + "brightsign", + "brightstonetherapy", + "brillapubliccharterschools", + "bringg", + "britishasiantrust", + "britive", + "brkz", + "broadsign", + "broadvoice", + "broadwayventures", + "brookecharterschools", + "brooklinen", + "brph", + "brunswickgroup", + "bruntworkwear", + "bswift", + "btgpactualchile", + "btig27", + "bubbleskincare", + "bugcrowd", + "builder", + "buildingdecarbonizationcoalition", + "buildkite", + "buncha", + "bungie", + "businessoffashion", + "butcherbox", + "butlr", + "butternutbox", + "buyersedgeplatformrecruiting", + "buynomics", + "buzzsolutions", + "bvnk", + "bwreferrals", + "c3el", + "c6bank", + "cabify", + "cadencehealth", + "cadrehospice", + "cafortune", + "cais", + "cakeai", + "calahealth", + "calendly", + "callrail", + "calm", + "calyxinstitute", + "calyxo", + "cameo", + "camp", + "campuscompact", + "cancostileandstone", + "candid", + "candidaturasdirecionadasxpinc", + "candidly", + "cannabisandglass", + "cannondale", + "canonical", + "canopyconnect", + "canopytax", + "canopyworks", + "capco", + "capintel", + "capitalbank", + "capitalfarmcredit", + "capitalgymnasticscedarpark", + "capitalgymnasticsroundrock", + "capitalontap", + "capitaltg", + "capstoneinvestmentadvisors", + "captivation", + "carbon", + "carbonchain", + "carbondirect", + "carbonfuture", + "carbonrobotics", + "cardata", + "cardinalpoint", + "careaccess", + "careerteam", + "cargomatic", + "cariadinc", + "caribou", + "cariboubiosciencesinc", + "carmichaellynch", + "carolinatitleloansinc", + "carrotfertility", + "carta", + "cascadeloans", + "casechek", + "caseguard", + "casestatus", + "cashcowlouisiana", + "catamountconstructors", + "catapultsports", + "catawiki", + "catchcreationllc", + "catdaddy", + "caylent", + "cayuse", + "cbemllc", + "cbinsights", + "cc", + "ccah", + "ccahremote", + "cclfg", + "cclim", + "cdbabyjobs", + "celerocommunicationsinc", + "celigo", + "cellanome", + "cellsignalingtechnology79", + "celonis", + "censys", + "centerforemploymentopportunities", + "centessapharmaceuticalsinc", + "centralreach", + "centriaautism", + "centriahealthcare", + "centrumhealth", + "centuracollege", + "cerebral", + "ceribell", + "ceros", + "certifiedgroup", + "chainguard", + "championhq", + "championsgroupholdings", + "chanzuckerberginitiative", + "chaosindustries", + "chaparralmedicalgroup", + "chariotdefense", + "charlesriverassociates", + "chartbeatinc", + "charterup", + "checkalt", + "checkbook", + "checkr", + "cheddar", + "chefman", + "chenmoore", + "chicagotradingcampus", + "childrenstreehouse", + "chile", + "chime", + "chorusinnovations", + "chowbus", + "christfellowship", + "cie", + "circleso", + "cision", + "cityoffortworth", + "citytherapeutics", + "cityvetinc", + "civicactions", + "civisanalytics", + "clara", + "clarishealth62", + "clariticloudinc", + "clarityinnovates", + "claros", + "claudecorps", + "cleancroptech", + "clear", + "clearfield", + "clearlinktechnologiesllc", + "clearscoretechnologylimited", + "clearstreet", + "clearviewhealthcarepartners", + "clearwayjobs", + "clenera", + "cleo", + "cleo-emea", + "cleoindia", + "clevelandguardiansbops", + "clickhouse", + "clicktherapeutics", + "climateai", + "climatecabinet", + "clockworksystems", + "closure-tech", + "cloudbeds", + "cloudbedsthirdpartyboard", + "cloudchamberen", + "cloudflare", + "cloudsek", + "cloverhealth", + "cloverly", + "clubcolors", + "clubmonaco", + "clutch", + "cmt", + "coactive", + "coast", + "cobaltio", + "cobaltservicepartners", + "cobblestoneenergy", + "cobblestoneenergy4", + "cobre", + "coconutsoftware", + "cocoon", + "codazen", + "code3", + "codeorg", + "codepath", + "coderoad", + "coefficient", + "cofertility", + "cofraholding", + "cogentbiosciences", + "cognite", + "cognitiv", + "cogstateinc", + "coherehealth", + "coherusbiosciences", + "coinbase", + "colabsoftware", + "colehourcoheninc", + "colemanresearch", + "collectively", + "collegetrack", + "colovore", + "commerceiq", + "commercetools", + "commonthreadcollective", + "communitymanager", + "commvault", + "comparaja", + "compeerfinancial", + "compliancygroupllc", + "comstock", + "concentric", + "concerthealth", + "conga", + "conifersaicareers", + "connectder", + "connectedcannabis", + "connectedcannabisco", + "connectwise", + "consensys", + "constantcontact", + "constellationsoftwareinc", + "constructionresources", + "consumeredge", + "consumerreports", + "contentful", + "convene", + "convenientmd", + "cooksys", + "copperco", + "cordance", + "cordellcordell", + "cordial81", + "corelight", + "coreone", + "coretelligent", + "coretrustpurchasinggroupllc", + "coreview", + "correlationone", + "cortex", + "cortica", + "corticamelmed", + "cortland", + "cosmoslabs", + "costar", + "cottinghambutlerinsuranceservicesinc", + "couchbaseinc", + "counterpart", + "coupanginternal", + "courierhealth", + "coursera", + "covar", + "coverahealth", + "cpisecurity", + "cpm", + "craftsmansocials", + "cranialtechnologies", + "creativefabrica", + "credible", + "crescent", + "crescolabs", + "cresta", + "crestwoodcareers", + "crexi", + "crfamilyofcompanies", + "criminaljusticeagency", + "crisprecruit", + "criticalmass", + "criticalmassgroup", + "crocodilecloth", + "croihealth", + "crossbeam", + "crunchyroll", + "crystaldynamics", + "csciconsulting", + "csgconsultants", + "css", + "cssmerge", + "culthealth", + "cultureamp", + "curaleaf", + "curative", + "curicapital", + "current", + "customcomputerspecialists", + "customerio", + "cuyana", + "cvx", + "cybersheath", + "cymulate", + "dagsterlabs", + "daiyafoodsinc", + "darkhorseemergency", + "darkwolfsolutions", + "dashlane", + "databento", + "databricks", + "datacamp", + "datagrail", + "dataiku", + "dataikujobs", + "datakindinc", + "datarails", + "datasocietyresearchinstitute", + "datasystemsanalystsinc", + "datavant2", + "datsolutions", + "davidzwirnergallery", + "day1academies", + "daybreakhealth", + "daylight", + "daymarkhealth", + "dbeaver", + "dcard", + "ddbhealth", + "ddome", + "debtbook", + "debutbiotech25", + "decimainternational", + "dedicatedit", + "deepintent", + "deeplocal", + "deepmind", + "defcon", + "defenseunicorns", + "definitivehc", + "definitivehcindia", + "definiumtherapeutics", + "degreed", + "deliveryassociates", + "democracyforward", + "democracypreppublicschools", + "densityai", + "dental365", + "denverbroncosteamllc", + "dept", + "descope", + "descript", + "designbridge", + "designedconveyorsystems", + "detect94", + "detroitlions", + "developmentseed", + "devfuturetalent", + "devrev", + "devries", + "devtechnology", + "dfinity", + "dhigroupinc", + "dhpace", + "dialpad", + "dianahealth94", + "dianthustherapeutics", + "dicefm-careers", + "digicert", + "digimarc", + "digisecuritysystems", + "digitalbiology", + "digitalbridge", + "digitalcurrencygroup", + "digitalextremes", + "diligentcorporation", + "diligentrobotics", + "dimagi", + "discmedicine", + "discord", + "distantjob", + "districtofcolumbiainternationalschool", + "distrokid", + "divcowest", + "divergent", + "dkatalislabs", + "dkbcodefactory", + "dlhcorporation", + "dlpbank", + "dlrgroup", + "dmgevents", + "dna", + "dnsfilter", + "doctolib", + "docugami", + "dodgshunmedlin", + "doitintl", + "dollarshaveclub", + "doma", + "donorbox", + "donorschoose", + "donorschoosefellowship", + "doordashaustralia", + "doordashcanada", + "doordashinternational", + "doordashmexico", + "doordashusa", + "dots", + "doublegood", + "doubleverify", + "downtownmusic", + "doximity", + "dragos", + "drdansanimalhospital", + "droit", + "droplet", + "drsquatch", + "drweng", + "dsi", + "duettoresearch", + "durable", + "durinmining", + "dv01", + "dvtrading", + "dxacirca", + "dynetherapeutics", + "dyopath", + "eamesinstitute", + "earlycareerprograms", + "earnin", + "eastharlemtutorialprogram", + "eastsidedermatology", + "easygo", + "ebanx", + "ebury", + "echodynecorp", + "eclinicalsolutions", + "eclipse", + "eclipsetrading", + "ecoatmgazelle", + "edelmanfinancialenginesllc", + "edgewoodpartnersinsurancecenter", + "edmentum", + "edo", + "edobestsandbox", + "educate", + "education", + "efficientcomputer", + "eikontherapeutics", + "eko", + "eldersburgveterinaryhospital", + "electrasteel", + "elementalimpact", + "elementbiosciences", + "elend", + "eleoshealth", + "eleventhhourgames", + "eliotcommunityhumanservices", + "elitedentalpartnersllc", + "elitetechnology", + "elliginthealth", + "elwoodtechnologies", + "emarketer", + "embroker", + "emergentlabsinc", + "emotainizioengage", + "emplifimonster", + "employerdirecthealthcare", + "empowerbrands", + "emslinqinc", + "enavatecareers", + "enchargeai36", + "encora10", + "encoura", + "endeavourinspiredinfrastructure", + "endorlabs", + "energage", + "energyexemplarllc", + "energyhub", + "energyimpactpartnerslp", + "energysolutions", + "energysolutionsinternships", + "enforce", + "engageseniortherapy", + "engine", + "engineersgate", + "englishcanada", + "enigmaio", + "enlacehealth", + "ennoblecare", + "enova", + "ensco", + "enscoskillbridge", + "entera", + "enterpret", + "entersekt", + "entradatherapeutics", + "entreehealth", + "envisionconsulting", + "enviva", + "envoyglobalinc", + "envoymortgage", + "eolapower", + "eonio", + "epickids", + "episodesix", + "episodesixlinkedin", + "eplusinc", + "eqtcorporation", + "equalexperts", + "equatic", + "equilibriumenergy", + "equipmentsharecom", + "eqvilentjobs", + "erasca", + "ergeon", + "ernesta", + "ernestpackagingsolutions", + "escribers", + "espirita", + "essential", + "etchedai", + "etchinc", + "etec", + "etelligentgroup", + "ethernovia", + "ethos", + "ethoslife", + "eucalyptus", + "eudia", + "eulerity", + "eve", + "eventbriteinc", + "eventsandinterns", + "ever", + "everagtester", + "evergreennephrology", + "evergreenservicesgroup", + "everlane", + "everlaw", + "everway", + "evgspecialtynetwork", + "evio", + "evismart", + "evolutionaryscale", + "evolutioniq", + "evolvevacationrental", + "exabeam", + "exadelinc", + "exame", + "excelsportsmanagement", + "exodus54", + "exoduspoint", + "experigreen", + "expertnetwork", + "explorasolutions", + "extend", + "extenteamcareers", + "externaljobboards", + "extrahopnetworks", + "eyeo", + "ezcaterinc", + "ezcatertalentcommunity", + "factored", + "factorialenergy", + "faeththerapeutics", + "faire", + "fairlife", + "fairmarkit", + "fairsteadescllc", + "falconx", + "fambrands", + "familyoffice", + "familywell", + "fanaticscollectibles", + "fanaticsfbg", + "fanaticsinc", + "faradayfuture", + "farmsanctuary", + "faropoint", + "fartherfinance", + "fashionnova", + "fastautoloansinc", + "fastpaydayloansfloridainc", + "fay", + "fccincinnati", + "federato", + "felixandfido", + "fender", + "fermat", + "ferocia", + "feverup", + "fgsglobal", + "fictiv", + "fieldstonebio", + "figma", + "figment", + "figure", + "figureai", + "filescom", + "filson", + "financialtimes33", + "finepointconsulting", + "finitestate", + "firaxis", + "fireworksai", + "firmpilotailawfirmmarketing", + "firmus", + "firstconnectinsurance", + "firsthand", + "firstnationalbankofamerica", + "firstprinciples", + "firststepsforkids", + "five9", + "fiveringsllc", + "fixify", + "flagshippioneeringinc", + "flash", + "flatironenergy", + "flaunt", + "fleetio", + "fletcherjonesautomotivegroup", + "flex", + "flexport", + "flighthub", + "flip", + "flipdish", + "flockhomes", + "flodesk", + "flohealth", + "flourish", + "flowtraders", + "fluxon", + "fluxx", + "flyr", + "flywheeldigital", + "focalsystems", + "focusfinancialpartners", + "focuspartnerswealth", + "foliahealth", + "follettsoftware", + "folxhealth", + "forafinancial", + "forbes", + "forcetherapeutics", + "foresitelabs", + "forgebiologics", + "forgeglobal", + "forgehealth", + "formaaiinc", + "formaaiinc3", + "formationbio", + "formhealth", + "formic", + "forta", + "forter", + "forthea", + "fortisfiresafety", + "fortra", + "fortrobotics", + "forumone", + "forwardnetworks", + "fosphamarketing", + "fossainc", + "found", + "foundationriskpartners", + "foundrydigital", + "fourhands", + "fourkites", + "fourthline", + "foxen", + "foxholetechnology", + "fractile", + "fractylhealthinc", + "freedassociates", + "freedomtogether", + "freeformfuturecorp", + "freenome", + "freenow", + "freestonecapitalmanagement", + "freshprints", + "fronterahealth", + "frontierdermatologyprovidercareers", + "fsastorecom", + "fsg", + "fueledcareers", + "fulfil", + "funga", + "fuseglobalpartners", + "fusionworldwide", + "future", + "futurhealth", + "g2vp", + "gaintheory", + "gaithersburg", + "galaxydigitalservices", + "galaxyservicepartners", + "galileo", + "gallaghereveliusjonesllp", + "gallup", + "galvanizeclimatesolutions", + "gametimeunited", + "gapinternational", + "garagedoormedics", + "gardacp", + "garnerhealth", + "gassouth", + "gatesventures", + "gather", + "gatherai", + "gatikaiinc", + "gavindebeckerassociates", + "gavindebeckerassociatesindeed", + "gcmgrosvenor", + "gearbox", + "gearboxquebec", + "gelbergroup", + "gelberhandshake", + "gelfandrennertfeldman", + "gemechanicalasp", + "gemini", + "genea", + "generalassembly", + "generalatlantic", + "generalcatalyst", + "generalmatter", + "generalproximity", + "genetixbiotherapeutics", + "genevatrading", + "genezenlabs", + "geniantllc", + "genius", + "geniussports", + "geniussportssn", + "genscript", + "gensyn", + "geocgi", + "georgiaautopawninc", + "geotab", + "geotekoperationslimited", + "getbuilt", + "getwhys", + "gibsondunn", + "gigaenergy", + "gillig", + "gingerlabsinc", + "ginkgobioworks", + "girleffect", + "gitai", + "gitlab", + "givecampus", + "givedirectly", + "givewell", + "glance", + "gleanwork", + "globalaccelerator", + "globalenergyallianceforpeopleandplanetgeappllc", + "globalhealthcareexchangeinc", + "globalizationpartners", + "globalli", + "globalwebindex", + "glossgenius", + "glossier", + "glyphicbiotechnologies", + "go1au", + "go1eu", + "go1us", + "goalcast", + "goalhonduras", + "goalsyria", + "goatgroup", + "gocardless", + "godfreydadichpartners", + "gofundme", + "goguardian", + "gokenamericallc", + "goldenstate", + "golf", + "gomotive", + "gongio", + "goodbysilversteinpartners", + "goodfire", + "goodhouse", + "goodinside", + "goodjobgames", + "goodnotes", + "goodr", + "goodsservices", + "goodwaygroup", + "goop", + "goremutualinsurance", + "gorjana", + "gotion", + "govtechbarbados", + "gr8tech", + "gradial", + "gradientai", + "grafanalabs", + "grahamcapitalmanagement", + "gramgamescareers", + "granum", + "graphcore", + "grasshopperasia", + "grassi", + "grassrootsanalytics", + "grayscaleinvestments", + "greatstate", + "greenbrookmedical", + "greenhouse", + "greenirony", + "greenpeace", + "greenplaces", + "greenpointtechnologies", + "greenthumbindustries", + "greenworkssunriseglobalmarketing", + "greyus", + "gridmaticinc", + "groma", + "groomecareers", + "group14", + "group14korea", + "grovecollaborative", + "grovelane", + "grover", + "growe", + "growetalents", + "gruns", + "grvty", + "gsgcareers", + "gsrmarkets", + "gtv", + "guardianrestoration", + "guardsquare", + "guerrilla-games", + "guidelighthealth", + "guidepoint", + "guidepointsecurity", + "guidepostmontessori", + "guildgaragegroup", + "gulfwindtechnology", + "gumgum", + "gusto", + "gwkinvestmentmanagementllc", + "gympass", + "gymshark", + "habitathealth", + "hackerrank", + "haigroup", + "haizelabs", + "hala", + "halcyon", + "hangar13", + "hankooktireamericacorp", + "hanover", + "hanwhaenergyusa", + "hanwharenewables", + "harbingermotors", + "harborglobal", + "harmonic", + "harnessinc", + "harpergroup", + "harrison&star", + "harrowhealth", + "harrys", + "hasbro", + "hatchcareers", + "haven", + "havenenglish", + "hazel", + "hbstudios", + "hc360", + "hcg", + "headlandsresearch", + "headlandstechnologiesllc", + "headoutcareers", + "headoutlinkedin", + "headoutreferrals", + "headspaceproviders", + "healthjoy", + "healthlink", + "hearcom", + "hearcomin", + "heartaerospace", + "heartflowinc", + "heartpaw", + "hebrewpublic", + "heliosx", + "helium", + "helium10", + "hellobackpack", + "hellofresh", + "hellommc", + "help", + "helpinghandsfamily", + "hereio", + "herselfhealth", + "hexagonbio", + "hexium", + "heygen", + "hiddenlayer", + "highdive", + "higherlogic", + "highmetric", + "highnote", + "hightouch", + "highwire", + "hillandknowlton", + "hillel", + "hillhousehome", + "hillpointe", + "hiper", + "hippo70", + "hiro", + "hivewatch", + "hlb90067", + "holder", + "holderconstruction", + "holisticindustries", + "hologram", + "homechef", + "homeconstructionregulatoryauthorityvolunteer", + "homeinstead", + "homelight", + "homemarketfoods", + "homesolutions", + "hometap", + "hometapjobs", + "homeward", + "honeathome", + "honehealth", + "honeycomb", + "hoodhp", + "hootsuite", + "hopscotchprimarycare", + "hopskipdrive", + "horacemannagents", + "horacemannservicecorporation", + "horizenlabs", + "horizenlabstalentpool", + "horizonindustrieslimited", + "hotmartcareersbr", + "hourglasscosmetics", + "housemarque", + "housinganywhere", + "hovercraft", + "hoyoverse", + "hpiq", + "hrpliving", + "hs", + "huddleup", + "hudl", + "hugeinc", + "humanagency", + "humaninterest", + "humanrightswatch", + "humansignal", + "humeai", + "hummingbirdregtech", + "hungryroot", + "huntress", + "hut8", + "hydritechemicalco", + "hyliion", + "hyphenconnect", + "ians", + "ibanfirst", + "icapitalnetwork", + "iconcareers", + "iconiq", + "idahotitleloansinc", + "ideo", + "idme", + "idmeuniversityrecruiting", + "idnow", + "ie", + "ifoodcarreiras", + "iftother", + "iherb", + "ilia", + "imaginepediatrics", + "imagineworldwide", + "imbue", + "immersive", + "immunefi", + "immunomeinc", + "impact", + "impinjexternal", + "impiricus", + "implicit", + "imvtcorporation", + "inalabconsulting", + "incadigitalinc", + "inceptive", + "inchargeenergy", + "incode", + "incognia", + "indicacoesifoodinterno", + "indigenouspactpbcinc", + "indigo", + "industrialelectricmanufacturing", + "industriouslabs", + "infinitumelectric", + "inflectionai", + "infotrust", + "infuse", + "ingenious", + "inhometherapy", + "inizio", + "iniziomedical", + "inkind", + "inmobi", + "innodatainc", + "insomniac", + "inspiraeducation", + "inspire11", + "inspiremedicalsystemsinc", + "inspiren", + "instabase", + "instawork", + "instead", + "instiglio", + "instride", + "instridehealth", + "insurify", + "insurityindia", + "intecrowd", + "integra", + "integralmolecular", + "intelluminc", + "inter", + "interbrand", + "intercom", + "interdependence", + "internaljobsatlush", + "interstellarlab", + "intersystems", + "interviewengineering", + "interwellhealth", + "interworks", + "inthepocket", + "intradiem", + "intrinsicrobotics", + "inversionspace", + "invivyd", + "inyova", + "ionq", + "ionqcontractors", + "iovancebiotherapeutics", + "iovino", + "iris", + "irradianttechnologies", + "isaac", + "isaraerospace", + "isccareers", + "isidor", + "ism", + "isomorphiclabs", + "ispottv", + "itd", + "iterable", + "iterativehealth", + "its", + "itslogisticsllc", + "ivxhealth", + "ixllearning", + "jadebiosciences", + "jamfilled", + "janestreet", + "jennikayne", + "jensenhughes", + "jjsnackfoods", + "jobsatphamily", + "jobsforthefuture", + "johnmcorcorancompany", + "joinaffect", + "joinforage", + "joinparadigm", + "jomboymedia", + "jordanparkgroup", + "joskoasp", + "journey", + "joya", + "jrmconstructionmanagementllc", + "jshiddenevents", + "jukeboxhealth", + "julieproducts", + "jumio", + "jumpcrypto", + "jumptrading", + "juno", + "just-global", + "justanswer", + "justfoodcompany", + "justfund", + "justprotein", + "justworks", + "juullabs", + "k2spacecorporation", + "kailera", + "kairospower", + "kalcon", + "kalepa", + "kalshi", + "karat", + "karbon", + "kardfinancialinc", + "kardigan", + "karya", + "kasa", + "katalyst", + "kateschwartzphysicaltherapyjobs", + "kayak", + "kayzen", + "keeleyconstruction", + "keelinfrastructure", + "keepersecurity", + "kellerpostman", + "kenjyatrusantgroup", + "kensingtoncorporate", + "kensingtontours", + "keplergroup", + "kernalbio", + "ketryx", + "keyfactorinc", + "khaerospace", + "khaite", + "khanacademy", + "khealthcareers", + "kickstarter", + "kidscountry", + "kikoff", + "kincellbio", + "kindbridgecorporation", + "kindsnacks", + "kinexus", + "kivaorg", + "kiwicoinc", + "knak", + "knightdivisiontactical", + "knit", + "knithealth", + "knock", + "knowbe4", + "knowde", + "knowledgecity", + "known", + "koboldmetals", + "koboldmetalsdrc", + "kodiak", + "kodiaksolutions", + "koleyjessen", + "kolmacintegratedbehavioralhealth", + "komodohealth", + "konovo", + "konux", + "koodoo", + "kostelanetzllp", + "krafton", + "kraftonamericas", + "kraftonindia", + "kreativekids", + "krollbondratingagency", + "kronosresearch", + "kunai", + "kuraoncology", + "kurosbiosciencesinc", + "kyowakirinusa90", + "la28careers", + "labelbox", + "labviva", + "lacarguy", + "ladder33", + "lakeasbury", + "lakefrontbiotherapeuticsinc", + "landdesign", + "landor", + "langanengineeringandenvironmentalservicesllc", + "laportefr", + "laporteusa", + "lasenza", + "lastpass", + "later", + "latitude", + "launch2", + "launchdarkly", + "launchpadtechnologiesinc", + "lawmatics", + "lawzero", + "layerhealth", + "layerzerolabs", + "lazarusenterprises", + "leaflink", + "leagueinc", + "learneo", + "learningnetwork", + "learnlux", + "learnupon", + "ledgy", + "legalservicesnyc", + "legatosecurity", + "legendcareers", + "legion", + "leit", + "lemurianlabs", + "lendingtree", + "leolabsinc", + "levanta", + "levelaccess", + "leveltenenergy", + "levelworks", + "levio", + "levitate", + "lexingtonmedical", + "lgairesearch", + "lgelectronics", + "li-thermal-works", + "liberate", + "licor", + "life360", + "lifelinkiii", + "lifeskillsautismacademy", + "lightfeatheriollc", + "lightforceorthodontics", + "lighthouse", + "lighthousebehavioralhealthsolutions", + "lightningai", + "lightspeeddms", + "lightspeedsystems", + "lilasciences", + "lineate", + "link", + "lionsheadprecisionmetals", + "liontree", + "lisc", + "lithic", + "litify", + "litmos", + "littlebutterflies", + "littlewordsproject", + "livecareer", + "livefront", + "lively43", + "liveparalleljobs", + "liveperson", + "liveviewtechnologiesinc", + "localcoin", + "localitymediallcdbafirstdue", + "lockwood", + "locusrobotics", + "loenbro", + "loftfederal", + "logicalintelligence", + "logicgate", + "logos", + "lokainc", + "looneyrickskiss", + "loonshotgames", + "lovable", + "loyal36", + "lp-professionalstaff", + "lpc", + "lucidbots", + "lucidmotors", + "lucidsoftware", + "luckybeverageco", + "ludorobotics", + "lumahealth", + "lumbermens", + "lumimeds", + "lunarenergy", + "luno", + "lush", + "lusternational", + "lyellimmunopharma", + "lynxanalytics", + "m0dbathenextthingltd", + "m2ingredients", + "m2xenergyinc", + "m9solutions", + "mabl", + "machinifyinc", + "mackaysposito", + "madano", + "maddoxindustrialtransformer", + "madisonlogicinc", + "maev", + "magnolia", + "magrathea", + "mainstreethealth", + "makeawishamerica", + "malbon", + "mammothbrands", + "mandm", + "manifoldai", + "manifoldbio", + "manscaped", + "mantl", + "mantrahealth", + "manychat", + "map", + "maravailifesciences", + "margaux", + "mariadbplc", + "markforged", + "markmorrisdancegroup", + "marksman", + "markspainrealestate-corp", + "marqeta", + "marqvision", + "marscousg", + "marshallwace", + "martellgrowthsolutions", + "maslanskycareers", + "masterclass", + "matchpointtx", + "materialbank", + "matherheadquarters", + "matherintysons", + "matherplace", + "mathersplendido", + "matik", + "matteprojects", + "mattermost", + "matx", + "mavenclinic", + "mavenrobotics", + "mavenscareers", + "mavensecuritiesholdingltd", + "maxcessinternational", + "maxmanufacturingcareers", + "maymobility", + "mbooth", + "mboothhealth", + "mcadams", + "mcclureoilcorporation", + "mcghealth", + "mcneeswallacenurickllc", + "mechanicallicensingcollective", + "medelitellc", + "medeloop", + "mediabrands", + "mediasmart", + "medispend", + "medistrava", + "meditelecare", + "medium", + "medrio", + "medsien", + "mejuri", + "melio", + "memic", + "mentalhealthcenterofdenver", + "mentimeter", + "merceradvisors", + "mercury", + "mercyforanimals", + "merge", + "mergeworld", + "meridianpartners", + "meruhealth", + "mesh", + "meshopticaltechnologies", + "messari", + "metalab", + "metapack", + "method", + "methodco", + "metisinc", + "metoxinternationalinc", + "metron", + "metronome", + "metropolis", + "mgproperties", + "mhi", + "michaelanthonycontractingcorp", + "microblink", + "midihealth", + "midpenhousing", + "midpointmarkets", + "mighty", + "mightynetworks", + "mill", + "millerremick", + "mindgrub", + "mindgym", + "mindtheproduct", + "mineralystherapeutics", + "minimal", + "minio", + "minitab", + "minno", + "miopartners", + "miqdigital", + "mirakl", + "mirakllabs", + "miris", + "misfitsmarket", + "mishimoto", + "missionhealthcare", + "missionlane", + "missionlanellc", + "mississippititleloansinc", + "missourititleloansinc", + "mithril", + "mitratech", + "mitsogoinc", + "mitsubishimotorsna", + "mixbook", + "mixpanel", + "mlbnetwork", + "mm", + "mntn", + "mobentertainment", + "mobiik", + "mobileanesthesiologists", + "mobilityware", + "mochihealth", + "moderawealthmanagement", + "modernhealth", + "modulrfinance", + "mogli", + "mojangab", + "moloco", + "momentenergy", + "momentous", + "momentumcompany3", + "momentumfinancialservicesgroup", + "momentus", + "monetserviceinc", + "moneyherogroup", + "moneymart", + "moneysmart", + "mongodb", + "monks", + "monocl", + "monsterenergy", + "monstro", + "monumentalsports", + "monzo", + "moodhealth", + "moon", + "moonlite", + "morganmorganjobsapplynow", + "morrisonmaierle", + "mos", + "mossnewyorkllc", + "motifneurotech", + "motional", + "motivity", + "movementstrategy", + "moveonorg", + "mozilla", + "mqreferrals", + "mrapplecareers", + "mrbeastyoutube", + "msfcareers", + "mthreerecruitingportal", + "muckrack", + "muonspace", + "murj", + "mwinternshipprogram", + "mx51", + "mxtechnologiesinc", + "myfitnesspal", + "myfundedfutures", + "myriad360", + "mythicalgames", + "n26", + "n2publishingglassdoor", + "nabis", + "nadiacare", + "nakedfarmer", + "nametag", + "nanit", + "nanonets", + "nanopathinc", + "nansen", + "narvar", + "nascompany", + "nashvillezoo", + "natera", + "national", + "nationalbusinesscapital", + "nationallutheraninc", + "nationalpublicradioinc", + "naturesbakery", + "naughtydog", + "navapbc", + "navierboat", + "navtechnologies", + "navvis", + "nearform", + "nearspacelabs", + "neo4j", + "neocybernetica", + "neonaerospace", + "neoris", + "neptunebio", + "neptunemedical", + "nerdy", + "nerostechnologies", + "netbrain", + "netdocuments", + "neteasegames", + "netlify", + "neuehealth", + "neuraflash", + "neurahealth", + "neuralink", + "nevadatitleandpaydayloansinc", + "newarkacademy", + "newengeninc", + "neweratech", + "neweratechnology", + "newglobesandbox", + "newlabcareers", + "newleafenergy", + "newlimit", + "newrelic", + "newsbreak", + "newsela", + "newsrevenuehub", + "newsweek", + "nexightgroup", + "nextinsurance66", + "nextroll", + "nextrollinc", + "nflcareers", + "ngmbiopharmaceuticals", + "nift", + "nightdivestudios", + "nimblerobotics", + "nimbus", + "ninedotholdingsinc", + "ninjatrader", + "ninjatradercontractors", + "nisc", + "nitricity", + "nlcventures", + "nmcareers", + "nmi", + "noahmedical", + "noble", + "nobuhoteltoronto", + "noctrixhealth", + "noctuatechnology", + "nomadhealth", + "nomina", + "nonprofitfinancefund", + "northbeam", + "northeastanimalclinic", + "northern911", + "northmarq", + "northpointrecoveryholdingsllc", + "northpointtechnology", + "northspyre", + "northwestadministratorsinc", + "notexternal", + "nova401", + "novacredit", + "novoed", + "noyocareers", + "nozominetworks", + "ntconcepts", + "nuancelabs", + "nubank", + "nucleus", + "numa", + "nurix", + "nutrafol", + "nutscom", + "nuvem", + "nve", + "nycedc", + "nysonian", + "o2epcminc", + "oafkenya", + "oasishealthpartners", + "oasissecurity", + "obexp", + "obrienveterinarygroup", + "obsidiansecurity", + "obsidiantherapeutics", + "oceanx", + "ocrolusinc", + "octaura", + "octus", + "oddball", + "odeko", + "odlesalescareers", + "offerup", + "offerzen", + "officehours", + "officespacesoftware", + "offshorelaunch", + "ogilvyhealthuk", + "ogilvyhealthusa", + "ogilvymena", + "ogroup", + "ohalogenetics", + "oklo", + "okx", + "oldcastlebuildingenvelope", + "olema", + "olipop", + "oliver", + "oliverseapac", + "oliverusa", + "olly", + "olsson", + "omadahealth", + "omenai", + "omidyarnetwork", + "omnicomhealth", + "onapsis", + "onbe", + "oncoverycare", + "oneacrefundethiopia", + "oneacrefundglobal", + "oneacrefundmalawi", + "oneacrefundnigeria", + "oneacrefundrwanda", + "oneacrefundtanzania", + "oneacrefunduganda", + "oneacrefundzambia", + "oneearthfuture", + "oneenergyrenewables", + "oneimaging", + "onemodel", + "onenergy", + "oneoncology", + "onepath", + "onesixsolutions27", + "onetrust", + "onexgeneral", + "onnitlabs", + "onxmaps", + "ooma", + "openap", + "opencoreventures", + "openeye", + "openfx", + "openly", + "opentable", + "openwork", + "operantai", + "operationscareers", + "ophelia", + "oportun", + "opploans", + "optimadermatologycareers", + "optimal", + "optimalcare", + "optimaldynamics", + "optimecare", + "optimove", + "optiverus", + "oralsurgerypartners", + "orangetwist", + "orchard", + "orchestra", + "orderly", + "orennia", + "oriongroup", + "orium", + "orkes", + "orthosportsmedphysicaltherapyjobs", + "oruka", + "osano", + "oscar", + "oshihealth", + "osmosis", + "otrcapital1", + "otter", + "ouihelp", + "oulahealth", + "ounceofcare", + "oura", + "outerspace", + "outpostspace", + "outrider", + "outschool", + "outsetmedical", + "outshine", + "overline", + "overstory", + "owenscompaniesasp", + "ownwell", + "oxosmedical", + "p72pi", + "pacaso", + "pacificfusion", + "pacificlegalfoundation", + "packardculliganwater", + "pacnyc", + "pacvue", + "pagaya", + "pagayais", + "pagerduty", + "paidyinc", + "pairteam", + "palaktakidsacademy", + "pallet", + "palmettocleantech", + "palosverdes", + "pandadoc", + "pandvil", + "panthalassa", + "pantheonpublic", + "pantherlabs", + "papa", + "papaya", + "paperlessparts", + "parachutehealth", + "parachutehome", + "parallel", + "parallellearning", + "paretocaptiveservicesllc", + "parloa", + "parrishdevaughn", + "parsecautomationcorp", + "parsleyhealth", + "particle41llc", + "passportbranddesign", + "pathrobotics", + "pathstream", + "pathward", + "pathwaysforlife", + "patientpoint", + "patients&purpose", + "patterndata", + "paveakatroveinformationtechnologies", + "pawapay", + "paxlabs", + "pay2dc", + "paynearmeinc", + "paypay", + "paypaycard", + "paytient", + "pdi", + "pdtpartners", + "peachpilot", + "pearceservices", + "pei", + "pelago", + "pendo", + "penninteractive", + "peregrinetechnologies", + "perfectserve", + "perform-careers", + "perionnetworkltd", + "perpay", + "perscholashires", + "persefoniaiinc", + "personalisinc", + "personalizedbeautydiscoveryincdbaipsy", + "petag", + "pfm", + "phaidra", + "phamily", + "phantomai", + "phasev", + "philadelphiaeagles", + "philliesbaseballoperations", + "philo", + "phoenixcontact", + "phonepe", + "phynetdermatology", + "physicsx", + "picarroinc", + "picoquantitativetrading", + "pieinsurance", + "piermontbank", + "pilothq", + "pineadvisorsolutions", + "pingidentity", + "pinn", + "pinterest", + "pinwheelapi", + "pipetechnologies", + "pirateship", + "pistontechnologies", + "pitchbookdata", + "pivotbio", + "pixability", + "placementsio", + "planetlabs", + "planetscale", + "planningcenter", + "platacard", + "platformscience", + "platinumdermphysicians", + "playsports", + "plootocareers", + "plos", + "plscareers", + "pluspower", + "pma", + "pmg", + "pmguk", + "podium81", + "point72", + "pointc", + "pointwild", + "pokemoncareers", + "polyai", + "polychaincapital", + "polygonus", + "pomelocare", + "pontera", + "poppulo", + "porschesouthbay", + "portable", + "porternovelli", + "portland-communications", + "poshmark", + "post-gazette", + "postman", + "postmanlaw", + "postpartumsupportinternational", + "postscript", + "powerdigitalmarketing", + "powerfinance", + "powerhousearts", + "practicebetter", + "practisinglawinstitute", + "prathaminternational", + "praxent", + "praxis", + "praxisprecisionmedicines", + "precisionaq", + "precisionmedicinegroup", + "precisionvehicleholdings", + "predictiveindex", + "presencelearning", + "presidents", + "presidentssummit", + "prevail", + "prezzee", + "pricefox", + "primemedicine", + "primerai", + "primexbt", + "prisma6", + "prismatic", + "privateequityinsights", + "privatehealthmanagement", + "procaresolutions", + "processstreet", + "prodigal", + "productpeople", + "profluent", + "project44", + "project44opportunities", + "projectaservicesgmbhcokg", + "prokidney", + "prolaio", + "prolific", + "pronto", + "prophecysimpledatalabs", + "prophero", + "prophet", + "propublica", + "prosek", + "proskill", + "prosperhealth", + "proteinqureinc", + "protillionbiosciences", + "protonai", + "prove", + "psibufet", + "pt", + "pubgmadison", + "public", + "publiclabel", + "pulley", + "pulse", + "pulsebiosciences", + "pulumicorporation", + "pumpcareers", + "purestorage", + "purplestrategies", + "purposemed", + "pursuit", + "pushpay", + "putnamassociatesllc", + "pyramidroofing", + "pzenainvestmentmanagement", + "qgenda", + "qohash", + "qphox", + "quadbridge", + "quadraturecapital", + "quaise", + "qualia", + "qualifieddigital", + "qualio", + "quanata", + "quansight", + "quantifind", + "quantumsi25", + "quantumspacellc", + "quartzbio", + "quberesearchandtechnologies", + "queracomputinginc", + "questbridge", + "quillbot", + "quilt", + "quince", + "quintoandar", + "quip", + "rabinmartin", + "racapitalmanagementllc", + "rackner", + "radar", + "radiantsecurity", + "radiclehealth", + "radixark", + "radixexperienced", + "radixuniversity", + "raft", + "railsware", + "raisin", + "ramosmarbleandgranite", + "rangeviewinc", + "rapidfortinc", + "rapidsos", + "rapp", + "rayeitconsulting", + "razorpaysoftwareprivatelimited", + "rclco", + "rcxsports", + "rdccareers", + "rdsourcing", + "rdstation", + "readysettechnologyinc", + "realcapital", + "realchemistry", + "reallygreatreading", + "realm", + "rebag", + "rebelliondefense", + "rebtel", + "rebuildmanufacturing", + "recall", + "recidiviz", + "recordedfuture", + "recruitingprograms", + "rectanglehealth", + "recursionpharmaceuticals", + "redcellpartners", + "reddit", + "redpeak", + "redventures", + "redwoodmaterials", + "redwoodsoftware", + "referralsuseonly", + "reflective", + "reformation", + "regscale", + "relationalai", + "relativity", + "relaygraduateschoolofeducation", + "relaypayments", + "relaypro", + "relaytherapeutics", + "released", + "relishworks", + "reltio", + "remedyhomehealthcare", + "remixtherapeutics", + "remodelhealth", + "remoracarbon", + "remotasks", + "remotecom", + "remotereferralboardinternaluseonly", + "renaissancelearning-emea", + "renaissancelearning-nam", + "renewedvision", + "renpsg", + "renttherunway", + "repisodic", + "reproductivefreedomforall", + "reprofreedomforallinternships", + "res", + "researchpartnership", + "residenthome", + "residential", + "resilience", + "resolvetosavelives", + "resortpass", + "restaurantsupply", + "resultsforamerica", + "revero", + "revivn", + "revloncorporate", + "rewardsnetwork", + "rexfordindustrial", + "rhombuspower", + "rhythmx-ai", + "rialtic", + "ribolifamilywines", + "ridgeline", + "rimestechnologies", + "ringtherapeutics", + "ripcpc", + "rise8", + "rithum", + "rithumliboard", + "ritterandbrogdenorthodontics", + "ritual", + "rivainternationalinc", + "rivaltechnologies", + "riverai", + "riversidenaturalfoodsltd", + "riviamind1", + "roadie", + "robertrauschenbergfoundation", + "robinhood", + "roblox", + "roboforce", + "robotsandpencils", + "rockbot", + "rocketlab", + "rocketlawyer", + "rocketmiles", + "rockstargames", + "roivantsciences", + "roku", + "roller", + "rondoenergy", + "roo", + "roofr", + "roofstock", + "root", + "rothesaygraduates", + "rothesaylife", + "route06", + "rpa", + "rubiconcarbon", + "ruelala", + "ruggable", + "ruggedrobotics", + "rumble", + "runwise", + "rushdownstudios", + "rushstreetinteractive", + "russett", + "rvi", + "rvohcontentfreelance", + "rvohealth", + "rxsense", + "rzero", + "rzr", + "safebreach", + "safetyworxs", + "sage49", + "sagebionetworks", + "salientmotion", + "saltbox", + "saltxc", + "samainc", + "samayaai", + "samsungresearchamerica", + "samsungresearchamericainternship", + "samsungsemiconductor", + "sandscapitalmanagementllc", + "sandstonecarecastlerock", + "saraworks", + "saxbys", + "saxllp", + "sayari", + "sbigrowth", + "sbusa", + "scaleai", + "scanner", + "schonfeld", + "schrdinger", + "science&purpose", + "science37", + "scileads", + "scm", + "scopely", + "scorpionenterprisesllc", + "scotch", + "scoutai", + "scoutmotors", + "scoutspace", + "scowtt", + "scsfinancial", + "sdcpinternshipprogram", + "seafireresortltd", + "seaporttherapeutics", + "seatown", + "seattlesoundersfc", + "secondharvest", + "secretariatadvisorsllc", + "securitize", + "securityscorecard", + "sedna", + "seed", + "seesaw", + "seisandbox", + "selffinancial", + "selinicapital", + "semafor", + "sendabiosciences", + "sendcloudnew", + "sensei", + "sensiblecare", + "seoulrobotics", + "septerna", + "serhant", + "serif", + "sertis", + "servicechampions", + "serviceexpertsllc", + "servicewizard", + "sesai", + "sesolabor", + "setpoint", + "setsales", + "sevenresearch", + "sezzle", + "sfox", + "shakepay", + "shapercapital", + "sharebite", + "sharkninjaoperatingllc", + "sharpelectronics", + "shearwater", + "shein", + "shennonbiotechnologies", + "shieldshealthsolutions", + "shift5", + "shifttechnology", + "shinola", + "shinolaretail", + "shipbobinc", + "shipmonk", + "shopltk", + "shopmy", + "showpad", + "shyftsolutionsllc", + "si", + "siboneinc", + "sidecarhealth", + "siei", + "sierrallc", + "sightlinemediagroup", + "sigmacomputing", + "sigmoid", + "signerscareers", + "signifyd95", + "silananotechnologies", + "silverado", + "silvr", + "silvus", + "silvus-international-opportunites", + "similarweb", + "simplesense", + "simpletechnologysolutions", + "simplextrading", + "simplifed", + "simplisafe", + "simpluris", + "simpplr", + "simtrabps", + "singlestore", + "singlestore-linkedin", + "sirenopt", + "sirum", + "sixfold", + "sixgeninc", + "sixspeed", + "sixthstreet", + "skedda", + "skildai-careers", + "skilledwoundcare", + "skinlaundry", + "skyepointdecisionsinc", + "skylighthq", + "skyryse", + "skysafe", + "sleepdoctor", + "slingshotaerospace", + "slingshotbiosciences", + "smaamerica", + "smallgirlspr2", + "smartasset", + "smartbear", + "smarterdx", + "smarterdxprivate", + "smartling", + "smartlyio", + "smartrent", + "smartsheet", + "smartypantsvitamins", + "smavagmbh", + "smithrx", + "snapmobileinc", + "snorkelai", + "snowcompanies", + "snsone", + "soci", + "socialfinance", + "sociallabsa", + "socialscienceresearchcouncil", + "socket", + "solarisbank", + "soldejaneiro", + "soldejaneirointernship", + "solidpower", + "sollishealth", + "solmentalhealth", + "soloioinc", + "solutions", + "sonatus", + "sonderaustralia", + "sonicwall", + "sonobello", + "sonyinteractiveentertainmentglobal", + "sonymusicasiacareers", + "sonymusiccanada", + "sonymusiccareersafrica", + "sonymusiccareersfrance", + "sonymusicentertainment", + "sonypicturesanimation", + "sonypicturesimageworks", + "soraunion", + "sorcero", + "sothebys", + "soundagriculture", + "soundcloud71", + "sourcegraph91", + "sourcemeridian", + "southernpovertylawcenter", + "southwesttitleloans", + "sovrn", + "spacekinetic", + "spaceship", + "spacex", + "spacexglobal", + "sparetech", + "sparkadvisors", + "sparkfund", + "spauldingridge", + "spcareers", + "specialty1", + "specterops", + "spectrumvascular", + "speechify", + "spinnakersupport", + "splashfinancial", + "splice", + "splitero", + "sportandspinephysicaltherapy", + "spothopper", + "spotme", + "spotter", + "springboard", + "springboardmentors", + "springfertility", + "springhealth66", + "sprintersportses", + "spsnorthamerica", + "spycloud", + "spyretherapeutics", + "squishable", + "stablekernel", + "stackadapt", + "stackav", + "stackblitz", + "stackcommerce", + "stackline", + "standardmetrics", + "stanley1913-us", + "stannesbelfieldschool", + "star-catcher", + "starburst", + "starcloud", + "starfaceworld", + "starfishneuroscience", + "starrez", + "startale", + "startcampus", + "stateroadah", + "steadfasthealth", + "stemhealthcare", + "sterlingtonpllc", + "sti", + "stirlingpdf", + "stockx", + "stone", + "stonekite", + "stonepatrocina", + "storycannabis", + "strandtherapeutics", + "stratacareers", + "stratainformationgroup", + "strategichr", + "strategicprojectpartners", + "stratolaunch", + "stressfree", + "striiminc", + "strike", + "stripe", + "strivehealth", + "strivepharmacy", + "striveworks", + "strongpointpartners", + "stubhubinc", + "studiokraftonboard", + "studsinc", + "studycontractors", + "stylusmedicine", + "submittable", + "subsplash", + "successacademycharterschool", + "successkpiinc", + "sudstop", + "summer", + "summitonevanderbilt", + "summittherapeutics", + "sumofus", + "sumologic", + "sumup", + "sunnyside", + "sunrise", + "sunset", + "suntimes", + "supergoop", + "supernal", + "superset", + "supersod", + "supplyhouse", + "supportingstrategies", + "surefirecyber", + "survata", + "surveymonkey", + "sustainabletalent", + "sustainablewestchester", + "sustainment", + "svetness", + "swanloveland", + "swayable", + "swellmedia", + "swiftsolar", + "synack", + "synacksrt", + "synaptrixlabs", + "sypnewsitetest", + "syskahennessy", + "systemstechnologyresearch", + "tactilemedical", + "tailscale", + "takealotcom", + "takealotgroup", + "taketwo", + "talkdesk2", + "talkspace", + "talkspacepsychiatry", + "talkspacetherapist", + "tandemlaunch", + "tandemmoneylimited", + "tanium", + "tankww", + "tapestryenergy", + "taskrabbit", + "tastylive", + "tastytrade", + "tatari", + "taxbit", + "taxvalet", + "tazewellpikeanimalclinic", + "teachinglab", + "teads1", + "teague", + "teamlfg", + "teammate", + "teammobot", + "teampicnic", + "teamrubicon", + "tebra", + "techholding", + "techstars57", + "tecovas", + "tefron", + "tegnainc", + "tekion", + "tekmetric", + "telnyx54", + "temporaltechnologies", + "temus", + "tenableinc", + "teneolinkedin", + "tennesseetitleloansinc", + "tenon", + "tenstorrent", + "tenstorrentuniversity", + "tenstreet", + "teravision", + "terraclear", + "terranorbitalcorporation", + "terzo", + "tesseratherapeutics", + "testlio", + "testnisc", + "texasairsystems", + "texascartitleandpaydayloanservicesinc", + "texaschillersystemsasp", + "textus", + "thalamusgme", + "thanx", + "thatlot", + "thatsnomoonentertainment", + "thealleninstitute", + "thebaltimorebanner", + "thebrattlegroup", + "thechempetitivegroupllc", + "thedoulanetwork", + "thedutchie", + "theeconomistgroup", + "theeverycompany", + "thefarmersdog", + "thefloridapanthers", + "thefork", + "thegialliancemanagementllccompany", + "thehealthmanagementacademy", + "thehutgroup", + "theiconic", + "theknotworldwide", + "theloomisagency", + "themaritimeaquarium", + "thematherevanston", + "themichaeljfoxfoundation", + "themjcos", + "themotleyfool", + "themuseumofscience", + "thena", + "thenewyorktimes", + "thenuclearcompany", + "theoncologyinstitute", + "theorchard", + "theplaceforchildrenwithautism", + "theplanningshopus", + "thesciongroupllc", + "thesiscareers", + "thesocialhub", + "thetradedesk", + "thevirtussolution", + "thevitacococompany", + "theweathercompany", + "thinkacademymy", + "thinkacademyus", + "thinkingmachines", + "thinkmarkets", + "thirdlove", + "thirdwaveautomation", + "thomasdoor", + "thomasvillechildcare", + "thoropass", + "thoughtworksreferral", + "threatlocker", + "thrivemarket", + "thtbc", + "tia", + "tide", + "tigera", + "tigergraph", + "tines", + "tintai", + "tippingpointcommunity", + "tlatechinc", + "tmc", + "toast", + "toastmastersinternational", + "togetherai", + "tollbit", + "tomofunfurbo", + "tomorrow", + "tomorrowhealth", + "toogoodtogo", + "toojaysdeli", + "topsort", + "topsteptrader", + "torcrobotics", + "torq", + "toshibaglobalcommercesolutions", + "totalrecon", + "totusmedicines", + "towerresearchcapital", + "tpcengineeringholdingsllc", + "tpgcareers", + "tpreducationllc", + "trace3", + "traegergrills", + "transactlyconnect", + "transcendinc", + "transcendtherapeutics", + "transfergo", + "transmarketgroup", + "trase", + "traveledgenetwork", + "treasuryprime", + "treelinebiosciences", + "tribalscale", + "trident", + "trigild", + "trinityparktalent", + "tripadvisor", + "tripledotstudios", + "triplewhale", + "triumvirateenvironmental", + "trivelta", + "triviumpoint", + "trove", + "trovohealth", + "trueanomalyinc", + "truebill", + "truecaller", + "truemedia", + "trufflesecurity", + "trustautomation", + "trustbank", + "trustwill", + "truveta", + "tsg", + "ttcglobal", + "tubescience-labs", + "tubescience52", + "tubitv", + "tucows", + "tudorgroup", + "turbineone", + "turbotenant", + "turing", + "turnkeycareers", "twilio", - "lakefieldveterinarygroup" + "twinhealth", + "twistbioscience", + "twitch", + "twosixtechnologies", + "typeface", + "typeform", + "tysonmendesllp", + "uareai", + "uasi", + "uberfreight", + "ubiquiti", + "ubiquitygp", + "udacity", + "udemybedi", + "ujet", + "ultimagenomics", + "umaeducationinc", + "umistone", + "unanet", + "unbounce", + "undercontrolroboticsinc", + "underdogfantasy", + "understood", + "understoodcare", + "unframe", + "unisonhomeownershipinvestors", + "unispace", + "uniswapfoundation", + "unitedfirm", + "unitedmasterstranslation", + "unitedmedia", + "uniteus", + "unknownworlds", + "unlock", + "unlockhealth", + "unrealsnacks", + "unybrands", + "up", + "upbound", + "upboundext", + "updater", + "upgrade", + "upkeep", + "upriteconstruction", + "upshop", + "upstack", + "upstart", + "upstatementrecruiting", + "upstreamusa", + "upwork", + "urbansky", + "urbansportsclub", + "ursamajor", + "usconec", + "usenourish", + "utahtitleloansinc", + "vacasa", + "vacationinc", + "vaco", + "vailhealthprivate", + "valaratomics", + "valohealth", + "valpro", + "valtech", + "vanmetre", + "vannevarlabs", + "vantagescore", + "vardaspace", + "varicent", + "vast", + "vaticlabs", + "vaxcyte", + "vectara", + "veeamsoftware", + "vegaamericas", + "veir", + "velocityelectronics", + "venncity", + "ventureglobal", + "veocorporatecareers", + "veracyte", + "verainstituteofjustice", + "veranahealth", + "veratherapeuticsinc", + "vercel", + "verisign", + "veristainc", + "veritasvetpartners", + "verkada", + "verkada2", + "verramobility", + "versaterm", + "verse", + "versprite", + "verstela", + "verve", + "vesalius", + "vestmark", + "vestwell", + "veterinaryemergencygroupst", + "veterinaryemergencyservices", + "veterinarypracticepartners", + "vetevolve", + "vettoai", + "vgw", + "vgw-canada", + "vgw-eu", + "via", + "via99", + "viamrobotics", + "vianttechnology", + "vibesllc", + "vikingglobalinvestors", + "vipvermontinformationprocessing2", + "viralnation", + "virbiotechnologyinc", + "virginbetsa", + "virtru", + "virtu", + "viseai", + "visia", + "visiersolutionsinc", + "visitingmedia", + "visualconcepts", + "vitalvoicesglobalpartnership", + "vitoriainternational", + "vitta", + "vivcourtevents", + "vivvi", + "vixxo", + "vmax", + "vmlenterprisesolutions", + "vogliodigitalmarketing", + "volastratherapeutics", + "vonage", + "vooban", + "vorbiopharma", + "voxmedia", + "voyagertechnologiesinc", + "voyagertherapeutics", + "vpawashington", + "vsapartners", + "vscfiresecurityinc", + "vsco39", + "vtex", + "vulcanelements", + "vulncheck", + "vynamic", + "vynyl", + "wakam", + "waldensecurity", + "wallapop", + "walleyecapital-external-students", + "wallstreetprep", + "wargamingen", + "warp", + "wasabi", + "waterloocoop", + "watershed", + "waverlyadvisorsllc", + "waymark", + "waymo", + "weave", + "webershandwick", + "webflow", + "wedgewoodpharmacy", + "weedmaps77", + "weee", + "weinsteinproperties", + "weissassetmanagement", + "welbehealth", + "wellist", + "wellsaidlabs", + "wellthy-care-network", + "weploy", + "westcancercenter", + "westmonroe4", + "wettermarkkeith", + "wfclainc", + "whalarinc", + "wheely", + "whisperaero", + "whitewatermidstream", + "whogivesacrap", + "whop", + "wight", + "wikimedia", + "wildalaskancompany", + "wildcardcreativegroup", + "wildlifestudios", + "wilsonelser", + "wilsonelserattorneys", + "wimanagementllc", + "wing", + "wingspan", + "winhomeinspection", + "winnerscirclegroupoftexas1", + "wisconsinautotitleloansinc", + "wisetack", + "withcoverage", + "withmeinc", + "wizardcommerce", + "wizixtechnologygroupinc", + "wolt", + "wonderschool", + "wongdoody", + "woo", + "wooga", + "woolpert", + "workatbackbase", + "workato", + "workera", + "workhelix", + "workithealth", + "workleap", + "workleapfr", + "workoverseas", + "workrightnw", + "workshop", + "workstream", + "workwize", + "world4822stri986des", + "worldlabs", + "worldquant", + "wovencare", + "wpp", + "wppmedia", + "wrike", + "wundercapital", + "ww", + "wwclinic", + "xai", + "xairatherapeutics", + "xantium", + "xapo61", + "xdinizioengage", + "xealth", + "xebiaapac", + "xebiacee", + "xebiausa", + "xendit", + "xohealthinc", + "xometry", + "xometryeurope", + "xometryturkey", + "xpengmotors", + "xpinc", + "xsolisinc", + "xtxmarketstechnologies", + "yalochatinc", + "yerbamadre", + "yext", + "yipitdata", + "yipitdatajobs", + "ylopo", + "yoodliinc", + "youcom", + "yugabyte", + "yurtsai", + "zafinlabsamericasinc", + "zam", + "zambold", + "zarminalihealth", + "zeffy", + "zenbusiness", + "zencoder", + "zengrc", + "zennioptical", + "zenoti", + "zephyrhome", + "zerostudios", + "zerotothree", + "zetachain", + "zetaglobal", + "zinnia", + "zinnov", + "zipcolimited", + "ziprecruiter", + "ziro", + "zocalohealth", + "zocdoc", + "zone5technologies", + "zonecompanysoftwareconsultingllc", + "zoo", + "zora", + "zscaler", + "zubiad", + "zuora", + "zupinnovation", + "zyngacareers", + "zyngaearlycareers" ] diff --git a/scraper-go/internal/interfaces/inhireTenants.json b/scraper-go/internal/interfaces/inhireTenants.json new file mode 100644 index 0000000..6fa0120 --- /dev/null +++ b/scraper-go/internal/interfaces/inhireTenants.json @@ -0,0 +1,3474 @@ +[ + { + "slug": "180seguros", + "tenantName": "180 Seguros", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "2future", + "tenantName": "2Future", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "3cplusnow", + "tenantName": "Grupo 3C", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "5ccompany", + "tenantName": "5Ccompany", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "aarin", + "tenantName": "Aarin Techfin", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "adias", + "tenantName": "A.Dias", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "advbox", + "tenantName": "ADVBOX - Software Jurídico", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "adyen", + "tenantName": "Adyen", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "agrosearch", + "tenantName": "AGROSearch", + "jobsCount": 43, + "listCompany": null, + "sampleJobs": 43 + }, + { + "slug": "agrotools", + "tenantName": "Agrotools", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "aiko", + "tenantName": "Aiko", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "aisolutions", + "tenantName": "AI Solutions", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "alana", + "tenantName": "Instituto Alana", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "alelo", + "tenantName": "Alelo", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "algar", + "tenantName": "Grupo Algar", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "alice", + "tenantName": "Alice", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "alicerce", + "tenantName": "Alicerce", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "alinemainericonsultoria", + "tenantName": "Aline Maineri Consultoria RH (TRD)", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "allintra", + "tenantName": "Allintra", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "alloyal", + "tenantName": "Alloyal", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "altoqi", + "tenantName": "AltoQi", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "alume", + "tenantName": "Alume", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "alun", + "tenantName": "FIAP/Alura/PM3", + "jobsCount": 24, + "listCompany": null, + "sampleJobs": 24 + }, + { + "slug": "alura-fiap-pm3", + "tenantName": "FIAP", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "amaro", + "tenantName": "AMARO", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "amcom", + "tenantName": "AMcom", + "jobsCount": 29, + "listCompany": null, + "sampleJobs": 29 + }, + { + "slug": "appfacilita", + "tenantName": "App Facilita", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "appmax", + "tenantName": "Appmax", + "jobsCount": 24, + "listCompany": null, + "sampleJobs": 24 + }, + { + "slug": "aprix", + "tenantName": "Aprix", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "arquivei", + "tenantName": "Arquivei", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "asa", + "tenantName": "ASA", + "jobsCount": 32, + "listCompany": null, + "sampleJobs": 32 + }, + { + "slug": "ascasconsultoria", + "tenantName": "ASCAS", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "ascsolutions", + "tenantName": "ASC Solutions", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "asper", + "tenantName": "Asper", + "jobsCount": 31, + "listCompany": null, + "sampleJobs": 31 + }, + { + "slug": "ateliware", + "tenantName": "Ateliware", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "atip", + "tenantName": "aTip", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "atlantico", + "tenantName": "Atlântico", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "atlastechnol", + "tenantName": "Atlas", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "autoglass", + "tenantName": "Grupo Autoglass", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "autom", + "tenantName": "Autom", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "automind", + "tenantName": "Automind", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "autopass", + "tenantName": "Autopass", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "auvotecnologia", + "tenantName": "Auvo Tecnologia", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "avel", + "tenantName": "Ável", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "avenue", + "tenantName": "Avenue", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "avivatec", + "tenantName": "Avivatec", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "azos", + "tenantName": "Azos", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "bancopine", + "tenantName": "Banco Pine", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "bancotoyota", + "tenantName": "Banco Toyota do Brasil", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "bankme", + "tenantName": "Bankme.", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "banni", + "tenantName": "Banni", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "barrosadvogados", + "tenantName": "Barros & Advogados", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "bemobi", + "tenantName": "Bemobi", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "benup", + "tenantName": "benup", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "bequestdigital", + "tenantName": "Bequest Academy", + "jobsCount": 32, + "listCompany": null, + "sampleJobs": 32 + }, + { + "slug": "bernhoeft", + "tenantName": "Bernhoeft", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "betaonline", + "tenantName": "BETA ONLINE", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "betha", + "tenantName": "Betha", + "jobsCount": 18, + "listCompany": null, + "sampleJobs": 18 + }, + { + "slug": "betmgm", + "tenantName": "BetMGM Brasil", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "bigdata", + "tenantName": "Big Data", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "bionexo", + "tenantName": "Bionexo", + "jobsCount": 22, + "listCompany": null, + "sampleJobs": 22 + }, + { + "slug": "bipa", + "tenantName": "Bipa", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "bixtecnologia", + "tenantName": "BIX Tecnologia - Consultoria de Dados", + "jobsCount": 20, + "listCompany": null, + "sampleJobs": 20 + }, + { + "slug": "blumeohaagua", + "tenantName": "Blume Alimentos | OH! A Água", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "board", + "tenantName": "Board", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "bradesco", + "tenantName": "Bradesco", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "brangemedia", + "tenantName": "Brange Media", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "brasilparalelo", + "tenantName": "Brasil Paralelo", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "breezyseguros", + "tenantName": "Breezy Seguros", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "bridge", + "tenantName": "Bridge & Co.", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "brivia", + "tenantName": "Brivia", + "jobsCount": 40, + "listCompany": null, + "sampleJobs": 40 + }, + { + "slug": "brmediagroup", + "tenantName": "BR Media Group", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "brms", + "tenantName": "BR MotorSport", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "brq", + "tenantName": "BRQ", + "jobsCount": 80, + "listCompany": null, + "sampleJobs": 80 + }, + { + "slug": "brqdigitalsolutions", + "tenantName": "BRQ", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "cactus", + "tenantName": "Cactus Gaming", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "cadmus", + "tenantName": "Cadmus", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "caf", + "tenantName": "CAF", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "capef", + "tenantName": "CAPEF", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "carbigdata", + "tenantName": "Carbigdata", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "cardapioweb", + "tenantName": "Cardápio Web", + "jobsCount": 23, + "listCompany": null, + "sampleJobs": 23 + }, + { + "slug": "carreiras", + "tenantName": "InHire", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "casabugre", + "tenantName": "Casa Bugre", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "cashme", + "tenantName": "CashMe", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "catarse", + "tenantName": "Catarse", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "cayena", + "tenantName": "Cayena", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "celcoin", + "tenantName": "Celcoin", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "celero", + "tenantName": "Celero", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "centraldevagas", + "tenantName": "Trinuscool", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "cerc", + "tenantName": "CERC", + "jobsCount": 19, + "listCompany": null, + "sampleJobs": 19 + }, + { + "slug": "certta", + "tenantName": "CAF", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "cesconbarrieu", + "tenantName": "Cescon Barrieu", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "chatguru", + "tenantName": "ChatGuru", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "cidadaniaja", + "tenantName": "Cidadania Já", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "cielo", + "tenantName": "Cielo", + "jobsCount": 22, + "listCompany": null, + "sampleJobs": 22 + }, + { + "slug": "cinga", + "tenantName": "Cinga Tech", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "cirandacultural", + "tenantName": "Grupo Ciranda Cultural", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "clam", + "tenantName": "Clam", + "jobsCount": 48, + "listCompany": null, + "sampleJobs": 48 + }, + { + "slug": "clavis", + "tenantName": "Clavis", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "clickbus", + "tenantName": "ClickBus", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "coaktion", + "tenantName": "coaktion", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "cobli", + "tenantName": "Cobli", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "cogna", + "tenantName": "Cogna Educação", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "construtoraelevacao", + "tenantName": "Construtora Elevação", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "construtorapatriani", + "tenantName": "Construtora Patriani", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "contaazul", + "tenantName": "Conta Azul", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "contabilizei", + "tenantName": "Contabilizei", + "jobsCount": 63, + "listCompany": null, + "sampleJobs": 63 + }, + { + "slug": "contasimples", + "tenantName": "Conta Simples", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "convenia", + "tenantName": "Convênia", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "conxconstrutora", + "tenantName": "Conx", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "cora", + "tenantName": "Cora", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "credaluga", + "tenantName": "CredAluga", + "jobsCount": 20, + "listCompany": null, + "sampleJobs": 20 + }, + { + "slug": "credipronto", + "tenantName": "CrediPronto", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "creditas", + "tenantName": "Creditas", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "crmbonus", + "tenantName": "CRMBonus", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "crown", + "tenantName": "Crown", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "csptecnologia", + "tenantName": "CSP Tecnologia", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "csu", + "tenantName": "CSU", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "ctctech", + "tenantName": "CTC", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "cubos", + "tenantName": "Cubos Tecnologia", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "cubosacademy", + "tenantName": "Cubos Academy", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "dafiti", + "tenantName": "Dafiti", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "dartz", + "tenantName": "Dartz Ideal Recruitment", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "db1", + "tenantName": "DB1 Group", + "jobsCount": 35, + "listCompany": null, + "sampleJobs": 35 + }, + { + "slug": "dbservices", + "tenantName": "DBServices", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "deal", + "tenantName": "Deal Group", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "dealersites", + "tenantName": "Dealer Space", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "deloitte", + "tenantName": "Deloitte", + "jobsCount": 227, + "listCompany": null, + "sampleJobs": 227 + }, + { + "slug": "demo", + "tenantName": "Demo", + "jobsCount": 246, + "listCompany": null, + "sampleJobs": 246 + }, + { + "slug": "dhg", + "tenantName": "Ocean Health", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "digix", + "tenantName": "Digix", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "dock", + "tenantName": "Dock", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "dotgroup", + "tenantName": "DOT Group", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "dqrtech", + "tenantName": "DQR Tech", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "dtidigital", + "tenantName": "dti digital", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "dtlabs", + "tenantName": "dtLabs", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "duxcompany", + "tenantName": "DUX Company", + "jobsCount": 35, + "listCompany": null, + "sampleJobs": 35 + }, + { + "slug": "ebanx", + "tenantName": "EBANX", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "eclipseworks", + "tenantName": "Eclipseworks", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "ecore", + "tenantName": "e-Core", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "ecx", + "tenantName": "ecx", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "edenred", + "tenantName": "Edenred", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "edge", + "tenantName": "EDGE", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "eiti", + "tenantName": "Eiti Gestão de Ti", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "elements", + "tenantName": "Elements", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "elogroup", + "tenantName": "EloGroup", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "emergeventures", + "tenantName": "Emerge Ventures", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "enjoei", + "tenantName": "Enjoei", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "enzrossi", + "tenantName": "EnzRossi", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "eqi", + "tenantName": "EQI", + "jobsCount": 32, + "listCompany": null, + "sampleJobs": 32 + }, + { + "slug": "erural", + "tenantName": "erural", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "escoladnc", + "tenantName": "Escola DNC", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "essenceit", + "tenantName": "Essence", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "essencialnutricao", + "tenantName": "Essencial Nutrição", + "jobsCount": 176, + "listCompany": null, + "sampleJobs": 176 + }, + { + "slug": "estapar-trial", + "tenantName": "Estapar", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "estrelabet", + "tenantName": "EstrelaBet", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "eveo", + "tenantName": "EVEO", + "jobsCount": 18, + "listCompany": null, + "sampleJobs": 18 + }, + { + "slug": "exa", + "tenantName": "EXA", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "exati", + "tenantName": "Exati", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "extremegroup", + "tenantName": "Extreme Digital Solutions", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "eyxo", + "tenantName": "Eyxo Content Co.", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "facilitapay", + "tenantName": "FacilitaPay", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "falconi", + "tenantName": "Falconi", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "farmtech", + "tenantName": "Farmtech", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "fbiz", + "tenantName": "Fbiz", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "fcamara", + "tenantName": "FCamara", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "fiberhome", + "tenantName": "FiberHome Brasil", + "jobsCount": 29, + "listCompany": null, + "sampleJobs": 29 + }, + { + "slug": "finallevel", + "tenantName": "Final Level", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "findup", + "tenantName": "FindUP", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "fintalk", + "tenantName": "Fintalk", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "fitenergia", + "tenantName": "FIT Energia", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "fitting", + "tenantName": "Fitting", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "flash", + "tenantName": "Flash", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "flemingeducacao", + "tenantName": "Fleming Educação", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "floraconsultoria", + "tenantName": "floraconsultoria", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "fluke", + "tenantName": "fluke", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "flutterbrazil", + "tenantName": "Flutter Brazil", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "fmu", + "tenantName": "FMU", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "foxbit", + "tenantName": "Foxbit", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "frameworkdigital", + "tenantName": "Framework", + "jobsCount": 38, + "listCompany": null, + "sampleJobs": 38 + }, + { + "slug": "fretadao", + "tenantName": "Fretadão", + "jobsCount": 19, + "listCompany": null, + "sampleJobs": 19 + }, + { + "slug": "frete", + "tenantName": "frete.com", + "jobsCount": 30, + "listCompany": null, + "sampleJobs": 30 + }, + { + "slug": "fricon", + "tenantName": "FRICON Brasil", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "fsbholding", + "tenantName": "FSB Holding", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "gauge", + "tenantName": "Gauge", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "gcservicos", + "tenantName": "GC Serviços", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "genyo", + "tenantName": "Genyo", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "geo3d", + "tenantName": "GEO3D ENGENHARIA DE MAPEAMENTO", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "gerdau", + "tenantName": "Gerdau", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "gex", + "tenantName": "GEX", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "globalsystem", + "tenantName": "Global System", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "gobravo", + "tenantName": "Go Bravo", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "goflow", + "tenantName": "GoFlow Consultoria", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "gogroup", + "tenantName": "Gogroup", + "jobsCount": 62, + "listCompany": null, + "sampleJobs": 62 + }, + { + "slug": "grupoagcapital", + "tenantName": "Grupo AG Capital", + "jobsCount": 28, + "listCompany": null, + "sampleJobs": 28 + }, + { + "slug": "grupoboticario", + "tenantName": "Grupo Boticário", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "grupocentral", + "tenantName": "Grupo Central", + "jobsCount": 19, + "listCompany": null, + "sampleJobs": 19 + }, + { + "slug": "grupocomolatti", + "tenantName": "Grupo Comolatti", + "jobsCount": 271, + "listCompany": null, + "sampleJobs": 271 + }, + { + "slug": "grupoeximio", + "tenantName": "Grupo Eximio", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "grupofoodnation", + "tenantName": "Grupo Food Nation", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "grupofour", + "tenantName": "Grupo Four", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "grupogcb", + "tenantName": "GCB", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "grupogcbinvestimentos", + "tenantName": "Grupo GCB", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "grupogopwr", + "tenantName": "GOPWR", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "grupoguiainvest", + "tenantName": "GuiaInvest", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "grupogvc", + "tenantName": "GVC", + "jobsCount": 27, + "listCompany": null, + "sampleJobs": 27 + }, + { + "slug": "grupolider", + "tenantName": "Grupo Lider", + "jobsCount": 181, + "listCompany": null, + "sampleJobs": 181 + }, + { + "slug": "grupolos", + "tenantName": "Grupo LOS", + "jobsCount": 33, + "listCompany": null, + "sampleJobs": 33 + }, + { + "slug": "grupomedeiros", + "tenantName": "Dunorte", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "grupopermaneo", + "tenantName": "Grupo Permaneo", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "grupoprotege", + "tenantName": "Grupo Protege", + "jobsCount": 204, + "listCompany": null, + "sampleJobs": 204 + }, + { + "slug": "gruposaga", + "tenantName": "Grupo Saga", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "gruposrm", + "tenantName": "SRM", + "jobsCount": 28, + "listCompany": null, + "sampleJobs": 28 + }, + { + "slug": "grupotaking", + "tenantName": "Grupo Taking", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "grupothink", + "tenantName": "Think IT", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "guaracamp", + "tenantName": "Guaracamp", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "gx2", + "tenantName": "Vem ser GX2", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "harpia", + "tenantName": "Harpia", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "heads", + "tenantName": "Heads", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "hh-escolaconquer", + "tenantName": "Escola Conquer", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "hh-lello", + "tenantName": "Lello Condomínios", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "hh-uello", + "tenantName": "Uello", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "hiltonbrasil", + "tenantName": "Hilton Brasil", + "jobsCount": 88, + "listCompany": null, + "sampleJobs": 88 + }, + { + "slug": "icatu", + "tenantName": "Icatu Seguros", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "icatuseguros", + "tenantName": "Icatu Seguros", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "icherry", + "tenantName": "icherry", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "iconit", + "tenantName": "ICON Solutions do Brasil", + "jobsCount": 62, + "listCompany": null, + "sampleJobs": 62 + }, + { + "slug": "idg", + "tenantName": "IDG", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "idwall", + "tenantName": "idwall", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "ifood", + "tenantName": "iFood", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "ihunters", + "tenantName": "iHunters", + "jobsCount": 22, + "listCompany": null, + "sampleJobs": 22 + }, + { + "slug": "imaflora", + "tenantName": "Imaflora", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "impulsogov", + "tenantName": "ImpulsoGov", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "indicium", + "tenantName": "Indicium", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "industriasanhembi", + "tenantName": "Anhembi", + "jobsCount": 20, + "listCompany": null, + "sampleJobs": 20 + }, + { + "slug": "infleet", + "tenantName": "Infleet", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "infotecbrasil", + "tenantName": "Infotec Brasil", + "jobsCount": 68, + "listCompany": null, + "sampleJobs": 68 + }, + { + "slug": "innoveahub", + "tenantName": "iNNOVEA HUB", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "insiderstore", + "tenantName": "Insider Store", + "jobsCount": 50, + "listCompany": null, + "sampleJobs": 50 + }, + { + "slug": "instacarro", + "tenantName": "InstaCarro", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "intera", + "tenantName": "Intera", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "intertechne", + "tenantName": "Intertechne", + "jobsCount": 87, + "listCompany": null, + "sampleJobs": 87 + }, + { + "slug": "involves", + "tenantName": "Involves", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "iorq", + "tenantName": "IORQ", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "ipnet", + "tenantName": "IPNET by Vivo", + "jobsCount": 20, + "listCompany": null, + "sampleJobs": 20 + }, + { + "slug": "isystems", + "tenantName": "iSystems", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "iteris", + "tenantName": "Iteris", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "itriad", + "tenantName": "ITRIAD", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "iventis", + "tenantName": "IVENTIS", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "jae", + "tenantName": "Jaé", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "jassy", + "tenantName": "J.Assy", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "jeitto", + "tenantName": "Jeitto", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "jota", + "tenantName": "JOTA", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "justos", + "tenantName": "Justos", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "kamino", + "tenantName": "Kamino", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "kanastra", + "tenantName": "Kanastra", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "kife", + "tenantName": "Kife", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "kiwify", + "tenantName": "Kiwify", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "klavi", + "tenantName": "klavi", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "kobe", + "tenantName": "Kobe Apps", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "kooperecooperativa", + "tenantName": "Koopere", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "kpmg", + "tenantName": "KPMG", + "jobsCount": 178, + "listCompany": null, + "sampleJobs": 178 + }, + { + "slug": "kstack", + "tenantName": "Kstack", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "labiexames", + "tenantName": "Labi Exames", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "labisaude", + "tenantName": "Labi Saúde", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "laboclin", + "tenantName": "Laboclin", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "labwarebrasil", + "tenantName": "LabWare", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "lastlink", + "tenantName": "Lastlink", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "leaderetalent", + "tenantName": "Leader & Talent", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "lello", + "tenantName": "Lello", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "lemonenergia", + "tenantName": "Lemon Energia", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "letrus", + "tenantName": "Letrus", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "lfaadvogados", + "tenantName": "LFA Advogados", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "liber", + "tenantName": "Líber", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "liggatelecom", + "tenantName": "Ligga", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "lightfarmstudios", + "tenantName": "Lightfarm", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "limaconsulting", + "tenantName": "Lima Consulting Group", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "link2pump", + "tenantName": "CTA Smart", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "linx", + "tenantName": "Linx", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "listo", + "tenantName": "Listo", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "livelo", + "tenantName": "Livelo", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "livemode", + "tenantName": "LiveMode", + "jobsCount": 24, + "listCompany": null, + "sampleJobs": 24 + }, + { + "slug": "livup", + "tenantName": "Liv Up", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "loft", + "tenantName": "Loft", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "loggi", + "tenantName": "Loggi", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "lotussolution", + "tenantName": "RH Lótus Solution", + "jobsCount": 20, + "listCompany": null, + "sampleJobs": 20 + }, + { + "slug": "luby", + "tenantName": "Luby", + "jobsCount": 25, + "listCompany": null, + "sampleJobs": 25 + }, + { + "slug": "lumine", + "tenantName": "Lumine", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "luzandreyaconsultoria", + "tenantName": "Luzandreya Consultoria e Assessoria de RH", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "lwsa", + "tenantName": "LWSA", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "lyncas", + "tenantName": "Lyncas", + "jobsCount": 30, + "listCompany": null, + "sampleJobs": 30 + }, + { + "slug": "magazineluiza", + "tenantName": "Magazine Luiza", + "jobsCount": 21, + "listCompany": null, + "sampleJobs": 21 + }, + { + "slug": "magazord", + "tenantName": "Magazord", + "jobsCount": 23, + "listCompany": null, + "sampleJobs": 23 + }, + { + "slug": "magnasearch", + "tenantName": "Magna Executive Search", + "jobsCount": 27, + "listCompany": null, + "sampleJobs": 27 + }, + { + "slug": "mapahds", + "tenantName": "Mapa HDS", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "match-mt", + "tenantName": "match.mt", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "matera", + "tenantName": "Matera", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "mathgroup", + "tenantName": "MATH", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "mazzatech", + "tenantName": "mazzatech", + "jobsCount": 90, + "listCompany": null, + "sampleJobs": 90 + }, + { + "slug": "mb", + "tenantName": "Mercado Bitcoin", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "mediahero", + "tenantName": "Media Hero", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "medway", + "tenantName": "Medway", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "meliuz", + "tenantName": "Méliuz", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "melnick", + "tenantName": "Melnick", + "jobsCount": 25, + "listCompany": null, + "sampleJobs": 25 + }, + { + "slug": "mena", + "tenantName": "MENA", + "jobsCount": 35, + "listCompany": null, + "sampleJobs": 35 + }, + { + "slug": "midhaus", + "tenantName": "Midhaus", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "minertecnologia", + "tenantName": "Miner Tecnologia", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "mirum", + "tenantName": "Mirum", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "mjv", + "tenantName": "MJV INNOVATION", + "jobsCount": 52, + "listCompany": null, + "sampleJobs": 52 + }, + { + "slug": "mksolutions", + "tenantName": "MK Solutions", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "monkey", + "tenantName": "Monkey", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "motim", + "tenantName": "MOTIM", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "mottu", + "tenantName": "Mottu", + "jobsCount": 65, + "listCompany": null, + "sampleJobs": 65 + }, + { + "slug": "movecta", + "tenantName": "Movecta", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "mspestudios", + "tenantName": "VEM PRA TURMA", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "multi7", + "tenantName": "M7 Soluções Financeiras", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "mundiale", + "tenantName": "Mundiale", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "mutant", + "tenantName": "Mutant", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "neogrid", + "tenantName": "Neogrid", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "neoway", + "tenantName": "Neoway", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "nexaas", + "tenantName": "Nexaas", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "nibo", + "tenantName": "Nibo", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "ninecon", + "tenantName": "Ninecon", + "jobsCount": 66, + "listCompany": null, + "sampleJobs": 66 + }, + { + "slug": "nomadglobal", + "tenantName": "Nomad", + "jobsCount": 30, + "listCompany": null, + "sampleJobs": 30 + }, + { + "slug": "nts", + "tenantName": "NTS Brasil", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "nubank", + "tenantName": "Nubank", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "nuclea", + "tenantName": "Núclea", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "nuvemshop-tiendanube", + "tenantName": "Nuvemshop", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "nuvidio", + "tenantName": "Nuvidio", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "objective", + "tenantName": "Objective", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "oceandrop", + "tenantName": "Ocean Drop", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "octafy", + "tenantName": "Octafy", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "olgari-hipokee", + "tenantName": "OVAL", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "olist", + "tenantName": "Olist", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "oliveiraeantunes", + "tenantName": "Oliveira & Antunes Advogados Associados", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "openlabs", + "tenantName": "Open Labs", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "orbia", + "tenantName": "Orbia", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "orizon", + "tenantName": "Orizon", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "paipe", + "tenantName": "Paipe", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "pantheon", + "tenantName": "Pantheon", + "jobsCount": 109, + "listCompany": null, + "sampleJobs": 109 + }, + { + "slug": "paytrack", + "tenantName": "Paytrack", + "jobsCount": 24, + "listCompany": null, + "sampleJobs": 24 + }, + { + "slug": "pedraagroindustrial", + "tenantName": "Pedra Agroindustrial", + "jobsCount": 42, + "listCompany": null, + "sampleJobs": 42 + }, + { + "slug": "peers", + "tenantName": "Peers", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "people", + "tenantName": "People", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "petitrh", + "tenantName": "Petit.RH", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "pier", + "tenantName": "Pier Seguradora", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "piposaude", + "tenantName": "Pipo Saúde", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "pitang", + "tenantName": "Pitang", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "platformbuilders", + "tenantName": "Platform Builders", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "pmfo", + "tenantName": "Portofino Multi Family Office", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "poliedroeducacao", + "tenantName": "Poliedro Educação", + "jobsCount": 38, + "listCompany": null, + "sampleJobs": 38 + }, + { + "slug": "portal", + "tenantName": "Portal Nova", + "jobsCount": 25, + "listCompany": null, + "sampleJobs": 25 + }, + { + "slug": "pottencial", + "tenantName": "Pottencial", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "praso", + "tenantName": "Praso", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "pravaler", + "tenantName": "Pravaler", + "jobsCount": 41, + "listCompany": null, + "sampleJobs": 41 + }, + { + "slug": "premiersoft", + "tenantName": "Premiersoft", + "jobsCount": 38, + "listCompany": null, + "sampleJobs": 38 + }, + { + "slug": "principia", + "tenantName": "Principia", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "priner", + "tenantName": "Priner", + "jobsCount": 268, + "listCompany": null, + "sampleJobs": 268 + }, + { + "slug": "pris", + "tenantName": "Pris", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "prizmamidia", + "tenantName": "Prizma Mídia", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "procfit", + "tenantName": "Procfit", + "jobsCount": 13, + "listCompany": null, + "sampleJobs": 13 + }, + { + "slug": "programmers", + "tenantName": "Programmers - AI, Data & Software", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "prospertechtalents", + "tenantName": "Prosper Tech Talents", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "qeevo", + "tenantName": "Qeevo", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "qitech", + "tenantName": "QI Tech", + "jobsCount": 32, + "listCompany": null, + "sampleJobs": 32 + }, + { + "slug": "qive", + "tenantName": "Qive", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "querodelivery", + "tenantName": "Querodelivery", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "queroeducacao", + "tenantName": "Qeevo", + "jobsCount": 34, + "listCompany": null, + "sampleJobs": 34 + }, + { + "slug": "queropassagem", + "tenantName": "Quero Passagem", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "quicksoft", + "tenantName": "QuickSoft", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "radix", + "tenantName": "Radix", + "jobsCount": 107, + "listCompany": null, + "sampleJobs": 107 + }, + { + "slug": "rankmyapp", + "tenantName": "RankMyApp", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "rarolabs", + "tenantName": "Raro Labs", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "reag", + "tenantName": "REAG", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "reclameaqui", + "tenantName": "Reclame AQUI", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "reidopitaco", + "tenantName": "Rei do Pitaco", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "remessaonline", + "tenantName": "Remessa Online", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "rentbrella", + "tenantName": "Rentbrella", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "rentcars", + "tenantName": "Rentcars", + "jobsCount": 19, + "listCompany": null, + "sampleJobs": 19 + }, + { + "slug": "residclub", + "tenantName": "Resid Club", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "rhitmotech", + "tenantName": "Rhitmo Tech", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "riogaleao", + "tenantName": "RIOgaleão", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "rnp", + "tenantName": "Rede RNP", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "rpo-abinbev", + "tenantName": "AB InBev", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "rpo-brmediagroup", + "tenantName": "BR Media Group - RPO", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "rpo-contasimples", + "tenantName": "Conta Simples - RPO", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "rpo-olist", + "tenantName": "Olist", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "same-engenharia", + "tenantName": "Same Engenharia", + "jobsCount": 11, + "listCompany": null, + "sampleJobs": 11 + }, + { + "slug": "samsung", + "tenantName": "Samsung", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "sanar", + "tenantName": "Sanar", + "jobsCount": 27, + "listCompany": null, + "sampleJobs": 27 + }, + { + "slug": "sancorsegurosbrasil", + "tenantName": "Sancor Seguros", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "schulze", + "tenantName": "Schulze", + "jobsCount": 28, + "listCompany": null, + "sampleJobs": 28 + }, + { + "slug": "seazone", + "tenantName": "Seazone", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "segura", + "tenantName": "Segura®", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "segurossura", + "tenantName": "Seguros SURA", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "semantix", + "tenantName": "Semantix", + "jobsCount": 72, + "listCompany": null, + "sampleJobs": 72 + }, + { + "slug": "senhasegura", + "tenantName": "Segura®", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "seuestagio", + "tenantName": "Seu Estágio", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "seventh", + "tenantName": "Seventh", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "shapedigital", + "tenantName": "Shape Digital", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "sharepeoplehub", + "tenantName": "Share People Hub", + "jobsCount": 75, + "listCompany": null, + "sampleJobs": 75 + }, + { + "slug": "sicredi", + "tenantName": "Sicredi", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "sidia", + "tenantName": "Sidia", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "silicon", + "tenantName": "Silicon", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "simdigital", + "tenantName": "sim.digital", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "skeelo", + "tenantName": "Skeelo", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "skyone", + "tenantName": "Skyone", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "softvaro", + "tenantName": "Softvaro", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "solfacil", + "tenantName": "Solfácil", + "jobsCount": 22, + "listCompany": null, + "sampleJobs": 22 + }, + { + "slug": "solutis", + "tenantName": "Solutis", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "sotran", + "tenantName": "Sotran", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "stone", + "tenantName": "Stone", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "stoque", + "tenantName": "Stoque", + "jobsCount": 18, + "listCompany": null, + "sampleJobs": 18 + }, + { + "slug": "strm-music", + "tenantName": "Strm Music", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "sunne", + "tenantName": "Sunne", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "superfrete", + "tenantName": "SuperFrete", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "superlogica", + "tenantName": "Superlógica Tecnologias", + "jobsCount": 18, + "listCompany": null, + "sampleJobs": 18 + }, + { + "slug": "sylvamo", + "tenantName": "Sylvamo", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "sympla", + "tenantName": "Sympla", + "jobsCount": 46, + "listCompany": null, + "sampleJobs": 46 + }, + { + "slug": "talentt", + "tenantName": "Talentt", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "talentx", + "tenantName": "TalentX Digital", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "tc", + "tenantName": "TC", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "tecer", + "tenantName": "tecer", + "jobsCount": 25, + "listCompany": null, + "sampleJobs": 25 + }, + { + "slug": "techlead", + "tenantName": "Techlead", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "tecsul", + "tenantName": "Tecsul", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "tegra", + "tenantName": "Tegra", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "tela", + "tenantName": "Saúde do TEA", + "jobsCount": 208, + "listCompany": null, + "sampleJobs": 208 + }, + { + "slug": "telavita", + "tenantName": "Telavita", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "tembici", + "tenantName": "Tembici", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "tempest", + "tenantName": "Tempest", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "tera", + "tenantName": "Tera", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "thewolvescompany", + "tenantName": "The Wolves Company", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "thunderstecnologia", + "tenantName": "Thunders Tecnologia", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "ticket360", + "tenantName": "Ticket360", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "tinnova", + "tenantName": "Tinnova", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "toroinvestimentos", + "tenantName": "Toro Investimentos", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "totvs", + "tenantName": "TOTVS", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "townsq", + "tenantName": "TownSq", + "jobsCount": 9, + "listCompany": null, + "sampleJobs": 9 + }, + { + "slug": "tpgroup", + "tenantName": "TPGROUP", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "tqi", + "tenantName": "TQI", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "trampay", + "tenantName": "Trampay", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "transfero", + "tenantName": "Transfero", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "tray", + "tenantName": "Tray", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "trinca", + "tenantName": "Trinca", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "trinnus", + "tenantName": "Trinnus RH", + "jobsCount": 21, + "listCompany": null, + "sampleJobs": 21 + }, + { + "slug": "tripla", + "tenantName": "Tripla", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "turbi", + "tenantName": "Turbi", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "turbo", + "tenantName": "Turbo", + "jobsCount": 16, + "listCompany": null, + "sampleJobs": 16 + }, + { + "slug": "ubots", + "tenantName": "Ubots", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "ume", + "tenantName": "Ume", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "unico", + "tenantName": "Unico", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "unimar", + "tenantName": "Unimar", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "uninter", + "tenantName": "Grupo UNINTER", + "jobsCount": 27, + "listCompany": null, + "sampleJobs": 27 + }, + { + "slug": "unionit", + "tenantName": "Union IT", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "unitech", + "tenantName": "Unitech", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "upda", + "tenantName": "UPDA", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "upflux", + "tenantName": "UpFlux", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "v360", + "tenantName": "V360", + "jobsCount": 8, + "listCompany": null, + "sampleJobs": 8 + }, + { + "slug": "v4company", + "tenantName": "V4 Company", + "jobsCount": 1035, + "listCompany": null, + "sampleJobs": 1035 + }, + { + "slug": "vagasbyintera", + "tenantName": "Vagas by Intera", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "vagasconfidenciais2", + "tenantName": "Vagas Confidenciais2", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "vatto", + "tenantName": "Vatto", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "venturus", + "tenantName": "Venturus", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "verticalrh", + "tenantName": "VerticalRH", + "jobsCount": 6, + "listCompany": null, + "sampleJobs": 6 + }, + { + "slug": "vertigo", + "tenantName": "Vertigo Tecnologia", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "vhsys", + "tenantName": "vhsys", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "via", + "tenantName": "Via", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "vinta", + "tenantName": "Vinta", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "viseu", + "tenantName": "Viseu Secondment", + "jobsCount": 55, + "listCompany": null, + "sampleJobs": 55 + }, + { + "slug": "vitru", + "tenantName": "Vitru Educação", + "jobsCount": 360, + "listCompany": null, + "sampleJobs": 360 + }, + { + "slug": "vivopcd", + "tenantName": "Vivo (Telefônica Brasil)", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "vixtra", + "tenantName": "Vixtra", + "jobsCount": 4, + "listCompany": null, + "sampleJobs": 4 + }, + { + "slug": "vobi", + "tenantName": "Vobi", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "vocedm", + "tenantName": "vocedm", + "jobsCount": 44, + "listCompany": null, + "sampleJobs": 44 + }, + { + "slug": "vockan", + "tenantName": "Vockan", + "jobsCount": 1, + "listCompany": null, + "sampleJobs": 1 + }, + { + "slug": "volttaenergy", + "tenantName": "Voltta", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "vortx", + "tenantName": "Vórtx", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "votorantim", + "tenantName": "Votorantim", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "vrental", + "tenantName": "VRental", + "jobsCount": 10, + "listCompany": null, + "sampleJobs": 10 + }, + { + "slug": "warren", + "tenantName": "Warren", + "jobsCount": 3, + "listCompany": null, + "sampleJobs": 3 + }, + { + "slug": "wavin", + "tenantName": "Orbia Amanco Wavin", + "jobsCount": 64, + "listCompany": null, + "sampleJobs": 64 + }, + { + "slug": "wehandle", + "tenantName": "wehandle", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "wellz", + "tenantName": "Wellz by Wellhub", + "jobsCount": 2, + "listCompany": null, + "sampleJobs": 2 + }, + { + "slug": "westwing", + "tenantName": "Westwing Brasil", + "jobsCount": 32, + "listCompany": null, + "sampleJobs": 32 + }, + { + "slug": "wevy-cloud", + "tenantName": "Wevy", + "jobsCount": 7, + "listCompany": null, + "sampleJobs": 7 + }, + { + "slug": "willbank", + "tenantName": "will bank", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + }, + { + "slug": "windcraft", + "tenantName": "WINDCRAFT", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "winnin", + "tenantName": "Winnin", + "jobsCount": 5, + "listCompany": null, + "sampleJobs": 5 + }, + { + "slug": "wsut", + "tenantName": "Wilson Sons Offshore", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "yandeh", + "tenantName": "Yandeh", + "jobsCount": 15, + "listCompany": null, + "sampleJobs": 15 + }, + { + "slug": "yolo", + "tenantName": "Yolo Recruit", + "jobsCount": 12, + "listCompany": null, + "sampleJobs": 12 + }, + { + "slug": "zallpy", + "tenantName": "Zallpy Digital", + "jobsCount": 26, + "listCompany": null, + "sampleJobs": 26 + }, + { + "slug": "zappts", + "tenantName": "Zappts", + "jobsCount": 17, + "listCompany": null, + "sampleJobs": 17 + }, + { + "slug": "zerezes", + "tenantName": "Zerezes", + "jobsCount": 41, + "listCompany": null, + "sampleJobs": 41 + }, + { + "slug": "zig", + "tenantName": "Zig", + "jobsCount": 14, + "listCompany": null, + "sampleJobs": 14 + }, + { + "slug": "zup", + "tenantName": "Zup", + "jobsCount": 0, + "listCompany": null, + "sampleJobs": 0 + } +] \ No newline at end of file diff --git a/scraper-go/internal/interfaces/leverCompanies.json b/scraper-go/internal/interfaces/leverCompanies.json index c29746a..856f10a 100644 --- a/scraper-go/internal/interfaces/leverCompanies.json +++ b/scraper-go/internal/interfaces/leverCompanies.json @@ -1,36 +1,310 @@ [ - { "slug": "netlify", "name": "Netlify" }, - { "slug": "vercel", "name": "Vercel" }, - { "slug": "postman", "name": "Postman" }, - { "slug": "sentry", "name": "Sentry" }, - { "slug": "supabase", "name": "Supabase" }, - { "slug": "planetscale", "name": "PlanetScale" }, - { "slug": "render", "name": "Render" }, - { "slug": "railway", "name": "Railway" }, - { "slug": "flyio", "name": "Fly.io" }, - { "slug": "typeform", "name": "Typeform" }, - { "slug": "hotjar", "name": "Hotjar" }, - { "slug": "contentful", "name": "Contentful" }, - { "slug": "algolia", "name": "Algolia" }, - { "slug": "front", "name": "Front" }, - { "slug": "segment", "name": "Segment" }, - { "slug": "linear", "name": "Linear" }, - { "slug": "retool", "name": "Retool" }, - { "slug": "loom", "name": "Loom" }, - { "slug": "superhuman", "name": "Superhuman" }, - { "slug": "pitch", "name": "Pitch" }, - { "slug": "mercury", "name": "Mercury" }, - { "slug": "alchemy", "name": "Alchemy" }, - { "slug": "chainlink", "name": "Chainlink" }, - { "slug": "openzeppelin", "name": "OpenZeppelin" }, - { "slug": "replicate", "name": "Replicate" }, - { "slug": "twelve-labs", "name": "Twelve Labs" }, - { "slug": "buffer", "name": "Buffer" }, - { "slug": "gitlab", "name": "GitLab" }, - { "slug": "automattic", "name": "Automattic" }, - { "slug": "wise", "name": "Wise" }, - { "slug": "klarna", "name": "Klarna" }, - { "slug": "glossier", "name": "Glossier" }, - { "slug": "patreon", "name": "Patreon" }, - { "slug": "robinhood", "name": "Robinhood" } + { + "site": "amicustherapeutics", + "companyName": "Amicus Therapeutics", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/amicustherapeutics", + "apiUrl": "https://api.eu.lever.co/v0/postings/amicustherapeutics?mode=json" + }, + { + "site": "blackshark", + "companyName": "Blackshark.ai", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/blackshark", + "apiUrl": "https://api.eu.lever.co/v0/postings/blackshark?mode=json" + }, + { + "site": "byborgenterprises", + "companyName": "Byborg Enterprises", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/byborgenterprises", + "apiUrl": "https://api.eu.lever.co/v0/postings/byborgenterprises?mode=json" + }, + { + "site": "coinspaid", + "companyName": "CoinsPaid", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/coinspaid", + "apiUrl": "https://api.eu.lever.co/v0/postings/coinspaid?mode=json" + }, + { + "site": "controlai", + "companyName": "ControlAI", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/controlai", + "apiUrl": "https://api.eu.lever.co/v0/postings/controlai?mode=json" + }, + { + "site": "coretechsecurity", + "companyName": "CoreTech Security Services", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/coretechsecurity", + "apiUrl": "https://api.eu.lever.co/v0/postings/coretechsecurity?mode=json" + }, + { + "site": "creatio", + "companyName": "Creatio", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/creatio", + "apiUrl": "https://api.eu.lever.co/v0/postings/creatio?mode=json" + }, + { + "site": "csmcy", + "companyName": "Columbia Shipmanagement", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/csmcy", + "apiUrl": "https://api.eu.lever.co/v0/postings/csmcy?mode=json" + }, + { + "site": "diabolocom", + "companyName": "Diabolocom", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/diabolocom", + "apiUrl": "https://api.eu.lever.co/v0/postings/diabolocom?mode=json" + }, + { + "site": "efficioconsulting", + "companyName": "Efficio", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/efficioconsulting", + "apiUrl": "https://api.eu.lever.co/v0/postings/efficioconsulting?mode=json" + }, + { + "site": "eneba", + "companyName": "Eneba", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/eneba", + "apiUrl": "https://api.eu.lever.co/v0/postings/eneba?mode=json" + }, + { + "site": "Expana", + "companyName": "Expana", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/Expana", + "apiUrl": "https://api.eu.lever.co/v0/postings/Expana?mode=json" + }, + { + "site": "frontier", + "companyName": "Frontier Developments", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/frontier", + "apiUrl": "https://api.eu.lever.co/v0/postings/frontier?mode=json" + }, + { + "site": "genomines", + "companyName": "GENOMINES", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/genomines", + "apiUrl": "https://api.eu.lever.co/v0/postings/genomines?mode=json" + }, + { + "site": "innogames", + "companyName": "InnoGames", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/innogames", + "apiUrl": "https://api.eu.lever.co/v0/postings/innogames?mode=json" + }, + { + "site": "ivcevidensia", + "companyName": "IVC", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/ivcevidensia", + "apiUrl": "https://api.eu.lever.co/v0/postings/ivcevidensia?mode=json" + }, + { + "site": "jacquemus", + "companyName": "JACQUEMUS", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/jacquemus", + "apiUrl": "https://api.eu.lever.co/v0/postings/jacquemus?mode=json" + }, + { + "site": "kaiko", + "companyName": "Kaiko", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/kaiko", + "apiUrl": "https://api.eu.lever.co/v0/postings/kaiko?mode=json" + }, + { + "site": "karllagerfeld", + "companyName": "Karl Lagerfeld", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/karllagerfeld", + "apiUrl": "https://api.eu.lever.co/v0/postings/karllagerfeld?mode=json" + }, + { + "site": "kwalee", + "companyName": "Kwalee", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/kwalee", + "apiUrl": "https://api.eu.lever.co/v0/postings/kwalee?mode=json" + }, + { + "site": "lovehoneygroup", + "companyName": "Lovehoney Group", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/lovehoneygroup", + "apiUrl": "https://api.eu.lever.co/v0/postings/lovehoneygroup?mode=json" + }, + { + "site": "mobileye", + "companyName": "Mobileye", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/mobileye", + "apiUrl": "https://api.eu.lever.co/v0/postings/mobileye?mode=json" + }, + { + "site": "olx", + "companyName": "OLX", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/olx", + "apiUrl": "https://api.eu.lever.co/v0/postings/olx?mode=json" + }, + { + "site": "oni", + "companyName": "ONI", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/oni", + "apiUrl": "https://api.eu.lever.co/v0/postings/oni?mode=json" + }, + { + "site": "opencast", + "companyName": "Opencast", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/opencast", + "apiUrl": "https://api.eu.lever.co/v0/postings/opencast?mode=json" + }, + { + "site": "optotune", + "companyName": "Optotune", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/optotune", + "apiUrl": "https://api.eu.lever.co/v0/postings/optotune?mode=json" + }, + { + "site": "orasio", + "companyName": "Orasio", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/orasio", + "apiUrl": "https://api.eu.lever.co/v0/postings/orasio?mode=json" + }, + { + "site": "pnlfin", + "companyName": "Finom", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/pnlfin", + "apiUrl": "https://api.eu.lever.co/v0/postings/pnlfin?mode=json" + }, + { + "site": "prima", + "companyName": "Prima", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/prima", + "apiUrl": "https://api.eu.lever.co/v0/postings/prima?mode=json" + }, + { + "site": "prosus", + "companyName": "Prosus", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/prosus", + "apiUrl": "https://api.eu.lever.co/v0/postings/prosus?mode=json" + }, + { + "site": "quadcode", + "companyName": "Quadcode", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/quadcode", + "apiUrl": "https://api.eu.lever.co/v0/postings/quadcode?mode=json" + }, + { + "site": "quanticdream", + "companyName": "QUANTIC DREAM", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/quanticdream", + "apiUrl": "https://api.eu.lever.co/v0/postings/quanticdream?mode=json" + }, + { + "site": "quantinuum", + "companyName": "Quantinuum", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/quantinuum", + "apiUrl": "https://api.eu.lever.co/v0/postings/quantinuum?mode=json" + }, + { + "site": "refugeerights", + "companyName": "International Refugee Assistance Project", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/refugeerights", + "apiUrl": "https://api.eu.lever.co/v0/postings/refugeerights?mode=json" + }, + { + "site": "seb", + "companyName": "SEB", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/seb", + "apiUrl": "https://api.eu.lever.co/v0/postings/seb?mode=json" + }, + { + "site": "silverfin", + "companyName": "Silverfin", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/silverfin", + "apiUrl": "https://api.eu.lever.co/v0/postings/silverfin?mode=json" + }, + { + "site": "sportalliance", + "companyName": "Sport Alliance GmbH", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/sportalliance", + "apiUrl": "https://api.eu.lever.co/v0/postings/sportalliance?mode=json" + }, + { + "site": "swave", + "companyName": "Swave", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/swave", + "apiUrl": "https://api.eu.lever.co/v0/postings/swave?mode=json" + }, + { + "site": "tradelink", + "companyName": "TradeLink", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/tradelink", + "apiUrl": "https://api.eu.lever.co/v0/postings/tradelink?mode=json" + }, + { + "site": "tsugu", + "companyName": "Tsugu AG", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/tsugu", + "apiUrl": "https://api.eu.lever.co/v0/postings/tsugu?mode=json" + }, + { + "site": "vwgds", + "companyName": "Volkswagen Group Digital Solutions [Portugal]", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/vwgds", + "apiUrl": "https://api.eu.lever.co/v0/postings/vwgds?mode=json" + }, + { + "site": "westernacher", + "companyName": "Westernacher Consulting", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/westernacher", + "apiUrl": "https://api.eu.lever.co/v0/postings/westernacher?mode=json" + }, + { + "site": "wypoon", + "companyName": "Wypoon Technologies", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/wypoon", + "apiUrl": "https://api.eu.lever.co/v0/postings/wypoon?mode=json" + }, + { + "site": "xm", + "companyName": "XM Careers", + "region": "eu", + "jobsUrl": "https://jobs.eu.lever.co/xm", + "apiUrl": "https://api.eu.lever.co/v0/postings/xm?mode=json" + } ] diff --git a/scraper-go/internal/jobstore/jobstore.go b/scraper-go/internal/jobstore/jobstore.go index f2e8d10..ce053ce 100644 --- a/scraper-go/internal/jobstore/jobstore.go +++ b/scraper-go/internal/jobstore/jobstore.go @@ -11,7 +11,7 @@ import ( "time" "unicode" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/redis/go-redis/v9" "golang.org/x/text/transform" "golang.org/x/text/unicode/norm" @@ -33,7 +33,7 @@ func New(rdb *redis.Client) *Store { // SaveBatch persiste vagas novas no Valkey, ignorando duplicatas. // Retorna o número de vagas efetivamente novas salvas. -func (s *Store) SaveBatch(ctx context.Context, jobs []models.Job) (int, error) { +func (s *Store) SaveBatch(ctx context.Context, jobs []domain.Job) (int, error) { if len(jobs) == 0 { return 0, nil } @@ -87,16 +87,16 @@ func (s *Store) SaveBatch(ctx context.Context, jobs []models.Job) (int, error) { } // GetAll retorna todas as vagas do índice. -func (s *Store) GetAll(ctx context.Context) ([]models.Job, error) { +func (s *Store) GetAll(ctx context.Context) ([]domain.Job, error) { ids, err := s.rdb.SMembers(ctx, indexKey).Result() if err != nil { return nil, fmt.Errorf("jobstore.GetAll: SMembers: %w", err) } if len(ids) == 0 { - return []models.Job{}, nil + return []domain.Job{}, nil } - jobs := make([]models.Job, 0, len(ids)) + jobs := make([]domain.Job, 0, len(ids)) for _, id := range ids { raw, err := s.rdb.Get(ctx, jobKeyPrefix+id).Result() @@ -109,7 +109,7 @@ func (s *Store) GetAll(ctx context.Context) ([]models.Job, error) { continue } - var job models.Job + var job domain.Job if err := json.Unmarshal([]byte(raw), &job); err != nil { slog.Warn("jobstore.GetAll: erro ao deserializar", "id", id, "error", err) continue @@ -123,7 +123,7 @@ func (s *Store) GetAll(ctx context.Context) ([]models.Job, error) { // GetSample retorna uma amostra limitada de vagas do índice global. // Também remove IDs órfãos encontrados durante a leitura. -func (s *Store) GetSample(ctx context.Context, limit int) ([]models.Job, error) { +func (s *Store) GetSample(ctx context.Context, limit int) ([]domain.Job, error) { if limit <= 0 { return s.GetAll(ctx) } @@ -150,10 +150,10 @@ func (s *Store) GetSample(ctx context.Context, limit int) ([]models.Job, error) } if len(ids) == 0 { - return []models.Job{}, nil + return []domain.Job{}, nil } - jobs := make([]models.Job, 0, len(ids)) + jobs := make([]domain.Job, 0, len(ids)) for _, id := range ids { raw, err := s.rdb.Get(ctx, jobKeyPrefix+id).Result() @@ -166,7 +166,7 @@ func (s *Store) GetSample(ctx context.Context, limit int) ([]models.Job, error) continue } - var job models.Job + var job domain.Job if err := json.Unmarshal([]byte(raw), &job); err != nil { slog.Warn("jobstore.GetSample: erro ao deserializar", "id", id, "error", err) continue @@ -189,7 +189,7 @@ func (s *Store) Count(ctx context.Context) (int64, error) { // StableID deriva um ID determinístico via SHA-256 truncado de título+empresa+local. // Exportado para o linkedin.go e outros adapters poderem setar job.ID corretamente. -func StableID(j *models.Job) string { +func StableID(j *domain.Job) string { title := normalizeForID(j.Title) company := normalizeForID(j.Company) location := normalizeForID(j.Location) diff --git a/scraper-go/internal/jobstore/jobstore_test.go b/scraper-go/internal/jobstore/jobstore_test.go index 6489f47..c3aeac4 100644 --- a/scraper-go/internal/jobstore/jobstore_test.go +++ b/scraper-go/internal/jobstore/jobstore_test.go @@ -10,8 +10,8 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" ) func newTestStore(t *testing.T) (*jobstore.Store, *miniredis.Miniredis) { @@ -33,7 +33,7 @@ func TestSaveBatch_IndexGlobalSemTTL(t *testing.T) { store, _ := newTestStore(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {Title: "Engenheiro Go", Company: "Acme", Location: "Brasil"}, } @@ -55,7 +55,7 @@ func TestSaveBatch_TTLVagaIndividual(t *testing.T) { store := jobstore.New(rdb) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {Title: "Dev Go", Company: "Empresa A", Location: "Brasil"}, } @@ -84,7 +84,7 @@ func TestSaveBatch_IndexSobreviventeAposExpiracao(t *testing.T) { ctx := context.Background() // Ciclo 1: semana atual - semana1 := []models.Job{ + semana1 := []domain.Job{ {Title: "Dev Go", Company: "Empresa A", Location: "Brasil"}, {Title: "Dev Python", Company: "Empresa B", Location: "Brasil"}, } @@ -93,7 +93,7 @@ func TestSaveBatch_IndexSobreviventeAposExpiracao(t *testing.T) { assert.Equal(t, 2, saved1) // Ciclo 2: semana seguinte, vagas novas + as antigas ainda no cache - semana2 := []models.Job{ + semana2 := []domain.Job{ {Title: "Dev Rust", Company: "Empresa C", Location: "Brasil"}, } saved2, err := store.SaveBatch(ctx, semana2) @@ -130,14 +130,14 @@ func TestSaveBatch_DeduplicacaoEntreCiclos(t *testing.T) { store := jobstore.New(rdb) ctx := context.Background() - vaga := models.Job{Title: "Dev Go", Company: "Acme", Location: "Brasil"} + vaga := domain.Job{Title: "Dev Go", Company: "Acme", Location: "Brasil"} - saved1, err := store.SaveBatch(ctx, []models.Job{vaga}) + saved1, err := store.SaveBatch(ctx, []domain.Job{vaga}) require.NoError(t, err) assert.Equal(t, 1, saved1) // Mesma vaga no próximo ciclo - saved2, err := store.SaveBatch(ctx, []models.Job{vaga}) + saved2, err := store.SaveBatch(ctx, []domain.Job{vaga}) require.NoError(t, err) assert.Equal(t, 0, saved2, "vaga duplicada não deve ser salva novamente") @@ -154,7 +154,7 @@ func TestGetAll_LimpaIDsOrfaos(t *testing.T) { store := jobstore.New(rdb) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {Title: "Dev Go", Company: "Acme", Location: "Brasil"}, {Title: "Dev Rust", Company: "Acme", Location: "Brasil"}, } @@ -182,7 +182,7 @@ func TestGetSample_RespeitaLimite(t *testing.T) { store, _ := newTestStore(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {Title: "Dev Go", Company: "Acme", Location: "Brasil"}, {Title: "Dev Rust", Company: "Globex", Location: "Brasil"}, {Title: "Dev Python", Company: "Initech", Location: "Brasil"}, diff --git a/scraper-go/internal/keywords/keywords copy.json b/scraper-go/internal/keywords/keywords copy.json new file mode 100644 index 0000000..3bb9a00 --- /dev/null +++ b/scraper-go/internal/keywords/keywords copy.json @@ -0,0 +1,18 @@ +{ + "KEYWORDS": [ + "php developer", + "laravel developer", + "symfony developer", + "javascript developer", + "typescript developer", + "node developer", + "node.js developer", + "nestjs developer", + "react developer", + "next.js developer", + "angular developer", + "vue developer", + "golang developer", + "go developer" + ] +} diff --git a/scraper-go/internal/keywords/keywords.json b/scraper-go/internal/keywords/keywords.json index 521baa1..1ea28a8 100644 --- a/scraper-go/internal/keywords/keywords.json +++ b/scraper-go/internal/keywords/keywords.json @@ -1,180 +1,227 @@ { "KEYWORDS": [ - "UX Designer", - "UI Designer", - "UX/UI Designer", - "Product Designer", - "Interaction Designer", - "Visual Designer", - "Service Designer", - "Design Lead", - "Product Manager", - "Product Owner", - "Product Analyst", - "Growth Product Manager", - "Technical Product Manager", - - "Frontend Developer", - "Frontend Engineer", - "Web Developer", - "React Developer", - "Next.js Developer", - "Vue Developer", - "Angular Developer", - "JavaScript Developer", - "TypeScript Developer", - "UI Engineer", - "Frontend Architect", - - "Backend Developer", - "Backend Engineer", - "Node.js Developer", - "NestJS Developer", - "Java Developer", - "Spring Boot Developer", - "Python Developer", - "Django Developer", - "Flask Developer", - "Ruby Developer", - "Rails Developer", - "PHP Developer", - "Laravel Developer", - "Go Developer", - "Golang Developer", - "Rust Developer", - ".NET Developer", - "C# Developer", - - "Fullstack Developer", - "Fullstack Engineer", - "Software Engineer", - "Software Developer", - "Application Developer", - "Platform Engineer", - - "Mobile Developer", - "iOS Developer", - "Android Developer", - "React Native Developer", - "Flutter Developer", - "Mobile Engineer", - - "Data Engineer", - "Data Analyst", - "Data Scientist", - "Machine Learning Engineer", - "ML Engineer", - "AI Engineer", - "Analytics Engineer", - "BI Analyst", - "Business Intelligence Developer", - - "DevOps Engineer", - "Site Reliability Engineer", - "SRE", - "Cloud Engineer", - "Platform Engineer", - "Infrastructure Engineer", - "Systems Engineer", - - "Security Engineer", - "Application Security Engineer", - "Cybersecurity Analyst", - "Information Security Engineer", - "DevSecOps Engineer", - - "QA Engineer", - "Test Engineer", - "Automation Engineer", - "QA Analyst", - "Software Tester", - "SDET", - - "Game Developer", - "Unity Developer", - "Unreal Developer", - - "Embedded Engineer", - "Firmware Engineer", - "Hardware Engineer", - "IoT Developer", - - "Blockchain Developer", - "Web3 Developer", - "Smart Contract Developer", - "Solidity Developer", - - "AR Developer", - "VR Developer", - "XR Developer", - - "Technical Lead", - "Tech Lead", - "Engineering Manager", - "Head of Engineering", - "CTO", - - "Startup Engineer", - "Founding Engineer", - - "Intern Software Engineer", - "Software Engineering Intern", - "Frontend Intern", - "Backend Intern", - "Data Intern", - "IT Intern", - "Estágio em Tecnologia", - "Estágio em Desenvolvimento", - "Estágio em Engenharia de Software", - "Trainee Developer", - "Trainee Software Engineer", - "Junior Developer", - "Mid-level Developer", - "Senior Developer", - "Lead Developer", - "Principal Engineer", - "Staff Engineer", - - "Remote Developer", - "Remote Software Engineer", - "Remote Frontend Developer", - "Remote Backend Developer", - - "Docker", - "Kubernetes", - "Terraform", - "Ansible", - "CI/CD", - "GitHub Actions", - "GitLab CI", - "Jenkins", - "Microservices", - "Distributed Systems", - "REST API", - "GraphQL", - "gRPC", - - "AWS", - "Azure", - "Google Cloud", - "GCP", - "Serverless", - "Lambda", - "Cloud Functions", - - "PostgreSQL", - "MySQL", - "MongoDB", - "Redis", - "Elasticsearch", - "Kafka", - "RabbitMQ", - "DynamoDB", - - "Agile", - "Scrum", - "Kanban", - "Lean", - "XP" + "php developer", + "laravel developer", + "symfony developer", + "javascript developer", + "typescript developer", + "node developer", + "node.js developer", + "nestjs developer", + "react developer", + "next.js developer", + "angular developer", + "vue developer", + "golang developer", + "go developer", + "software engineer", + "software developer", + "application developer", + "platform engineer", + "systems engineer", + "frontend developer", + "frontend engineer", + "ui engineer", + "frontend architect", + "backend developer", + "backend engineer", + "java developer", + "spring boot developer", + "python developer", + "django developer", + "flask developer", + "rust developer", + ".net developer", + "c# developer", + "ruby on rails developer", + "full stack developer", + "full stack engineer", + "mobile developer", + "android developer", + "ios developer", + "react native developer", + "flutter developer", + "data engineer", + "data scientist", + "data analyst", + "analytics engineer", + "machine learning engineer", + "ai engineer", + "mlops engineer", + "bi developer", + "business intelligence analyst", + "devops engineer", + "site reliability engineer", + "sre", + "cloud engineer", + "infrastructure engineer", + "qa engineer", + "qa analyst", + "automation engineer", + "test engineer", + "sdet", + "security engineer", + "cybersecurity engineer", + "information security analyst", + "application security engineer", + "devsecops engineer", + "ux designer", + "ui designer", + "ux/ui designer", + "product designer", + "visual designer", + "interaction designer", + "service designer", + "ux researcher", + "ux writer", + "design system designer", + "product manager", + "technical product manager", + "product owner", + "product analyst", + "business analyst", + "salesforce developer", + "salesforce engineer", + "salesforce administrator", + "salesforce consultant", + "salesforce architect", + "salesforce technical lead", + "salesforce apex developer", + "salesforce lightning developer", + "salesforce lwc developer", + "salesforce integration developer", + "salesforce commerce cloud developer", + "salesforce marketing cloud developer", + "salesforce service cloud developer", + "salesforce sales cloud developer", + "salesforce cpq developer", + "mulesoft developer", + "mulesoft engineer", + "mulesoft integration developer", + "mulesoft architect", + "integration engineer", + "integration developer", + "api developer", + "api integration engineer", + "esb developer", + "sap abap developer", + "sap abap consultant", + "sap hana developer", + "sap hana consultant", + "sap fiori developer", + "sap ui5 developer", + "sap btp developer", + "sap cpi developer", + "sap pi developer", + "sap po developer", + "sap bw consultant", + "sap basis consultant", + "sap mm consultant", + "sap sd consultant", + "sap fi consultant", + "sap co consultant", + "sap pp consultant", + "sap qm consultant", + "sap wm consultant", + "sap ewm consultant", + "sap successfactors consultant", + "oracle developer", + "oracle apex developer", + "oracle pl/sql developer", + "dynamics 365 developer", + "power platform developer", + "power apps developer", + "power automate developer", + "blockchain developer", + "solidity developer", + "web3 developer", + "embedded software engineer", + "firmware engineer", + "iot engineer", + "unity developer", + "unreal engine developer", + "game developer", + "technical lead", + "tech lead", + "engineering manager", + "head of engineering", + "cto", + "principal engineer", + "staff engineer", + "junior developer", + "mid-level developer", + "senior developer", + "lead developer", + "software engineering intern", + "frontend intern", + "backend intern", + "data intern", + "it intern", + "estágio em tecnologia", + "estágio em desenvolvimento", + "estágio em engenharia de software", + "trainee developer", + "trainee software engineer", + "design lead", + "web developer", + "ruby developer", + "rails developer", + "fullstack developer", + "fullstack engineer", + "mobile engineer", + "ml engineer", + "bi analyst", + "business intelligence developer", + "cybersecurity analyst", + "information security engineer", + "software tester", + "unreal developer", + "embedded engineer", + "hardware engineer", + "iot developer", + "smart contract developer", + "ar developer", + "vr developer", + "xr developer", + "startup engineer", + "founding engineer", + "intern software engineer", + "remote developer", + "remote software engineer", + "remote frontend developer", + "remote backend developer", + "docker", + "kubernetes", + "terraform", + "ansible", + "ci/cd", + "github actions", + "gitlab ci", + "jenkins", + "microservices", + "distributed systems", + "rest api", + "graphql", + "grpc", + "aws", + "azure", + "google cloud", + "gcp", + "serverless", + "lambda", + "cloud functions", + "postgresql", + "mysql", + "mongodb", + "redis", + "elasticsearch", + "kafka", + "rabbitmq", + "dynamodb", + "agile", + "scrum", + "kanban", + "lean", + "xp" ] } diff --git a/scraper-go/internal/keywords/keywords.test.json b/scraper-go/internal/keywords/keywords.test.json index e7a6337..5c0079b 100644 --- a/scraper-go/internal/keywords/keywords.test.json +++ b/scraper-go/internal/keywords/keywords.test.json @@ -7,3 +7,184 @@ "Interaction Designer" ] } + +// { +// "KEYWORDS": [ +// "UX Designer", +// "UI Designer", +// "UX/UI Designer", +// "Product Designer", +// "Interaction Designer", +// "Visual Designer", +// "Service Designer", +// "Design Lead", +// "Product Manager", +// "Product Owner", +// "Product Analyst", +// "Growth Product Manager", +// "Technical Product Manager", + +// "Frontend Developer", +// "Frontend Engineer", +// "Web Developer", +// "React Developer", +// "Next.js Developer", +// "Vue Developer", +// "Angular Developer", +// "JavaScript Developer", +// "TypeScript Developer", +// "UI Engineer", +// "Frontend Architect", + +// "Backend Developer", +// "Backend Engineer", +// "Node.js Developer", +// "NestJS Developer", +// "Java Developer", +// "Spring Boot Developer", +// "Python Developer", +// "Django Developer", +// "Flask Developer", +// "Ruby Developer", +// "Rails Developer", +// "PHP Developer", +// "Laravel Developer", +// "Go Developer", +// "Golang Developer", +// "Rust Developer", +// ".NET Developer", +// "C# Developer", + +// "Fullstack Developer", +// "Fullstack Engineer", +// "Software Engineer", +// "Software Developer", +// "Application Developer", +// "Platform Engineer", + +// "Mobile Developer", +// "iOS Developer", +// "Android Developer", +// "React Native Developer", +// "Flutter Developer", +// "Mobile Engineer", + +// "Data Engineer", +// "Data Analyst", +// "Data Scientist", +// "Machine Learning Engineer", +// "ML Engineer", +// "AI Engineer", +// "Analytics Engineer", +// "BI Analyst", +// "Business Intelligence Developer", + +// "DevOps Engineer", +// "Site Reliability Engineer", +// "SRE", +// "Cloud Engineer", +// "Platform Engineer", +// "Infrastructure Engineer", +// "Systems Engineer", + +// "Security Engineer", +// "Application Security Engineer", +// "Cybersecurity Analyst", +// "Information Security Engineer", +// "DevSecOps Engineer", + +// "QA Engineer", +// "Test Engineer", +// "Automation Engineer", +// "QA Analyst", +// "Software Tester", +// "SDET", + +// "Game Developer", +// "Unity Developer", +// "Unreal Developer", + +// "Embedded Engineer", +// "Firmware Engineer", +// "Hardware Engineer", +// "IoT Developer", + +// "Blockchain Developer", +// "Web3 Developer", +// "Smart Contract Developer", +// "Solidity Developer", + +// "AR Developer", +// "VR Developer", +// "XR Developer", + +// "Technical Lead", +// "Tech Lead", +// "Engineering Manager", +// "Head of Engineering", +// "CTO", + +// "Startup Engineer", +// "Founding Engineer", + +// "Intern Software Engineer", +// "Software Engineering Intern", +// "Frontend Intern", +// "Backend Intern", +// "Data Intern", +// "IT Intern", +// "Estágio em Tecnologia", +// "Estágio em Desenvolvimento", +// "Estágio em Engenharia de Software", +// "Trainee Developer", +// "Trainee Software Engineer", +// "Junior Developer", +// "Mid-level Developer", +// "Senior Developer", +// "Lead Developer", +// "Principal Engineer", +// "Staff Engineer", + +// "Remote Developer", +// "Remote Software Engineer", +// "Remote Frontend Developer", +// "Remote Backend Developer", + +// "Docker", +// "Kubernetes", +// "Terraform", +// "Ansible", +// "CI/CD", +// "GitHub Actions", +// "GitLab CI", +// "Jenkins", +// "Microservices", +// "Distributed Systems", +// "REST API", +// "GraphQL", +// "gRPC", + +// "AWS", +// "Azure", +// "Google Cloud", +// "GCP", +// "Serverless", +// "Lambda", +// "Cloud Functions", + +// "PostgreSQL", +// "MySQL", +// "MongoDB", +// "Redis", +// "Elasticsearch", +// "Kafka", +// "RabbitMQ", +// "DynamoDB", + +// "Agile", +// "Scrum", +// "Kanban", +// "Lean", +// "XP" +// ] +// } diff --git a/scraper-go/internal/keywords/normalize.go b/scraper-go/internal/keywords/normalize.go index c7bcb96..95f2e71 100644 --- a/scraper-go/internal/keywords/normalize.go +++ b/scraper-go/internal/keywords/normalize.go @@ -7,7 +7,7 @@ func NormalizeKeywords(keywords []string) []string { var result []string for _, k := range keywords { - k = strings.TrimSpace(k) + k = strings.ToLower(strings.TrimSpace(k)) if k == "" { continue diff --git a/scraper-go/internal/keywords/normalize_test.go b/scraper-go/internal/keywords/normalize_test.go new file mode 100644 index 0000000..db3e56b --- /dev/null +++ b/scraper-go/internal/keywords/normalize_test.go @@ -0,0 +1,15 @@ +package keywords + +import ( + "reflect" + "testing" +) + +func TestNormalizeKeywordsTrimsLowercasesAndDeduplicates(t *testing.T) { + got := NormalizeKeywords([]string{" Go ", "go", "", " Node.js ", "NODE.js"}) + want := []string{"go", "node.js"} + + if !reflect.DeepEqual(got, want) { + t.Fatalf("NormalizeKeywords() = %#v, want %#v", got, want) + } +} diff --git a/scraper-go/internal/models/job.go b/scraper-go/internal/models/job.go index 669882e..34ed3b9 100644 --- a/scraper-go/internal/models/job.go +++ b/scraper-go/internal/models/job.go @@ -1,40 +1,8 @@ package models -type Job struct { - ID string `json:"id"` - Title string `json:"title"` - Company string `json:"company"` - Location string `json:"location"` - URL string `json:"url"` - Salary string `json:"salary,omitempty"` - Modality string `json:"modality,omitempty"` - Description string `json:"description,omitempty"` - PostedAt string `json:"postedAt,omitempty"` - Source string `json:"source"` - Sources []string `json:"sources"` - Keyword string `json:"keyword"` - Keywords []string `json:"keywords"` -} +import "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" -type ScrapeRequest struct { - Keywords []string `json:"keywords"` - SearchLocation string `json:"searchLocation"` - SearchGeoID string `json:"searchGeoId"` - SearchLanguage string `json:"searchLanguage"` - JobTypes string `json:"jobTypes"` - TimeFilter string `json:"timeFilter"` - RemoteOnly bool `json:"remoteOnly"` - Sources []string `json:"sources"` - ResultsPerPage int `json:"resultsPerPage"` - MaxPagesPerKeyword int `json:"maxPagesPerKeyword"` - WaitBetweenSearchesMs int `json:"waitBetweenSearchesMs"` - PageTimeoutMs int `json:"pageTimeoutMs"` - MaxConcurrency int `json:"maxConcurrency"` -} - -type ScrapeResponse struct { - Jobs []Job `json:"jobs"` - Total int `json:"total"` - CachedAt string `json:"cachedAt"` - FromCache bool `json:"fromCache"` -} +type Job = domain.Job +type Classification = domain.Classification +type ScrapeRequest = domain.ScrapeRequest +type ScrapeResponse = domain.ScrapeResponse diff --git a/scraper-go/internal/pipeline/cache_key.go b/scraper-go/internal/pipeline/cache_key.go index 472027e..43a1712 100644 --- a/scraper-go/internal/pipeline/cache_key.go +++ b/scraper-go/internal/pipeline/cache_key.go @@ -2,30 +2,54 @@ package pipeline import ( "sort" + "strconv" "strings" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" ) func BuildCacheKey(config SearchConfig) string { - keywords := normalizeKeywords(config.Keywords) + normalizedKeywords := keywords.NormalizeKeywords(config.Keywords) + sort.Strings(normalizedKeywords) + + sources := normalizeCacheValues(config.Sources) return strings.Join([]string{ "jobs", - strings.Join(keywords, ","), - strings.ToLower(strings.TrimSpace(config.SearchLocation)), - config.JobTypes, - config.TimeFilter, + strings.Join(normalizedKeywords, ","), + normalizeCacheValue(config.SearchLocation), + normalizeCacheValue(config.SearchGeoID), + normalizeCacheValue(config.SearchLanguage), + normalizeCacheValue(config.JobTypes), + normalizeCacheValue(config.TimeFilter), + strconv.FormatBool(config.RemoteOnly), + strings.Join(sources, ","), + strconv.Itoa(config.ResultsPerPage), + strconv.Itoa(config.MaxPagesPerKeyword), + strconv.Itoa(config.WaitBetweenSearchesMs), + strconv.Itoa(config.PageTimeoutMs), + strconv.Itoa(config.MaxConcurrency), }, ":") } -func normalizeKeywords(keywords []string) []string { - var normalized []string +func normalizeCacheValue(value string) string { + return strings.ToLower(strings.TrimSpace(value)) +} - for _, k := range keywords { - k = strings.ToLower(strings.TrimSpace(k)) +func normalizeCacheValues(values []string) []string { + unique := make(map[string]struct{}, len(values)) + normalized := make([]string, 0, len(values)) - if k != "" { - normalized = append(normalized, k) + for _, value := range values { + value = normalizeCacheValue(value) + if value == "" { + continue + } + if _, exists := unique[value]; exists { + continue } + unique[value] = struct{}{} + normalized = append(normalized, value) } sort.Strings(normalized) diff --git a/scraper-go/internal/pipeline/cache_key_test.go b/scraper-go/internal/pipeline/cache_key_test.go new file mode 100644 index 0000000..e01a6f9 --- /dev/null +++ b/scraper-go/internal/pipeline/cache_key_test.go @@ -0,0 +1,79 @@ +package pipeline + +import "testing" + +func TestBuildCacheKeyNormalizesAndDeduplicatesKeywords(t *testing.T) { + a := BuildCacheKey(SearchConfig{ + Keywords: []string{" Go ", "go", "", "Node"}, + SearchLocation: " Brasil ", + }) + b := BuildCacheKey(SearchConfig{ + Keywords: []string{"node", "GO"}, + SearchLocation: "brasil", + }) + + if a != b { + t.Fatalf("expected equivalent normalized cache keys, got %q and %q", a, b) + } +} + +func TestBuildCacheKeyIncludesResultAndExecutionConfig(t *testing.T) { + base := SearchConfig{ + Keywords: []string{"go"}, + SearchLocation: "Brasil", + SearchGeoID: "106057199", + SearchLanguage: "pt", + JobTypes: "C,F", + TimeFilter: "r604800", + RemoteOnly: true, + Sources: []string{"LinkedIn", "Adzuna"}, + ResultsPerPage: 20, + MaxPagesPerKeyword: 3, + WaitBetweenSearchesMs: 1000, + PageTimeoutMs: 15000, + MaxConcurrency: 10, + } + + cases := []struct { + name string + mutate func(*SearchConfig) + }{ + {name: "search geo id", mutate: func(c *SearchConfig) { c.SearchGeoID = "92000000" }}, + {name: "search language", mutate: func(c *SearchConfig) { c.SearchLanguage = "en" }}, + {name: "remote only", mutate: func(c *SearchConfig) { c.RemoteOnly = false }}, + {name: "sources", mutate: func(c *SearchConfig) { c.Sources = []string{"LinkedIn"} }}, + {name: "results per page", mutate: func(c *SearchConfig) { c.ResultsPerPage = 50 }}, + {name: "max pages per keyword", mutate: func(c *SearchConfig) { c.MaxPagesPerKeyword = 10 }}, + {name: "wait between searches", mutate: func(c *SearchConfig) { c.WaitBetweenSearchesMs = 2000 }}, + {name: "page timeout", mutate: func(c *SearchConfig) { c.PageTimeoutMs = 30000 }}, + {name: "max concurrency", mutate: func(c *SearchConfig) { c.MaxConcurrency = 25 }}, + } + + baseKey := BuildCacheKey(base) + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + changed := base + changed.Sources = append([]string(nil), base.Sources...) + tc.mutate(&changed) + + if key := BuildCacheKey(changed); key == baseKey { + t.Fatalf("expected cache key to change for %s", tc.name) + } + }) + } +} + +func TestBuildCacheKeyNormalizesSourcesOrder(t *testing.T) { + a := BuildCacheKey(SearchConfig{ + Keywords: []string{"go"}, + Sources: []string{" LinkedIn ", "adzuna", "linkedin"}, + }) + b := BuildCacheKey(SearchConfig{ + Keywords: []string{"go"}, + Sources: []string{"ADZUNA", "linkedin"}, + }) + + if a != b { + t.Fatalf("expected equivalent source sets to share cache key, got %q and %q", a, b) + } +} diff --git a/scraper-go/internal/pipeline/pipeline.go b/scraper-go/internal/pipeline/pipeline.go index 6507251..87cff9e 100644 --- a/scraper-go/internal/pipeline/pipeline.go +++ b/scraper-go/internal/pipeline/pipeline.go @@ -9,25 +9,32 @@ import ( "time" "unicode" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/classifier" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/dedup" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/metrics" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" "github.com/prometheus/client_golang/prometheus" "github.com/redis/go-redis/v9" "golang.org/x/text/transform" "golang.org/x/text/unicode/norm" ) -const defaultMaxConcurrency = 150 +const defaultMaxConcurrency = 40 type result struct { - jobs []models.Job + jobs []domain.Job err error } -func Run(ctx context.Context, adapterList []adapters.Adapter, req models.ScrapeRequest) []models.Job { +type adapterTask struct { + adapter ports.JobSource + keywords []string + batch bool +} + +func Run(ctx context.Context, adapterList []ports.JobSource, req domain.ScrapeRequest) []domain.Job { pipelineStart := time.Now() maxConcurrency := req.MaxConcurrency @@ -35,15 +42,15 @@ func Run(ctx context.Context, adapterList []adapters.Adapter, req models.ScrapeR maxConcurrency = defaultMaxConcurrency } - type task struct { - adapter adapters.Adapter - keyword string - } - - tasks := make([]task, 0, len(adapterList)*len(req.Keywords)) + tasks := make([]adapterTask, 0, len(adapterList)*len(req.Keywords)) for _, a := range adapterList { + if _, ok := a.(ports.BatchJobSource); ok { + tasks = append(tasks, adapterTask{adapter: a, keywords: req.Keywords, batch: true}) + continue + } + for _, kw := range req.Keywords { - tasks = append(tasks, task{adapter: a, keyword: kw}) + tasks = append(tasks, adapterTask{adapter: a, keywords: []string{kw}}) } } @@ -55,14 +62,14 @@ func Run(ctx context.Context, adapterList []adapters.Adapter, req models.ScrapeR wg.Add(1) sem <- struct{}{} - go func(t task) { + go func(t adapterTask) { defer wg.Done() defer func() { <-sem }() source := t.adapter.SourceName() timer := prometheus.NewTimer(metrics.ScrapeDurationSeconds.WithLabelValues(source)) - jobs, err := t.adapter.Search(ctx, t.keyword, req) + jobs, err := runAdapterTask(ctx, t, req) timer.ObserveDuration() metrics.ScrapeRunsTotal.WithLabelValues(source).Inc() @@ -73,7 +80,8 @@ func Run(ctx context.Context, adapterList []adapters.Adapter, req models.ScrapeR metrics.ScrapeErrorsTotal.WithLabelValues(source).Inc() slog.Warn("adapter falhou", "source", source, - "keyword", t.keyword, + "keywords", len(t.keywords), + "batch", t.batch, "error", err, ) return @@ -83,7 +91,8 @@ func Run(ctx context.Context, adapterList []adapters.Adapter, req models.ScrapeR slog.Info("adapter concluído", "source", source, - "keyword", t.keyword, + "keywords", len(t.keywords), + "batch", t.batch, "count", len(jobs), ) }(t) @@ -94,7 +103,7 @@ func Run(ctx context.Context, adapterList []adapters.Adapter, req models.ScrapeR close(results) }() - var allJobs []models.Job + var allJobs []domain.Job for r := range results { if r.err == nil { allJobs = append(allJobs, r.jobs...) @@ -102,14 +111,29 @@ func Run(ctx context.Context, adapterList []adapters.Adapter, req models.ScrapeR } deduped := dedup.DedupeJobs(allJobs) + classified := classifier.ClassifyJobs(deduped) metrics.PipelineRunDuration.Observe(time.Since(pipelineStart).Seconds()) - metrics.PipelineJobsTotal.Observe(float64(len(deduped))) + metrics.PipelineJobsTotal.Observe(float64(len(classified))) + + return classified +} - return deduped +func runAdapterTask(ctx context.Context, t adapterTask, req domain.ScrapeRequest) ([]domain.Job, error) { + if t.batch { + batchAdapter := t.adapter.(ports.BatchJobSource) + return batchAdapter.SearchBatch(ctx, t.keywords, req) + } + + keyword := "" + if len(t.keywords) > 0 { + keyword = t.keywords[0] + } + + return t.adapter.Search(ctx, keyword, req) } -func IndexJobsInValkey(ctx context.Context, rdb *redis.Client, jobs []models.Job, keywords []string) { +func IndexJobsInValkey(ctx context.Context, rdb *redis.Client, jobs []domain.Job, keywords []string) { if rdb == nil || len(jobs) == 0 { return } @@ -170,6 +194,10 @@ func IndexJobsInValkey(ctx context.Context, rdb *redis.Client, jobs []models.Job for _, key := range structuredIndexKeys(job) { kwIndex[key] = append(kwIndex[key], id) } + + for _, key := range classificationIndexKeys(job) { + kwIndex[key] = append(kwIndex[key], id) + } } // Publica os índices de keyword com RENAME atômico @@ -202,7 +230,56 @@ func IndexJobsInValkey(ctx context.Context, rdb *redis.Client, jobs []models.Job ) } -func structuredIndexKeys(job models.Job) []string { +func classificationIndexKeys(job domain.Job) []string { + if job.Classification == nil { + return nil + } + + classification := job.Classification + if !classification.InScope { + return nil + } + + values := make([]string, 0, 1+len(classification.RelatedFamilies)+len(classification.Technologies)) + + if classification.PrimaryFamily != "" { + normalized := normalizeIndexValue(classification.PrimaryFamily) + if normalized != "" { + values = append(values, + fmt.Sprintf("scraper:jobs:family:%s", normalized), + fmt.Sprintf("scraper:jobs:keyword:%s", normalized), + ) + } + } + for _, family := range classification.RelatedFamilies { + normalized := normalizeIndexValue(family) + if normalized != "" { + values = append(values, + fmt.Sprintf("scraper:jobs:family:%s", normalized), + fmt.Sprintf("scraper:jobs:keyword:%s", normalized), + ) + } + } + for _, technology := range classification.Technologies { + normalized := normalizeIndexValue(technology) + if normalized != "" { + values = append(values, + fmt.Sprintf("scraper:jobs:technology:%s", normalized), + fmt.Sprintf("scraper:jobs:keyword:%s", normalized), + ) + } + } + if classification.Seniority != "" { + normalized := normalizeIndexValue(classification.Seniority) + if normalized != "" { + values = append(values, fmt.Sprintf("scraper:jobs:seniority:%s", normalized)) + } + } + + return uniqueStrings(values) +} + +func structuredIndexKeys(job domain.Job) []string { values := map[string]string{ "level": inferLevel(job), "model": inferWorkModel(job), @@ -245,7 +322,7 @@ func normalizeIndexValue(value string) string { return strings.Join(strings.Fields(b.String()), " ") } -func keywordSearchText(job models.Job) string { +func keywordSearchText(job domain.Job) string { return normalizeIndexValue(strings.Join([]string{ job.Title, job.Company, @@ -255,7 +332,7 @@ func keywordSearchText(job models.Job) string { }, " ")) } -func searchableJobText(job models.Job) string { +func searchableJobText(job domain.Job) string { return normalizeIndexValue(strings.Join([]string{ job.Title, job.Company, @@ -325,7 +402,7 @@ func uniqueStrings(values []string) []string { return result } -func inferLevel(job models.Job) string { +func inferLevel(job domain.Job) string { text := searchableJobText(job) if containsAny(text, "estagio", "intern", "trainee") { @@ -341,7 +418,7 @@ func inferLevel(job models.Job) string { return "pleno" } -func inferWorkModel(job models.Job) string { +func inferWorkModel(job domain.Job) string { text := searchableJobText(job) if containsAny(text, "hibrido", "hybrid") { @@ -372,7 +449,7 @@ func inferWorkModel(job models.Job) string { return "presencial" } -func inferContract(job models.Job) string { +func inferContract(job domain.Job) string { text := searchableJobText(job) if containsAny(text, "cooperado", "cooperativa") { diff --git a/scraper-go/internal/pipeline/pipeline_test.go b/scraper-go/internal/pipeline/pipeline_test.go index 5a03907..f42a041 100644 --- a/scraper-go/internal/pipeline/pipeline_test.go +++ b/scraper-go/internal/pipeline/pipeline_test.go @@ -10,8 +10,8 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/jobstore" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/pipeline" ) @@ -34,7 +34,7 @@ func TestIndexJobsInValkey_SemJanelaVazia(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {Title: "Engenheiro Go", Company: "Acme", Location: "Brasil", Description: "vaga de go golang"}, } keywords := []string{"go", "golang"} @@ -68,7 +68,7 @@ func TestIndexJobsInValkey_TTLKeywordAlinhado(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {Title: "Dev Go", Company: "Acme", Location: "Brasil", Description: "golang backend"}, } keywords := []string{"go"} @@ -89,7 +89,7 @@ func TestIndexJobsInValkey_IndexGlobalSemTTL(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {Title: "Dev Go", Company: "Acme", Location: "Brasil"}, } @@ -109,7 +109,7 @@ func TestIndexJobsInValkey_KeywordComposta(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ { Title: "Dev Node.js", Company: "Empresa A", @@ -144,7 +144,7 @@ func TestIndexJobsInValkey_SubTermosIndexadosIndividualmente(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ { Title: "Dev Node.js", Company: "Acme", @@ -171,7 +171,7 @@ func TestIndexJobsInValkey_NormalizaAliasesDeTecnologia(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ { Title: "Node.js Developer", Company: "Acme", @@ -201,7 +201,7 @@ func TestIndexJobsInValkey_IndicesEstruturados(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ { Title: "Desenvolvedor Node.js Júnior PJ", Company: "Acme", @@ -232,11 +232,54 @@ func TestIndexJobsInValkey_IndicesEstruturados(t *testing.T) { } } +func TestIndexJobsInValkey_IndicesDeClassificacao(t *testing.T) { + rdb, _ := newTestRedis(t) + ctx := context.Background() + + classification := domain.Classification{ + PrimaryFamily: "backend", + RelatedFamilies: []string{"platform"}, + Technologies: []string{"go", "postgresql"}, + Seniority: "senior", + InScope: true, + Confidence: 0.91, + } + jobs := []domain.Job{ + { + Title: "Senior Software Engineer - APIs", + Company: "Acme", + Location: "Brasil", + Description: "Go e PostgreSQL", + Classification: &classification, + }, + } + + pipeline.IndexJobsInValkey(ctx, rdb, jobs, []string{"software engineer"}) + expectedID := jobstore.StableID(&jobs[0]) + + keys := []string{ + "scraper:jobs:family:backend", + "scraper:jobs:family:platform", + "scraper:jobs:technology:go", + "scraper:jobs:technology:postgresql", + "scraper:jobs:keyword:backend", + "scraper:jobs:keyword:go", + "scraper:jobs:keyword:postgresql", + "scraper:jobs:seniority:senior", + } + + for _, key := range keys { + members, err := rdb.SMembers(ctx, key).Result() + require.NoError(t, err) + assert.Contains(t, members, expectedID, "índice de classificação %s deve conter a vaga", key) + } +} + func TestIndexJobsInValkey_DetectaBrasilPorEstadoECidade(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ { Title: "Junior/midlevel Java Developer - Remote Work", Company: "BairesDev", @@ -267,7 +310,7 @@ func TestIndexJobsInValkey_ClassificacaoIgnoraKeywordDaBusca(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ { Title: "Software Engineer", Company: "Acme", @@ -303,7 +346,7 @@ func TestIndexJobsInValkey_VagasSemIDIgnoradas(t *testing.T) { rdb, _ := newTestRedis(t) ctx := context.Background() - jobs := []models.Job{ + jobs := []domain.Job{ {}, // vaga completamente vazia — StableID retorna "" {Title: "Dev Go", Company: "Acme", Location: "Brasil"}, } diff --git a/scraper-go/internal/pipeline/scrape.go b/scraper-go/internal/pipeline/scrape.go index 238b0f3..3e2f61b 100644 --- a/scraper-go/internal/pipeline/scrape.go +++ b/scraper-go/internal/pipeline/scrape.go @@ -4,36 +4,58 @@ import ( "context" "log/slog" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/keywords" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" "github.com/redis/go-redis/v9" ) type SearchConfig struct { - Keywords []string `json:"keywords"` - SearchLocation string `json:"searchLocation"` - JobTypes string `json:"jobTypes"` - TimeFilter string `json:"timeFilter"` - RemoteOnly bool `json:"remoteOnly"` - MaxConcurrency int `json:"maxConcurrency"` + Keywords []string `json:"keywords"` + SearchLocation string `json:"searchLocation"` + SearchGeoID string `json:"searchGeoId"` + SearchLanguage string `json:"searchLanguage"` + JobTypes string `json:"jobTypes"` + TimeFilter string `json:"timeFilter"` + RemoteOnly bool `json:"remoteOnly"` + Sources []string `json:"sources"` + ResultsPerPage int `json:"resultsPerPage"` + MaxPagesPerKeyword int `json:"maxPagesPerKeyword"` + WaitBetweenSearchesMs int `json:"waitBetweenSearchesMs"` + PageTimeoutMs int `json:"pageTimeoutMs"` + MaxConcurrency int `json:"maxConcurrency"` +} + +func normalizeSearchConfig(config SearchConfig) SearchConfig { + config.Keywords = keywords.NormalizeKeywords(config.Keywords) + return config } func ScrapeAllSources( ctx context.Context, config SearchConfig, + adapterList []ports.JobSource, rdb *redis.Client, -) ([]models.Job, error) { +) ([]domain.Job, error) { + config = normalizeSearchConfig(config) slog.Info("starting scrape", "keywords", config.Keywords) - adapterList := adapters.GetAdapters(rdb) + adapterList = filterAdaptersByCadence(ctx, rdb, adapterList) - req := models.ScrapeRequest{ - Keywords: config.Keywords, - SearchLocation: config.SearchLocation, - JobTypes: config.JobTypes, - TimeFilter: config.TimeFilter, - RemoteOnly: config.RemoteOnly, - MaxConcurrency: config.MaxConcurrency, + req := domain.ScrapeRequest{ + Keywords: config.Keywords, + SearchLocation: config.SearchLocation, + SearchGeoID: config.SearchGeoID, + SearchLanguage: config.SearchLanguage, + JobTypes: config.JobTypes, + TimeFilter: config.TimeFilter, + RemoteOnly: config.RemoteOnly, + Sources: config.Sources, + ResultsPerPage: config.ResultsPerPage, + MaxPagesPerKeyword: config.MaxPagesPerKeyword, + WaitBetweenSearchesMs: config.WaitBetweenSearchesMs, + PageTimeoutMs: config.PageTimeoutMs, + MaxConcurrency: config.MaxConcurrency, } jobs := Run(ctx, adapterList, req) diff --git a/scraper-go/internal/pipeline/search.go b/scraper-go/internal/pipeline/search.go index f68583e..2304930 100644 --- a/scraper-go/internal/pipeline/search.go +++ b/scraper-go/internal/pipeline/search.go @@ -9,12 +9,13 @@ import ( "github.com/redis/go-redis/v9" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/cache" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/inflight" - "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/models" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" ) type SearchResult struct { - Jobs []models.Job `json:"jobs"` + Jobs []domain.Job `json:"jobs"` Total int `json:"total"` CachedAt time.Time `json:"cachedAt"` FromCache bool `json:"fromCache"` @@ -24,9 +25,11 @@ func SearchJobs( ctx context.Context, c cache.Cache, config SearchConfig, + adapterList []ports.JobSource, ttl time.Duration, rdb *redis.Client, ) (SearchResult, error) { + config = normalizeSearchConfig(config) cacheKey := BuildCacheKey(config) if result, found, err := cache.GetAs[SearchResult](c, ctx, cacheKey); err != nil { @@ -44,7 +47,7 @@ func SearchJobs( return result, nil } - jobs, err := ScrapeAllSources(ctx, config, rdb) + jobs, err := ScrapeAllSources(ctx, config, adapterList, rdb) if err != nil { return SearchResult{}, fmt.Errorf("pipeline.SearchJobs: scrape: %w", err) } diff --git a/scraper-go/internal/pipeline/source_schedule.go b/scraper-go/internal/pipeline/source_schedule.go new file mode 100644 index 0000000..fc5f601 --- /dev/null +++ b/scraper-go/internal/pipeline/source_schedule.go @@ -0,0 +1,60 @@ +package pipeline + +import ( + "context" + "log/slog" + "strings" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/ports" + "github.com/redis/go-redis/v9" +) + +const ( + joobleRunGateKey = "scraper:source:jooble:last_run" + joobleRunCadence = 12 * time.Hour +) + +func filterAdaptersByCadence(ctx context.Context, rdb *redis.Client, adapterList []ports.JobSource) []ports.JobSource { + filtered := make([]ports.JobSource, 0, len(adapterList)) + + for _, adapter := range adapterList { + if !strings.EqualFold(adapter.SourceName(), "Jooble") { + filtered = append(filtered, adapter) + continue + } + + if shouldRunJooble(ctx, rdb) { + filtered = append(filtered, adapter) + } + } + + return filtered +} + +func shouldRunJooble(ctx context.Context, rdb *redis.Client) bool { + if rdb == nil { + slog.Warn("jooble: Valkey indisponível, cadência de 12h não será aplicada") + return true + } + + now := time.Now() + ok, err := rdb.SetNX(ctx, joobleRunGateKey, now.Format(time.RFC3339), joobleRunCadence).Result() + if err != nil { + slog.Warn("jooble: falha ao verificar cadência de execução, adapter será liberado", "error", err) + return true + } + if ok { + slog.Info("jooble: execução liberada", "cadence", joobleRunCadence) + return true + } + + ttl, err := rdb.TTL(ctx, joobleRunGateKey).Result() + if err != nil { + slog.Info("jooble: execução ignorada pela cadência de 12h") + return false + } + + slog.Info("jooble: execução ignorada pela cadência de 12h", "next_available_in", ttl.Round(time.Minute)) + return false +} diff --git a/scraper-go/internal/pipeline/source_schedule_test.go b/scraper-go/internal/pipeline/source_schedule_test.go new file mode 100644 index 0000000..13374de --- /dev/null +++ b/scraper-go/internal/pipeline/source_schedule_test.go @@ -0,0 +1,99 @@ +package pipeline + +import ( + "context" + "testing" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/adapters" + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" + "github.com/alicebob/miniredis/v2" + "github.com/redis/go-redis/v9" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +type sourceScheduleTestAdapter struct { + source string +} + +func (a sourceScheduleTestAdapter) SourceName() string { + return a.source +} + +func (a sourceScheduleTestAdapter) Search(context.Context, string, domain.ScrapeRequest) ([]domain.Job, error) { + return nil, nil +} + +type batchRunTestAdapter struct { + searchCalls int + batchCalls int + keywords []string +} + +func (a *batchRunTestAdapter) SourceName() string { + return "Batch Test" +} + +func (a *batchRunTestAdapter) Search(context.Context, string, domain.ScrapeRequest) ([]domain.Job, error) { + a.searchCalls++ + return nil, nil +} + +func (a *batchRunTestAdapter) SearchBatch(_ context.Context, keywords []string, _ domain.ScrapeRequest) ([]domain.Job, error) { + a.batchCalls++ + a.keywords = append([]string(nil), keywords...) + return nil, nil +} + +func TestFilterAdaptersByCadenceAllowsJoobleTwicePerDay(t *testing.T) { + mr, err := miniredis.Run() + require.NoError(t, err) + t.Cleanup(mr.Close) + + rdb := redis.NewClient(&redis.Options{Addr: mr.Addr()}) + t.Cleanup(func() { _ = rdb.Close() }) + + ctx := context.Background() + adapterList := []adapters.Adapter{ + sourceScheduleTestAdapter{source: "Jooble"}, + sourceScheduleTestAdapter{source: "The Muse"}, + } + + firstRun := filterAdaptersByCadence(ctx, rdb, []adapters.Adapter{ + adapterList[0], + adapterList[1], + }) + require.Len(t, firstRun, 2) + + secondRun := filterAdaptersByCadence(ctx, rdb, []adapters.Adapter{ + adapterList[0], + adapterList[1], + }) + require.Len(t, secondRun, 1) + assert.Equal(t, "The Muse", secondRun[0].SourceName()) + + mr.FastForward(12*time.Hour + time.Second) + + thirdRun := filterAdaptersByCadence(ctx, rdb, []adapters.Adapter{ + adapterList[0], + adapterList[1], + }) + require.Len(t, thirdRun, 2) +} + +func TestShouldRunJoobleAllowsWhenRedisUnavailable(t *testing.T) { + assert.True(t, shouldRunJooble(context.Background(), nil)) +} + +func TestRunUsesBatchAdapterOnceForAllKeywords(t *testing.T) { + adapter := &batchRunTestAdapter{} + + Run(context.Background(), []adapters.Adapter{adapter}, domain.ScrapeRequest{ + Keywords: []string{"go", "java", "python"}, + }) + + assert.Equal(t, 0, adapter.searchCalls) + assert.Equal(t, 1, adapter.batchCalls) + assert.Equal(t, []string{"go", "java", "python"}, adapter.keywords) +} diff --git a/scraper-go/internal/ports/job_source.go b/scraper-go/internal/ports/job_source.go new file mode 100644 index 0000000..5641811 --- /dev/null +++ b/scraper-go/internal/ports/job_source.go @@ -0,0 +1,42 @@ +package ports + +import ( + "context" + "time" + + "github.com/Benevanio/Jobs_Scraper_Global/scraper-go/internal/domain" +) + +type JobSource interface { + SourceName() string + Search(ctx context.Context, keyword string, req domain.ScrapeRequest) ([]domain.Job, error) +} + +type BatchJobSource interface { + JobSource + SearchBatch(ctx context.Context, keywords []string, req domain.ScrapeRequest) ([]domain.Job, error) +} + +type JobRepository interface { + SaveBatch(ctx context.Context, jobs []domain.Job) (int, error) + GetAll(ctx context.Context) ([]domain.Job, error) + GetSample(ctx context.Context, limit int) ([]domain.Job, error) + Count(ctx context.Context) (int64, error) +} + +type KeywordRepository interface { + Load(ctx context.Context) ([]string, error) + Save(ctx context.Context, keywords []string) error +} + +type CacheRepository interface { + Get(ctx context.Context, key string, target any) (bool, error) + Set(ctx context.Context, key string, value any, ttl time.Duration) error + Delete(ctx context.Context, key string) error + SetNX(ctx context.Context, key string, value any, ttl time.Duration) (bool, error) +} + +type MetricsRecorder interface { + RecordSourceRun(source string, duration time.Duration, jobs int, err error) + RecordPipelineRun(duration time.Duration, jobs int) +}