From f69c61b48338412c9c64e8400f17af5d4c2a8ec2 Mon Sep 17 00:00:00 2001 From: 314dro Date: Wed, 19 Nov 2025 03:52:03 -0300 Subject: [PATCH 1/3] =?UTF-8?q?feat:=20adiciona=20spider=20e=20play=20para?= =?UTF-8?q?=20Pol=C3=AAmica=20Para=C3=ADba?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Implementa PolemicaParaibaSpider com filtros de URL - Implementa PolemicaParaibaPlay para extração alternativa - Atualiza crawl.py para importar novo spider - Adiciona comandos no Makefile para crawl_polemicaparaiba e scrape_polemicaparaiba - Atualiza documentação de help no Makefile --- Makefile | 22 +++++- crawl.py | 6 +- plays/__init__.py | 2 + plays/polemicaparaiba.py | 142 +++++++++++++++++++++++++++++++++++++ spiders/__init__.py | 2 + spiders/polemicaparaiba.py | 103 +++++++++++++++++++++++++++ 6 files changed, 275 insertions(+), 2 deletions(-) create mode 100644 plays/polemicaparaiba.py create mode 100644 spiders/polemicaparaiba.py diff --git a/Makefile b/Makefile index 12663c4..01dc856 100644 --- a/Makefile +++ b/Makefile @@ -82,7 +82,13 @@ scrape_uol: scrape_folha: docker compose exec scraper python scrape_no_openai.py --platform folha.uol.com.br - + +scrape_cartacapital: + docker compose exec scraper python scrape_no_openai.py --platform cartacapital.com.br + +scrape_polemicaparaiba: + docker compose exec scraper python scrape_no_openai.py --platform polemicaparaiba.com.br + # Crawler para todos os portais ou específicos crawl: docker compose run --rm scraper python crawl.py @@ -112,6 +118,12 @@ crawl_ig: crawl_folha: docker compose run scraper python crawl.py folhaspider +crawl_cartacapital: + docker compose run scraper python crawl.py cartacapitalspider + +crawl_polemicaparaiba: + docker compose run scraper python crawl.py polemicaparaibaspider + # Workflow completo de coleta de URLs crawl_all_working: @echo "Executando crawl de todos os portais funcionais..." @@ -123,6 +135,8 @@ crawl_all_working: @make crawl_aliadosBrasil @make crawl_ig @make crawl_folha + @make crawl_cartacapital + @make crawl_polemicaparaiba @echo "Crawl de todos os portais concluído!" # Workflow completo de scraping @@ -136,6 +150,8 @@ scrape_all_working: @make scrape_r7 @make scrape_uol @make scrape_folha + @make scrape_cartacapital + @make scrape_polemicaparaiba @echo "Scraping de todos os portais concluído!" # Pipeline completo: crawl + scrape @@ -182,6 +198,8 @@ help: @echo " make crawl_aliadosBrasil - Coleta URLs do portal AliadosBrasil" @echo " make crawl_ig - Coleta URLs do portal IG" @echo " make crawl_folha - Coleta URLs do portal Folha" + @echo " make crawl_cartacapital - Coleta URLs do portal CartaCapital" + @echo " make crawl_polemicaparaiba - Coleta URLs do portal Polêmica Paraíba" @echo "" @echo "=== COMANDOS DE SCRAPING (Extração de Anúncios) ===" @echo " make scrape_all_working - Executa scraping de todos os portais funcionais" @@ -193,6 +211,8 @@ help: @echo " make scrape_r7 - Scraping do portal R7" @echo " make scrape_uol - Scraping do portal UOL" @echo " make scrape_folha - Scraping do portal Folha" + @echo " make scrape_cartacapital - Scraping do portal CartaCapital" + @echo " make scrape_polemicaparaiba - Scraping do portal Polêmica Paraíba" @echo "" @echo "=== WORKFLOWS COMPLETOS ===" @echo " make pipeline_complete - Executa crawl + scraping de todos os portais" diff --git a/crawl.py b/crawl.py index 2cf45e9..113f2cf 100755 --- a/crawl.py +++ b/crawl.py @@ -4,10 +4,12 @@ from plog import logger from spiders.base import BaseSpider +from spiders.polemicaparaiba import PolemicaParaibaSpider process = CrawlerProcess( settings={ - "ITEM_PIPELINES": {"pipelines.PostgresPipeline": 300}, + # Disable pipelines to avoid SQLAlchemy import hanging + "ITEM_PIPELINES": {}, # Add browser-like headers "DEFAULT_REQUEST_HEADERS": { "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", @@ -23,6 +25,8 @@ "RETRY_ENABLED": True, "RETRY_TIMES": 3, "RETRY_HTTP_CODES": [500, 502, 503, 504, 400, 403, 404, 408], + # Fix for import hanging - force AsyncIO reactor early + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", } ) diff --git a/plays/__init__.py b/plays/__init__.py index 338fe28..6455148 100644 --- a/plays/__init__.py +++ b/plays/__init__.py @@ -11,6 +11,7 @@ from plays.gazetaDoPovo import GazetaDoPovoPlay from plays.maisGoias import MaisGoias from plays.aliadosBrasil import AliadosBrasilPlay +from plays.cartaCapital import CartaCapitalPlay __all__ = [ ClicRBSPlay, @@ -26,4 +27,5 @@ GazetaDoPovoPlay, MaisGoias, AliadosBrasilPlay, + CartaCapitalPlay, ] diff --git a/plays/polemicaparaiba.py b/plays/polemicaparaiba.py new file mode 100644 index 0000000..f0d678c --- /dev/null +++ b/plays/polemicaparaiba.py @@ -0,0 +1,142 @@ +import requests +import re +from urllib.parse import urljoin, urlparse +from plays.base import BasePlay +from plays.items import URLItem + + +class PolemicaParaibaPlay(BasePlay): + """ + Play para extrair URLs do site polemicaparaiba.com.br + """ + + name = "polemicaparaiba" + start_urls = [ + "https://www.polemicaparaiba.com.br/", + "https://www.polemicaparaiba.com.br/politica/", + "https://www.polemicaparaiba.com.br/paraiba/", + "https://www.polemicaparaiba.com.br/brasil/", + "https://www.polemicaparaiba.com.br/entretenimento/", + ] + + def allow_url(self, url): + """ + Determina se uma URL deve ser incluída. + """ + parsed_url = urlparse(url) + + # Lista de padrões que devem ser bloqueados + blocked_patterns = [ + r'/wp-content/', + r'/wp-admin/', + r'/wp-json/', + r'/feed/', + r'/author/', + r'/tag/', + r'/category/', + r'/page/', + r'/search/', + r'\?', + r'#', + r'xmlrpc\.php', + r'\.xml$', + r'\.pdf$', + r'\.jpg$', + r'\.jpeg$', + r'\.png$', + r'\.gif$', + r'\.css$', + r'\.js$', + r'\.ico$', + ] + + # Verifica se a URL contém algum padrão bloqueado + for pattern in blocked_patterns: + if re.search(pattern, url, re.IGNORECASE): + return False + + # Verifica se é do domínio polemicaparaiba.com.br + if not parsed_url.netloc.endswith('polemicaparaiba.com.br'): + return False + + # Verifica se parece ser uma URL de artigo/notícia + path = parsed_url.path + + # Permitir URLs que parecem ser de notícias + path_segments = [seg for seg in path.strip('/').split('/') if seg] + + # Permitir páginas principais e de categoria + if len(path_segments) <= 1: + return True + + # Permitir URLs que parecem ser de artigos + if len(path_segments) >= 2 and not path.endswith('/'): + return True + + # Permitir URLs de categorias específicas + allowed_categories = [ + 'politica', 'paraiba', 'brasil', 'entretenimento', + 'esportes', 'economia', 'tecnologia', 'mundo' + ] + + if len(path_segments) >= 1 and path_segments[0] in allowed_categories: + return True + + return False + + def extract_urls_from_content(self, content, base_url): + """ + Extrai URLs do conteúdo HTML. + """ + urls = set() + + # Padrão para encontrar links href + href_pattern = r'href=["\']([^"\'>]+)["\']' + matches = re.findall(href_pattern, content, re.IGNORECASE) + + for match in matches: + # Converte links relativos em absolutos + absolute_url = urljoin(base_url, match) + + # Verifica se a URL deve ser incluída + if self.allow_url(absolute_url): + urls.add(absolute_url) + + return list(urls) + + def run(self): + """ + Executa o play para extrair URLs. + """ + all_urls = set() + + for start_url in self.start_urls: + try: + print(f"Processando: {start_url}") + + # Faz a requisição + response = requests.get( + start_url, + headers={ + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36' + }, + timeout=30 + ) + + if response.status_code == 200: + # Extrai URLs do conteúdo + urls = self.extract_urls_from_content(response.text, start_url) + all_urls.update(urls) + print(f"Encontradas {len(urls)} URLs em {start_url}") + else: + print(f"Erro {response.status_code} ao acessar {start_url}") + + except Exception as e: + print(f"Erro ao processar {start_url}: {e}") + + # Converte para lista de URLItem + url_items = [URLItem(url=url) for url in sorted(all_urls)] + + print(f"\nTotal de URLs extraídas: {len(url_items)}") + + return url_items diff --git a/spiders/__init__.py b/spiders/__init__.py index ecb5e1f..49baf3a 100644 --- a/spiders/__init__.py +++ b/spiders/__init__.py @@ -12,6 +12,7 @@ from spiders.gazetaDoPovo import GazetaDoPovoSpider from spiders.maisGoias import MaisGoiasSpider from spiders.aliadosBrasil import AliadosBrasilSpider +from spiders.cartaCapital import CartaCapitalSpider __all__ = [ BaseSpider, @@ -28,4 +29,5 @@ GazetaDoPovoSpider, MaisGoiasSpider, AliadosBrasilSpider, + CartaCapitalSpider, ] diff --git a/spiders/polemicaparaiba.py b/spiders/polemicaparaiba.py new file mode 100644 index 0000000..c51f665 --- /dev/null +++ b/spiders/polemicaparaiba.py @@ -0,0 +1,103 @@ +import re +from urllib.parse import urljoin, urlparse +from spiders.base import BaseSpider +from spiders.items import URLItem + + +class PolemicaParaibaSpider(BaseSpider): + name = "polemicaparaibaspider" + start_urls = [ + "https://www.polemicaparaiba.com.br/", + "https://www.polemicaparaiba.com.br/politica/", + "https://www.polemicaparaiba.com.br/paraiba/", + "https://www.polemicaparaiba.com.br/brasil/", + "https://www.polemicaparaiba.com.br/entretenimento/", + ] + + custom_settings = { + "DEPTH_LIMIT": 1, + "DOWNLOAD_DELAY": 1.5, + "RETRY_TIMES": 3, + "RETRY_HTTP_CODES": [500, 502, 503, 504, 400, 403, 404, 408], + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", + } + + def allow_url(self, url): + """ + Determina se uma URL deve ser incluída. + """ + parsed_url = urlparse(url) + + # Lista de padrões que devem ser bloqueados + blocked_patterns = [ + r'/wp-content/', + r'/wp-admin/', + r'/wp-json/', + r'/feed/', + r'/author/', + r'/tag/', + r'/category/', + r'/page/', + r'/search/', + r'\?', + r'#', + r'xmlrpc\.php', + r'\.xml$', + r'\.pdf$', + r'\.jpg$', + r'\.jpeg$', + r'\.png$', + r'\.gif$', + r'\.css$', + r'\.js$', + r'\.ico$', + ] + + # Verifica se a URL contém algum padrão bloqueado + for pattern in blocked_patterns: + if re.search(pattern, url, re.IGNORECASE): + return False + + # Verifica se é do domínio polemicaparaiba.com.br + if not parsed_url.netloc.endswith('polemicaparaiba.com.br'): + return False + + # Verifica se parece ser uma URL de artigo/notícia + # URLs de artigos geralmente têm uma estrutura específica + path = parsed_url.path + + # Permitir URLs que parecem ser de notícias (com pelo menos 3 segmentos no path) + path_segments = [seg for seg in path.strip('/').split('/') if seg] + + # Permitir páginas principais e de categoria + if len(path_segments) <= 1: + return True + + # Permitir URLs que parecem ser de artigos + # (com pelo menos 2 segmentos e não terminando em /) + if len(path_segments) >= 2 and not path.endswith('/'): + return True + + # Permitir URLs de categorias específicas + allowed_categories = [ + 'politica', 'paraiba', 'brasil', 'entretenimento', + 'esportes', 'economia', 'tecnologia', 'mundo' + ] + + if len(path_segments) >= 1 and path_segments[0] in allowed_categories: + return True + + return False + + def parse(self, response): + # Extrair todos os links da página + links = response.css('a::attr(href)').getall() + + for link in links: + if link: + # Converte links relativos em absolutos + absolute_url = urljoin(response.url, link) + + # Verifica se a URL deve ser incluída + if self.allow_url(absolute_url): + yield URLItem(url=absolute_url) From 693ae42937b5d24a04a8692d471baa13976eda0a Mon Sep 17 00:00:00 2001 From: 314dro Date: Wed, 19 Nov 2025 04:07:07 -0300 Subject: [PATCH 2/3] feat: spider e play do Carta Capital --- create_db.py | 2 + plays/cartaCapital.py | 199 ++++++++++++++++++++++++++++++++++++++++ spiders/cartaCapital.py | 113 +++++++++++++++++++++++ 3 files changed, 314 insertions(+) create mode 100644 plays/cartaCapital.py create mode 100644 spiders/cartaCapital.py diff --git a/create_db.py b/create_db.py index 617b85b..778ef95 100644 --- a/create_db.py +++ b/create_db.py @@ -42,6 +42,8 @@ ("Gazeta do Povo", "https://www.gazetadopovo.com.br/", "gazetadopovo"), ("Mais Goiás", "https://www.maisgoias.com.br/", "maisgoias"), ("Aliados Brasil", "https://www.aliadosbrasiloficial.com.br/", "aliadosbrasil"), + ("Carta Capital", "https://www.cartacapital.com.br/", "cartacapital"), + ] portals_to_add = [] for portal in portals: diff --git a/plays/cartaCapital.py b/plays/cartaCapital.py new file mode 100644 index 0000000..b0764f0 --- /dev/null +++ b/plays/cartaCapital.py @@ -0,0 +1,199 @@ +from playwright.sync_api import sync_playwright + +from plays.base import BasePlay +from plays.items import EntryItem +from plog import logger + + +class CartaCapitalPlay(BasePlay): + name = "cartacapital" + + @classmethod + def match(cls, url: str) -> bool: + return "cartacapital.com.br" in url + + def run(self) -> EntryItem: + """ + Abre a página da notícia, extrai título, descrição, corpo e tags + e devolve um EntryItem. + """ + with sync_playwright() as p: + browser = self.launch_browser( + p, + viewport={"width": 1366, "height": 768}, + ) + page = browser.new_page() + + logger.info(f"[{self.name}] Opening URL '{self.url}'...") + page.goto(self.url, timeout=180_000, wait_until="networkidle") + + title = self._extract_title(page) + description = self._extract_description(page) + body = self._extract_body(page) + tags = self._extract_tags(page) + + browser.close() + + return EntryItem( + title=title, + url=self.url, + description=description, + body=body, + tags=tags, + ) + + def _extract_title(self, page) -> str: + """ + CartaCapital tem título principal em

. + """ + selectors = [ + "article h1", + "main h1", + "h1.entry-title", + "h1", + ] + + for selector in selectors: + try: + locator = page.locator(selector) + if locator.count() > 0: + text = locator.first.inner_text().strip() + if text: + logger.info(f"[{self.name}] Title extracted: {text[:80]}...") + return text + except Exception as exc: + logger.warning( + f"[{self.name}] Failed to extract title with '{selector}': {exc}" + ) + + logger.error(f"[{self.name}] Failed to extract title") + return "" + + def _extract_description(self, page) -> str: + """ + Subtítulo / linha de apoio (se existir). + Se não achar, cai no primeiro parágrafo do corpo. + """ + selectors = [ + ".subtitle", + ".sub-title", + ".lead", + ".summary", + ".excerpt", + ] + + for selector in selectors: + try: + locator = page.locator(selector) + if locator.count() > 0: + text = locator.first.inner_text().strip() + if text and len(text) > 20: + logger.info( + f"[{self.name}] Description extracted: {text[:80]}..." + ) + return text + except Exception as exc: + logger.warning( + f"[{self.name}] Failed to extract description with '{selector}': {exc}" + ) + + first_paragraph = self._first_paragraph(page) + if first_paragraph: + logger.info( + f"[{self.name}] Description fallback from first paragraph: " + f"{first_paragraph[:80]}..." + ) + return first_paragraph + + def _first_paragraph(self, page) -> str: + """ + Pega o primeiro parágrafo "decente" do artigo. + """ + try: + locator = page.locator("article p") + if locator.count() == 0: + locator = page.locator("main article p") + if locator.count() == 0: + locator = page.locator("main p") + + for i in range(locator.count()): + text = locator.nth(i).inner_text().strip() + if text and len(text) > 40: + return text + except Exception: + pass + return "" + + def _extract_body(self, page) -> str: + """ + Extrai o texto completo do corpo da matéria. + """ + selectors = [ + "article .entry-content", + "article .post-content", + "article .content", + "article", + "main article", + ] + + for selector in selectors: + try: + locator = page.locator(selector) + if locator.count() == 0: + continue + + root = locator.first + + try: + root.locator( + "script, style, .ad, .advertisement, nav, footer" + ).evaluate_all("els => els.forEach(e => e.remove())") + except Exception: + pass + + text = root.inner_text().strip() + if text and len(text) > 200: + logger.info( + f"[{self.name}] Body extracted ({len(text)} chars)" + ) + return text + except Exception as exc: + logger.warning( + f"[{self.name}] Failed to extract body with '{selector}': {exc}" + ) + + logger.error(f"[{self.name}] Failed to extract body") + return "" + + def _extract_tags(self, page) -> list[str]: + """ + Extrai tags/tópicos se houver bloco de tags. + Na CartaCapital geralmente aparecem como links em blocos de 'ENTENDA MAIS SOBRE', etc. + """ + selectors = [ + ".tags a", + ".post-tags a", + ".article-tags a", + ".c-tags a", + "[rel='tag']", + ] + + tags: list[str] = [] + + for selector in selectors: + try: + loc = page.locator(selector) + count = loc.count() + if count == 0: + continue + + for i in range(count): + tag_text = loc.nth(i).inner_text().strip() + if tag_text and tag_text not in tags: + tags.append(tag_text) + except Exception as exc: + logger.warning( + f"[{self.name}] Failed to extract tags with '{selector}': {exc}" + ) + + return tags diff --git a/spiders/cartaCapital.py b/spiders/cartaCapital.py new file mode 100644 index 0000000..499ff3d --- /dev/null +++ b/spiders/cartaCapital.py @@ -0,0 +1,113 @@ +import scrapy +from urllib.parse import urlparse + +from spiders.base import BaseSpider +from spiders.items import URLItem + + +class CartaCapitalSpider(BaseSpider): + name = "cartacapitalspider" + allowed_domains = ["www.cartacapital.com.br", "cartacapital.com.br"] + start_urls = [ + "https://www.cartacapital.com.br/", + "https://www.cartacapital.com.br/mais-recentes/", + ] + + custom_settings = { + **BaseSpider.custom_settings, + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", + "COOKIES_ENABLED": True, + "DOWNLOAD_DELAY": 1.5, + "DEFAULT_REQUEST_HEADERS": { + "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8", + "Accept-Language": "pt-BR,pt;q=0.9,en-US;q=0.8,en;q=0.7", + "User-Agent": ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/121.0 Safari/537.36" + ), + }, + } + + def allow_url(self, url: str) -> bool: + p = urlparse(url) + + if p.netloc not in ("www.cartacapital.com.br", "cartacapital.com.br"): + return False + + path = p.path.rstrip("/") or "/" + + if path == "/": + return False + + blacklist_prefixes = ( + "/tag/", + "/tags/", + "/author/", + "/autores/", + "/assine", + "/manifesto", + "/princípios", + "/principios", + "/expediente", + "/sobre-nos", + "/media-kit", + "/newsletter", + "/wp-content", + "/wp-json", + "/wp-admin", + "/central-de-ajuda", + ) + if any(path.startswith(prefix) for prefix in blacklist_prefixes): + return False + + if path.endswith(".pdf"): + return False + + segments = [seg for seg in path.split("/") if seg] + if len(segments) < 2: + return False + + slug = segments[-1] + if slug.count("-") < 2 and len(slug) < 15: + return False + + return True + + def start_requests(self): + for url in self.start_urls: + yield scrapy.Request( + url, + callback=self.parse, + dont_filter=True, + meta={ + "dont_redirect": True, + "handle_httpstatus_list": [403, 404], + }, + ) + + def parse(self, response: scrapy.http.Response): + seen_urls: set[str] = set() + + selectors = [ + "a.h-education__item::attr(href)", + "main a[href*='cartacapital.com.br']::attr(href)", + "article a[href]::attr(href)", + "h2 a[href]::attr(href)", + "h3 a[href]::attr(href)", + ] + + for selector in selectors: + for href in response.css(selector).getall(): + if not href: + continue + + full_url = response.urljoin(href) + full_url = full_url.split("#", 1)[0].split("?", 1)[0] + + if full_url in seen_urls: + continue + seen_urls.add(full_url) + + if self.allow_url(full_url): + yield URLItem(url=full_url) From 256c1b91ccb295691bd870caf90be440f32a2904 Mon Sep 17 00:00:00 2001 From: 314dro Date: Wed, 3 Dec 2025 08:45:21 -0300 Subject: [PATCH 3/3] (Bug): Corrigindo duplicidade e erros --- Makefile | 10 --- crawl.py | 8 +-- plays/polemicaparaiba.py | 142 ------------------------------------- spiders/polemicaparaiba.py | 103 --------------------------- 4 files changed, 2 insertions(+), 261 deletions(-) delete mode 100644 plays/polemicaparaiba.py delete mode 100644 spiders/polemicaparaiba.py diff --git a/Makefile b/Makefile index 01dc856..1eb7dd7 100644 --- a/Makefile +++ b/Makefile @@ -86,9 +86,6 @@ scrape_folha: scrape_cartacapital: docker compose exec scraper python scrape_no_openai.py --platform cartacapital.com.br -scrape_polemicaparaiba: - docker compose exec scraper python scrape_no_openai.py --platform polemicaparaiba.com.br - # Crawler para todos os portais ou específicos crawl: docker compose run --rm scraper python crawl.py @@ -121,9 +118,6 @@ crawl_folha: crawl_cartacapital: docker compose run scraper python crawl.py cartacapitalspider -crawl_polemicaparaiba: - docker compose run scraper python crawl.py polemicaparaibaspider - # Workflow completo de coleta de URLs crawl_all_working: @echo "Executando crawl de todos os portais funcionais..." @@ -136,7 +130,6 @@ crawl_all_working: @make crawl_ig @make crawl_folha @make crawl_cartacapital - @make crawl_polemicaparaiba @echo "Crawl de todos os portais concluído!" # Workflow completo de scraping @@ -151,7 +144,6 @@ scrape_all_working: @make scrape_uol @make scrape_folha @make scrape_cartacapital - @make scrape_polemicaparaiba @echo "Scraping de todos os portais concluído!" # Pipeline completo: crawl + scrape @@ -199,7 +191,6 @@ help: @echo " make crawl_ig - Coleta URLs do portal IG" @echo " make crawl_folha - Coleta URLs do portal Folha" @echo " make crawl_cartacapital - Coleta URLs do portal CartaCapital" - @echo " make crawl_polemicaparaiba - Coleta URLs do portal Polêmica Paraíba" @echo "" @echo "=== COMANDOS DE SCRAPING (Extração de Anúncios) ===" @echo " make scrape_all_working - Executa scraping de todos os portais funcionais" @@ -212,7 +203,6 @@ help: @echo " make scrape_uol - Scraping do portal UOL" @echo " make scrape_folha - Scraping do portal Folha" @echo " make scrape_cartacapital - Scraping do portal CartaCapital" - @echo " make scrape_polemicaparaiba - Scraping do portal Polêmica Paraíba" @echo "" @echo "=== WORKFLOWS COMPLETOS ===" @echo " make pipeline_complete - Executa crawl + scraping de todos os portais" diff --git a/crawl.py b/crawl.py index 113f2cf..7cc6b98 100755 --- a/crawl.py +++ b/crawl.py @@ -4,12 +4,10 @@ from plog import logger from spiders.base import BaseSpider -from spiders.polemicaparaiba import PolemicaParaibaSpider process = CrawlerProcess( settings={ - # Disable pipelines to avoid SQLAlchemy import hanging - "ITEM_PIPELINES": {}, + "ITEM_PIPELINES": {"pipelines.PostgresPipeline": 300}, # Add browser-like headers "DEFAULT_REQUEST_HEADERS": { "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", @@ -25,8 +23,6 @@ "RETRY_ENABLED": True, "RETRY_TIMES": 3, "RETRY_HTTP_CODES": [500, 502, 503, 504, 400, 403, 404, 408], - # Fix for import hanging - force AsyncIO reactor early - "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", } ) @@ -58,4 +54,4 @@ def run(spider_name=None): if spider_name: logger.info(f"Running spider: {spider_name}") run(spider_name) - logger.info("Done!") + logger.info("Done!") \ No newline at end of file diff --git a/plays/polemicaparaiba.py b/plays/polemicaparaiba.py deleted file mode 100644 index f0d678c..0000000 --- a/plays/polemicaparaiba.py +++ /dev/null @@ -1,142 +0,0 @@ -import requests -import re -from urllib.parse import urljoin, urlparse -from plays.base import BasePlay -from plays.items import URLItem - - -class PolemicaParaibaPlay(BasePlay): - """ - Play para extrair URLs do site polemicaparaiba.com.br - """ - - name = "polemicaparaiba" - start_urls = [ - "https://www.polemicaparaiba.com.br/", - "https://www.polemicaparaiba.com.br/politica/", - "https://www.polemicaparaiba.com.br/paraiba/", - "https://www.polemicaparaiba.com.br/brasil/", - "https://www.polemicaparaiba.com.br/entretenimento/", - ] - - def allow_url(self, url): - """ - Determina se uma URL deve ser incluída. - """ - parsed_url = urlparse(url) - - # Lista de padrões que devem ser bloqueados - blocked_patterns = [ - r'/wp-content/', - r'/wp-admin/', - r'/wp-json/', - r'/feed/', - r'/author/', - r'/tag/', - r'/category/', - r'/page/', - r'/search/', - r'\?', - r'#', - r'xmlrpc\.php', - r'\.xml$', - r'\.pdf$', - r'\.jpg$', - r'\.jpeg$', - r'\.png$', - r'\.gif$', - r'\.css$', - r'\.js$', - r'\.ico$', - ] - - # Verifica se a URL contém algum padrão bloqueado - for pattern in blocked_patterns: - if re.search(pattern, url, re.IGNORECASE): - return False - - # Verifica se é do domínio polemicaparaiba.com.br - if not parsed_url.netloc.endswith('polemicaparaiba.com.br'): - return False - - # Verifica se parece ser uma URL de artigo/notícia - path = parsed_url.path - - # Permitir URLs que parecem ser de notícias - path_segments = [seg for seg in path.strip('/').split('/') if seg] - - # Permitir páginas principais e de categoria - if len(path_segments) <= 1: - return True - - # Permitir URLs que parecem ser de artigos - if len(path_segments) >= 2 and not path.endswith('/'): - return True - - # Permitir URLs de categorias específicas - allowed_categories = [ - 'politica', 'paraiba', 'brasil', 'entretenimento', - 'esportes', 'economia', 'tecnologia', 'mundo' - ] - - if len(path_segments) >= 1 and path_segments[0] in allowed_categories: - return True - - return False - - def extract_urls_from_content(self, content, base_url): - """ - Extrai URLs do conteúdo HTML. - """ - urls = set() - - # Padrão para encontrar links href - href_pattern = r'href=["\']([^"\'>]+)["\']' - matches = re.findall(href_pattern, content, re.IGNORECASE) - - for match in matches: - # Converte links relativos em absolutos - absolute_url = urljoin(base_url, match) - - # Verifica se a URL deve ser incluída - if self.allow_url(absolute_url): - urls.add(absolute_url) - - return list(urls) - - def run(self): - """ - Executa o play para extrair URLs. - """ - all_urls = set() - - for start_url in self.start_urls: - try: - print(f"Processando: {start_url}") - - # Faz a requisição - response = requests.get( - start_url, - headers={ - 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36' - }, - timeout=30 - ) - - if response.status_code == 200: - # Extrai URLs do conteúdo - urls = self.extract_urls_from_content(response.text, start_url) - all_urls.update(urls) - print(f"Encontradas {len(urls)} URLs em {start_url}") - else: - print(f"Erro {response.status_code} ao acessar {start_url}") - - except Exception as e: - print(f"Erro ao processar {start_url}: {e}") - - # Converte para lista de URLItem - url_items = [URLItem(url=url) for url in sorted(all_urls)] - - print(f"\nTotal de URLs extraídas: {len(url_items)}") - - return url_items diff --git a/spiders/polemicaparaiba.py b/spiders/polemicaparaiba.py deleted file mode 100644 index c51f665..0000000 --- a/spiders/polemicaparaiba.py +++ /dev/null @@ -1,103 +0,0 @@ -import re -from urllib.parse import urljoin, urlparse -from spiders.base import BaseSpider -from spiders.items import URLItem - - -class PolemicaParaibaSpider(BaseSpider): - name = "polemicaparaibaspider" - start_urls = [ - "https://www.polemicaparaiba.com.br/", - "https://www.polemicaparaiba.com.br/politica/", - "https://www.polemicaparaiba.com.br/paraiba/", - "https://www.polemicaparaiba.com.br/brasil/", - "https://www.polemicaparaiba.com.br/entretenimento/", - ] - - custom_settings = { - "DEPTH_LIMIT": 1, - "DOWNLOAD_DELAY": 1.5, - "RETRY_TIMES": 3, - "RETRY_HTTP_CODES": [500, 502, 503, 504, 400, 403, 404, 408], - "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", - } - - def allow_url(self, url): - """ - Determina se uma URL deve ser incluída. - """ - parsed_url = urlparse(url) - - # Lista de padrões que devem ser bloqueados - blocked_patterns = [ - r'/wp-content/', - r'/wp-admin/', - r'/wp-json/', - r'/feed/', - r'/author/', - r'/tag/', - r'/category/', - r'/page/', - r'/search/', - r'\?', - r'#', - r'xmlrpc\.php', - r'\.xml$', - r'\.pdf$', - r'\.jpg$', - r'\.jpeg$', - r'\.png$', - r'\.gif$', - r'\.css$', - r'\.js$', - r'\.ico$', - ] - - # Verifica se a URL contém algum padrão bloqueado - for pattern in blocked_patterns: - if re.search(pattern, url, re.IGNORECASE): - return False - - # Verifica se é do domínio polemicaparaiba.com.br - if not parsed_url.netloc.endswith('polemicaparaiba.com.br'): - return False - - # Verifica se parece ser uma URL de artigo/notícia - # URLs de artigos geralmente têm uma estrutura específica - path = parsed_url.path - - # Permitir URLs que parecem ser de notícias (com pelo menos 3 segmentos no path) - path_segments = [seg for seg in path.strip('/').split('/') if seg] - - # Permitir páginas principais e de categoria - if len(path_segments) <= 1: - return True - - # Permitir URLs que parecem ser de artigos - # (com pelo menos 2 segmentos e não terminando em /) - if len(path_segments) >= 2 and not path.endswith('/'): - return True - - # Permitir URLs de categorias específicas - allowed_categories = [ - 'politica', 'paraiba', 'brasil', 'entretenimento', - 'esportes', 'economia', 'tecnologia', 'mundo' - ] - - if len(path_segments) >= 1 and path_segments[0] in allowed_categories: - return True - - return False - - def parse(self, response): - # Extrair todos os links da página - links = response.css('a::attr(href)').getall() - - for link in links: - if link: - # Converte links relativos em absolutos - absolute_url = urljoin(response.url, link) - - # Verifica se a URL deve ser incluída - if self.allow_url(absolute_url): - yield URLItem(url=absolute_url)