Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,9 @@ scrape_uol:
scrape_folha:
docker compose exec scraper python scrape_no_openai.py --platform folha.uol.com.br

scrape_congressoemfoco:
docker compose exec scraper python scrape_no_openai.py --platform congressoemfoco.com.br

# Crawler para todos os portais ou específicos
crawl:
docker compose run --rm scraper python crawl.py
Expand Down Expand Up @@ -112,6 +115,9 @@ crawl_ig:
crawl_folha:
docker compose run scraper python crawl.py folhaspider

crawl_congressoemfoco:
docker compose run scraper python crawl.py congressoemfocospider

# Workflow completo de coleta de URLs
crawl_all_working:
@echo "Executando crawl de todos os portais funcionais..."
Expand All @@ -123,6 +129,7 @@ crawl_all_working:
@make crawl_aliadosBrasil
@make crawl_ig
@make crawl_folha
@make crawl_congressoemfoco
@echo "Crawl de todos os portais concluído!"

# Workflow completo de scraping
Expand All @@ -136,6 +143,7 @@ scrape_all_working:
@make scrape_r7
@make scrape_uol
@make scrape_folha
@make scrape_congressoemfoco
@echo "Scraping de todos os portais concluído!"

# Pipeline completo: crawl + scrape
Expand Down Expand Up @@ -182,6 +190,7 @@ help:
@echo " make crawl_aliadosBrasil - Coleta URLs do portal AliadosBrasil"
@echo " make crawl_ig - Coleta URLs do portal IG"
@echo " make crawl_folha - Coleta URLs do portal Folha"
@echo " make crawl_congressoemfoco - Coleta URLs do portal Congresso em Foco"
@echo ""
@echo "=== COMANDOS DE SCRAPING (Extração de Anúncios) ==="
@echo " make scrape_all_working - Executa scraping de todos os portais funcionais"
Expand All @@ -193,6 +202,7 @@ help:
@echo " make scrape_r7 - Scraping do portal R7"
@echo " make scrape_uol - Scraping do portal UOL"
@echo " make scrape_folha - Scraping do portal Folha"
@echo " make scrape_congressoemfoco - Scraping do portal Congresso em Foco"
@echo ""
@echo "=== WORKFLOWS COMPLETOS ==="
@echo " make pipeline_complete - Executa crawl + scraping de todos os portais"
Expand Down
1 change: 1 addition & 0 deletions create_db.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@
("Gazeta do Povo", "https://www.gazetadopovo.com.br/", "gazetadopovo"),
("Mais Goiás", "https://www.maisgoias.com.br/", "maisgoias"),
("Aliados Brasil", "https://www.aliadosbrasiloficial.com.br/", "aliadosbrasil"),
("Congresso em Foco", "https://www.congressoemfoco.com.br/", "congressoemfoco"),
]
portals_to_add = []
for portal in portals:
Expand Down
2 changes: 2 additions & 0 deletions plays/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from plays.gazetaDoPovo import GazetaDoPovoPlay
from plays.maisGoias import MaisGoias
from plays.aliadosBrasil import AliadosBrasilPlay
from plays.congressoEmFoco import CongressoEmFocoPlay

__all__ = [
ClicRBSPlay,
Expand All @@ -26,4 +27,5 @@
GazetaDoPovoPlay,
MaisGoias,
AliadosBrasilPlay,
CongressoEmFocoPlay,
]
77 changes: 77 additions & 0 deletions plays/congressoEmFoco.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
import time

from playwright.sync_api import sync_playwright

from plays.base import BasePlay
from plays.items import EntryItem
from plog import logger


class CongressoEmFocoPlay(BasePlay):
name = "congressoemfoco"

@classmethod
def match(cls, url):
return "congressoemfoco.com.br" in url

def pre_run(self):
pass

def run(self) -> EntryItem:
with sync_playwright() as p:
browser = self.launch_browser(p, viewport={"width": 1920, "height": 1080})
page = browser.new_page()
logger.info(f"[{self.name}] Opening URL '{self.url}'...")
page.goto(self.url, timeout=180_000)

# Wait for the main content to be visible
page.wait_for_selector("h1", timeout=30000)

# Extract article title
entry_title = ""
try:
entry_title = page.locator("h1").first.inner_text()
except Exception as e:
logger.warning(f"[{self.name}] Failed to extract title: {str(e)}")

# Extract description/subtitle
description = ""
try:
description = page.locator("h2.asset__summary").first.inner_text()
except Exception as e:
logger.warning(f"[{self.name}] Failed to extract description: {str(e)}")

# Extract article body
body = ""
try:
body_element = page.locator("div.asset__content div.html-content")
if body_element.count() > 0:
body = body_element.inner_text()
if not body.strip():
logger.warning(f"[{self.name}] Extracted body is empty")
else:
logger.warning(f"[{self.name}] Could not find body element")
except Exception as e:
logger.warning(f"[{self.name}] Failed to extract body: {str(e)}")

# Extract tags
tags = []
try:
tag_elements = page.locator("div.asset__tags-links a.asset__tags-link")
for i in range(tag_elements.count()):
tag_text = tag_elements.nth(i).inner_text().strip()
if tag_text:
tags.append(tag_text)
except Exception as e:
logger.warning(f"[{self.name}] Failed to extract tags: {str(e)}")

return EntryItem(
title=entry_title,
url=self.url,
description=description,
body=body,
tags=tags,
)

def post_run(self, output):
return output
2 changes: 2 additions & 0 deletions spiders/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
from spiders.gazetaDoPovo import GazetaDoPovoSpider
from spiders.maisGoias import MaisGoiasSpider
from spiders.aliadosBrasil import AliadosBrasilSpider
from spiders.congressoEmFoco import CongressoEmFocoSpider

__all__ = [
BaseSpider,
Expand All @@ -28,4 +29,5 @@
GazetaDoPovoSpider,
MaisGoiasSpider,
AliadosBrasilSpider,
CongressoEmFocoSpider,
]
106 changes: 106 additions & 0 deletions spiders/congressoEmFoco.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
import scrapy

from spiders.base import BaseSpider
from spiders.items import URLItem
from urllib.parse import urlparse


class CongressoEmFocoSpider(BaseSpider):
name = "congressoemfocospider"
start_urls = ["https://www.congressoemfoco.com.br/"]
allowed_domains = ["congressoemfoco.com.br"]

custom_settings = {
**BaseSpider.custom_settings,
"COOKIES_ENABLED": True,
"DOWNLOAD_DELAY": 3,
"DEFAULT_REQUEST_HEADERS": {
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language": "pt-BR,pt;q=0.9,en-US;q=0.8,en;q=0.7",
"Accept-Encoding": "gzip, deflate, br",
"Connection": "keep-alive",
"Upgrade-Insecure-Requests": "1",
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36",
"Cache-Control": "max-age=0",
},
"DOWNLOAD_HANDLERS": {
"https": "scrapy_playwright.handler.ScrapyPlaywrightDownloadHandler",
},
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
"PLAYWRIGHT_BROWSER_TYPE": "firefox",
}

def allow_url(self, url: str) -> bool:
# Validate domain
if not url or "congressoemfoco.com.br" not in url:
return False

p = urlparse(url)
path = p.path.rstrip("/")

# Blacklist sections
blacklist = {
"/videos",
"/podcasts",
"/newsletter",
"/sobre",
"/contato",
"/politica-de-privacidade",
"/termos-de-uso",
"/assine",
"/busca",
}

if path in blacklist:
self.logger.info(f"Blacklisted URL: {url}")
return False

# Validate news patterns
if "/noticia/" not in path and "/artigo/" not in path:
return False

# Validate segments
segments = [seg for seg in path.split("/") if seg]
if len(segments) < 2:
self.logger.info(f"URL with less than 2 segments: {url}")
return False

slug = segments[-1]
# Long slugs by hyphens or by character length
if slug.count("-") >= 3 or len(slug) > 30:
return True

return False

def start_requests(self):
for url in self.start_urls:
yield scrapy.Request(
url,
callback=self.parse,
dont_filter=True,
meta={
"playwright": True,
"playwright_include_page": True,
"playwright_page_goto_kwargs": {"wait_until": "networkidle"},
},
)

def parse(self, response):
page = response.meta.get("playwright_page")
if page:
page.close()

# Extract all links from page
links = response.css("a::attr(href)").getall()
self.logger.info(f"Found {len(links)} links under homepage")

for raw in links:
if not raw:
continue

# Build absolute URL, then strip fragments/queries
full = response.urljoin(raw)
full = full.split("#", 1)[0].split("?", 1)[0]

if self.allow_url(full):
yield URLItem(url=full)