Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,10 @@ scrape_uol:

scrape_folha:
docker compose exec scraper python scrape_no_openai.py --platform folha.uol.com.br


scrape_cartacapital:
docker compose exec scraper python scrape_no_openai.py --platform cartacapital.com.br

# Crawler para todos os portais ou específicos
crawl:
docker compose run --rm scraper python crawl.py
Expand Down Expand Up @@ -112,6 +115,9 @@ crawl_ig:
crawl_folha:
docker compose run scraper python crawl.py folhaspider

crawl_cartacapital:
docker compose run scraper python crawl.py cartacapitalspider

# Workflow completo de coleta de URLs
crawl_all_working:
@echo "Executando crawl de todos os portais funcionais..."
Expand All @@ -123,6 +129,7 @@ crawl_all_working:
@make crawl_aliadosBrasil
@make crawl_ig
@make crawl_folha
@make crawl_cartacapital
@echo "Crawl de todos os portais concluído!"

# Workflow completo de scraping
Expand All @@ -136,6 +143,7 @@ scrape_all_working:
@make scrape_r7
@make scrape_uol
@make scrape_folha
@make scrape_cartacapital
@echo "Scraping de todos os portais concluído!"

# Pipeline completo: crawl + scrape
Expand Down Expand Up @@ -182,6 +190,7 @@ help:
@echo " make crawl_aliadosBrasil - Coleta URLs do portal AliadosBrasil"
@echo " make crawl_ig - Coleta URLs do portal IG"
@echo " make crawl_folha - Coleta URLs do portal Folha"
@echo " make crawl_cartacapital - Coleta URLs do portal CartaCapital"
@echo ""
@echo "=== COMANDOS DE SCRAPING (Extração de Anúncios) ==="
@echo " make scrape_all_working - Executa scraping de todos os portais funcionais"
Expand All @@ -193,6 +202,7 @@ help:
@echo " make scrape_r7 - Scraping do portal R7"
@echo " make scrape_uol - Scraping do portal UOL"
@echo " make scrape_folha - Scraping do portal Folha"
@echo " make scrape_cartacapital - Scraping do portal CartaCapital"
@echo ""
@echo "=== WORKFLOWS COMPLETOS ==="
@echo " make pipeline_complete - Executa crawl + scraping de todos os portais"
Expand Down
2 changes: 1 addition & 1 deletion crawl.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,4 +54,4 @@ def run(spider_name=None):
if spider_name:
logger.info(f"Running spider: {spider_name}")
run(spider_name)
logger.info("Done!")
logger.info("Done!")
2 changes: 2 additions & 0 deletions create_db.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,8 @@
("Gazeta do Povo", "https://www.gazetadopovo.com.br/", "gazetadopovo"),
("Mais Goiás", "https://www.maisgoias.com.br/", "maisgoias"),
("Aliados Brasil", "https://www.aliadosbrasiloficial.com.br/", "aliadosbrasil"),
("Carta Capital", "https://www.cartacapital.com.br/", "cartacapital"),

]
portals_to_add = []
for portal in portals:
Expand Down
2 changes: 2 additions & 0 deletions plays/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from plays.gazetaDoPovo import GazetaDoPovoPlay
from plays.maisGoias import MaisGoias
from plays.aliadosBrasil import AliadosBrasilPlay
from plays.cartaCapital import CartaCapitalPlay

__all__ = [
ClicRBSPlay,
Expand All @@ -26,4 +27,5 @@
GazetaDoPovoPlay,
MaisGoias,
AliadosBrasilPlay,
CartaCapitalPlay,
]
199 changes: 199 additions & 0 deletions plays/cartaCapital.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,199 @@
from playwright.sync_api import sync_playwright

from plays.base import BasePlay
from plays.items import EntryItem
from plog import logger


class CartaCapitalPlay(BasePlay):
name = "cartacapital"

@classmethod
def match(cls, url: str) -> bool:
return "cartacapital.com.br" in url

def run(self) -> EntryItem:
"""
Abre a página da notícia, extrai título, descrição, corpo e tags
e devolve um EntryItem.
"""
with sync_playwright() as p:
browser = self.launch_browser(
p,
viewport={"width": 1366, "height": 768},
)
page = browser.new_page()

logger.info(f"[{self.name}] Opening URL '{self.url}'...")
page.goto(self.url, timeout=180_000, wait_until="networkidle")

title = self._extract_title(page)
description = self._extract_description(page)
body = self._extract_body(page)
tags = self._extract_tags(page)

browser.close()

return EntryItem(
title=title,
url=self.url,
description=description,
body=body,
tags=tags,
)

def _extract_title(self, page) -> str:
"""
CartaCapital tem título principal em <h1>.
"""
selectors = [
"article h1",
"main h1",
"h1.entry-title",
"h1",
]

for selector in selectors:
try:
locator = page.locator(selector)
if locator.count() > 0:
text = locator.first.inner_text().strip()
if text:
logger.info(f"[{self.name}] Title extracted: {text[:80]}...")
return text
except Exception as exc:
logger.warning(
f"[{self.name}] Failed to extract title with '{selector}': {exc}"
)

logger.error(f"[{self.name}] Failed to extract title")
return ""

def _extract_description(self, page) -> str:
"""
Subtítulo / linha de apoio (se existir).
Se não achar, cai no primeiro parágrafo do corpo.
"""
selectors = [
".subtitle",
".sub-title",
".lead",
".summary",
".excerpt",
]

for selector in selectors:
try:
locator = page.locator(selector)
if locator.count() > 0:
text = locator.first.inner_text().strip()
if text and len(text) > 20:
logger.info(
f"[{self.name}] Description extracted: {text[:80]}..."
)
return text
except Exception as exc:
logger.warning(
f"[{self.name}] Failed to extract description with '{selector}': {exc}"
)

first_paragraph = self._first_paragraph(page)
if first_paragraph:
logger.info(
f"[{self.name}] Description fallback from first paragraph: "
f"{first_paragraph[:80]}..."
)
return first_paragraph

def _first_paragraph(self, page) -> str:
"""
Pega o primeiro parágrafo "decente" do artigo.
"""
try:
locator = page.locator("article p")
if locator.count() == 0:
locator = page.locator("main article p")
if locator.count() == 0:
locator = page.locator("main p")

for i in range(locator.count()):
text = locator.nth(i).inner_text().strip()
if text and len(text) > 40:
return text
except Exception:
pass
return ""

def _extract_body(self, page) -> str:
"""
Extrai o texto completo do corpo da matéria.
"""
selectors = [
"article .entry-content",
"article .post-content",
"article .content",
"article",
"main article",
]

for selector in selectors:
try:
locator = page.locator(selector)
if locator.count() == 0:
continue

root = locator.first

try:
root.locator(
"script, style, .ad, .advertisement, nav, footer"
).evaluate_all("els => els.forEach(e => e.remove())")
except Exception:
pass

text = root.inner_text().strip()
if text and len(text) > 200:
logger.info(
f"[{self.name}] Body extracted ({len(text)} chars)"
)
return text
except Exception as exc:
logger.warning(
f"[{self.name}] Failed to extract body with '{selector}': {exc}"
)

logger.error(f"[{self.name}] Failed to extract body")
return ""

def _extract_tags(self, page) -> list[str]:
"""
Extrai tags/tópicos se houver bloco de tags.
Na CartaCapital geralmente aparecem como links em blocos de 'ENTENDA MAIS SOBRE', etc.
"""
selectors = [
".tags a",
".post-tags a",
".article-tags a",
".c-tags a",
"[rel='tag']",
]

tags: list[str] = []

for selector in selectors:
try:
loc = page.locator(selector)
count = loc.count()
if count == 0:
continue

for i in range(count):
tag_text = loc.nth(i).inner_text().strip()
if tag_text and tag_text not in tags:
tags.append(tag_text)
except Exception as exc:
logger.warning(
f"[{self.name}] Failed to extract tags with '{selector}': {exc}"
)

return tags
2 changes: 2 additions & 0 deletions spiders/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
from spiders.gazetaDoPovo import GazetaDoPovoSpider
from spiders.maisGoias import MaisGoiasSpider
from spiders.aliadosBrasil import AliadosBrasilSpider
from spiders.cartaCapital import CartaCapitalSpider

__all__ = [
BaseSpider,
Expand All @@ -28,4 +29,5 @@
GazetaDoPovoSpider,
MaisGoiasSpider,
AliadosBrasilSpider,
CartaCapitalSpider,
]
Loading