-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathscraper.py
More file actions
178 lines (154 loc) · 5.9 KB
/
Copy pathscraper.py
File metadata and controls
178 lines (154 loc) · 5.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
"""Fetches Substack Notes using the authenticated session cookie."""
import os
import httpx
from urllib.parse import unquote
from typing import Optional
def _headers(cookie: str) -> dict:
# Cookie may arrive URL-encoded from browser copy — decode it
decoded = unquote(cookie)
return {
"User-Agent": (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
),
"Accept": "application/json",
"Referer": "https://substack.com/notes",
"Cookie": f"substack.sid={decoded}",
}
def is_session_valid(cookie: str) -> bool:
"""
Probe an auth-gated endpoint to confirm the session cookie is still live.
The Notes feed itself returns a generic feed when logged out (no 401), so we
hit /subscriptions which returns 401 only when the session is invalid/expired.
"""
if not cookie:
return False
try:
with httpx.Client(follow_redirects=False) as client:
resp = client.get(
"https://substack.com/api/v1/subscriptions",
headers=_headers(cookie),
timeout=10,
)
# 401 = expired/invalid session. Anything else (200/400) = authenticated.
return resp.status_code != 401
except Exception:
# Network error — don't cry wolf about an expired cookie
return True
def fetch_notes_feed(
cookie: str,
limit: int = 80,
pages: int = 6,
) -> list[dict]:
"""Fetches from the authenticated reader feed, paginating up to `pages` times."""
all_items = []
cursor = None
with httpx.Client(follow_redirects=True) as client:
for _ in range(pages):
params = {"limit": min(limit, 25)}
if cursor:
params["cursor"] = cursor
resp = client.get(
"https://substack.com/api/v1/reader/feed",
params=params,
headers=_headers(cookie),
timeout=15,
)
resp.raise_for_status()
data = resp.json()
items = data.get("items", [])
if not items:
break
all_items.extend(items)
cursor = data.get("nextCursor")
if not cursor or len(all_items) >= limit:
break
return _normalize(all_items)
def fetch_publication_posts(subdomain: str, limit: int = 10) -> list[dict]:
"""Fetches recent posts from a specific Substack publication (no auth needed)."""
base = f"https://{subdomain}.substack.com"
with httpx.Client(follow_redirects=True) as client:
resp = client.get(
f"{base}/api/v1/posts",
params={"limit": limit},
headers={"User-Agent": "Mozilla/5.0", "Accept": "application/json"},
timeout=10,
)
if resp.status_code != 200:
return []
posts = resp.json()
return [
{
"id": p.get("id"),
"type": "post",
"author": subdomain,
"title": p.get("title", ""),
"body": p.get("truncated_body_text") or p.get("description") or p.get("subtitle", ""),
"url": p.get("canonical_url", ""),
"reaction_count": p.get("reaction_count", 0),
"comment_count": p.get("comment_count", 0),
"restacks": p.get("restacks", 0),
"post_date": p.get("post_date", ""),
"subdomain": subdomain,
"subscriber_count": None,
}
for p in posts
if p.get("type") == "newsletter"
]
def _normalize(items: list) -> list[dict]:
"""Normalises reader feed items into a consistent shape."""
results = []
for item in items:
# Only process notes (not reposts of articles etc.)
if item.get("context", {}).get("type") != "note":
continue
c = item.get("comment") or {}
pub = item.get("publication") or c.get("user_primary_publication") or {}
# Author name/handle sit directly on the comment object
author_name = c.get("name") or c.get("handle") or "unknown"
author_handle = c.get("handle") or ""
# Extract plain text from body_json paragraphs if body is empty
body = c.get("body") or ""
if not body and c.get("body_json"):
try:
bj = c["body_json"]
if isinstance(bj, str):
import json
bj = json.loads(bj)
body = " ".join(
n.get("text", "")
for block in bj.get("content", [])
for n in block.get("content", [])
if n.get("type") == "text"
)
except Exception:
pass
if not str(body).strip():
continue
# Build note URL — prefer a direct inbox link, fall back to author's publication
subdomain = pub.get("subdomain") or ""
custom_domain = pub.get("custom_domain") or ""
post_id = c.get("post_id")
note_id = c.get("id")
if note_id:
# Substack resolves this to the canonical profile note URL
note_url = f"https://substack.com/note/c-{note_id}"
elif subdomain:
note_url = f"https://{subdomain}.substack.com"
else:
note_url = ""
results.append({
"id": c.get("id"),
"type": "note",
"author": author_name,
"author_handle": author_handle,
"body": str(body)[:2000],
"url": note_url,
"reaction_count": c.get("reaction_count", 0),
"comment_count": c.get("children_count", 0),
"restacks": c.get("restacks", 0),
"post_date": c.get("date", ""),
"subdomain": subdomain,
"subscriber_count": pub.get("subscriber_count") or pub.get("subscriberCount"),
})
return results