-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathcastScraper.py
More file actions
77 lines (57 loc) · 2.08 KB
/
Copy pathcastScraper.py
File metadata and controls
77 lines (57 loc) · 2.08 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
## Main Project ##
#castScraper by Enver Uslu [6-27-2023]
#I made this project for my programming portfolio,
#educational use and practice for web scraping only.
#github.com/enverUslu
from urllib.request import urlopen
from bs4 import BeautifulSoup
import pandas as pd
import requests
from imdb import IMDb
# Creating an IMDb object
ia = IMDb()
# Asking the user for a movie name
filmName = input("Enter a movie name: ")
# Searching for movies using the given name
movies = ia.search_movie(filmName)
# Checking if any movies are found - hoping they are.
if len(movies) > 0:
# Get the first movie from the search results
movie = movies[0]
# Get the IMDb ID of the movie
imdb_id = movie.getID()
# Construct the IMDb URL using the IMDb ID
url = f"https://www.imdb.com/title/tt{imdb_id}/fullcredits"
# Print the URL
print("URL:", url)
else:
print("No movies found.")
if movie is not None:
# Get the title of the movie
title = movie.get('title')
# Display the movie's name
print("Movie Title:", title)
print("----------")
else:
print("Movie not found.")
print("----------")
print("--CAST--")
# URL of the page to scrape
# Send a GET request to the URL
response = requests.get(url)
# Create a BeautifulSoup object with the response text
soup = BeautifulSoup(response.text, 'html.parser')
# Find all the table rows containing actor information
actor_rows = soup.select('tr.odd, tr.even')
# Loop through the actor rows and extract the actor names and roles
for row in actor_rows:
# Extract actor name
actor_name_element = row.select_one('td.primary_photo a img')
actor_name = actor_name_element.get('alt') if actor_name_element else "N/A"
# Extract actor role
actor_role_element = row.select_one('td.character a')
actor_role = actor_role_element.text.strip() if actor_role_element else "N/A"
if actor_name != "N/A" and actor_role != "N/A":
print(f"Actor: {actor_name}")
print(f"Role: {actor_role}")
print("----------")