mirror of
https://github.com/TheAlgorithms/Python.git
synced 2026-09-28 21:45:27 +08:00
* deps: migrate from httpx to httpx2 (pydantic's maintained fork) Mechanical rename of httpx -> httpx2 (API-compatible fork of httpx 0.28.1): pyproject.toml deps, PEP 723 inline-script headers, and all import/call sites. Excludes uv.lock (the keeper's allow-list rejects .lock files); the lock refresh needs a separate maintainer-merged PR. Refs #15081 * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * deps: drop tweepy + migrate remaining requests refs to httpx2 - maths/allocation_number.py: docstring example uses httpx2, not requests - web_programming/get_imdbtop.py.DISABLED: import httpx2 instead of requests - remove web_programming/get_user_tweets.py.DISABLED (a Twitter API how-to, not an algorithm) and drop the tweepy dependency that was its only user and the last high-level dep pulling in requests - uv.lock intentionally untouched (keeper allow-list) Refs #15081 --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
54 lines
1.5 KiB
Plaintext
54 lines
1.5 KiB
Plaintext
import bs4
|
|
import httpx2
|
|
|
|
|
|
def get_movie_data_from_soup(soup: bs4.element.ResultSet) -> dict[str, str]:
|
|
return {
|
|
"name": soup.h3.a.text,
|
|
"genre": soup.find("span", class_="genre").text.strip(),
|
|
"rating": soup.strong.text,
|
|
"page_link": f"https://www.imdb.com{soup.a.get('href')}",
|
|
}
|
|
|
|
|
|
def get_imdb_top_movies(num_movies: int = 5) -> tuple:
|
|
"""Get the top num_movies most highly rated movies from IMDB and
|
|
return a tuple of dicts describing each movie's name, genre, rating, and URL.
|
|
|
|
Args:
|
|
num_movies: The number of movies to get. Defaults to 5.
|
|
|
|
Returns:
|
|
A list of tuples containing information about the top n movies.
|
|
|
|
>>> len(get_imdb_top_movies(5))
|
|
5
|
|
>>> len(get_imdb_top_movies(-3))
|
|
0
|
|
>>> len(get_imdb_top_movies(4.99999))
|
|
4
|
|
"""
|
|
num_movies = int(float(num_movies))
|
|
if num_movies < 1:
|
|
return ()
|
|
base_url = (
|
|
"https://www.imdb.com/search/title?title_type="
|
|
f"feature&sort=num_votes,desc&count={num_movies}"
|
|
)
|
|
source = bs4.BeautifulSoup(httpx2.get(base_url).content, "html.parser")
|
|
return tuple(
|
|
get_movie_data_from_soup(movie)
|
|
for movie in source.find_all("div", class_="lister-item mode-advanced")
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import json
|
|
|
|
num_movies = int(input("How many movies would you like to see? "))
|
|
print(
|
|
", ".join(
|
|
json.dumps(movie, indent=4) for movie in get_imdb_top_movies(num_movies)
|
|
)
|
|
)
|