Files
Python/web_programming/get_amazon_product_data.py
priya-sundaram-devandpre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> bfa655ecea deps: migrate from httpx to httpx2 (pydantic's maintained fork) (#15192)
* deps: migrate from httpx to httpx2 (pydantic's maintained fork)

Mechanical rename of httpx -> httpx2 (API-compatible fork of httpx 0.28.1):
pyproject.toml deps, PEP 723 inline-script headers, and all import/call sites.
Excludes uv.lock (the keeper's allow-list rejects .lock files); the lock
refresh needs a separate maintainer-merged PR.

Refs #15081

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* deps: drop tweepy + migrate remaining requests refs to httpx2

- maths/allocation_number.py: docstring example uses httpx2, not requests
- web_programming/get_imdbtop.py.DISABLED: import httpx2 instead of requests
- remove web_programming/get_user_tweets.py.DISABLED (a Twitter API how-to,
  not an algorithm) and drop the tweepy dependency that was its only user and
  the last high-level dep pulling in requests
- uv.lock intentionally untouched (keeper allow-list)

Refs #15081

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2026-09-05 09:26:45 +02:00

113 lines
3.6 KiB
Python

"""
This file provides a function which will take a product name as input from the user,
and fetch from Amazon information about products of this name or category. The product
information will include title, URL, price, ratings, and the discount available.
"""
# /// script
# requires-python = ">=3.13"
# dependencies = [
# "beautifulsoup4",
# "httpx2",
# "pandas",
# ]
# ///
from itertools import zip_longest
import httpx2
from bs4 import BeautifulSoup
from pandas import DataFrame
def get_amazon_product_data(product: str = "laptop") -> DataFrame:
"""
Take a product name or category as input and return product information from Amazon
including title, URL, price, ratings, and the discount available.
"""
url = f"https://www.amazon.in/laptop/s?k={product}"
header = {
"User-Agent": (
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36"
"(KHTML, like Gecko)Chrome/44.0.2403.157 Safari/537.36"
),
"Accept-Language": "en-US, en;q=0.5",
}
soup = BeautifulSoup(
httpx2.get(url, headers=header, timeout=10).text, features="lxml"
)
# Initialize a Pandas dataframe with the column titles
data_frame = DataFrame(
columns=[
"Product Title",
"Product Link",
"Current Price of the product",
"Product Rating",
"MRP of the product",
"Discount",
]
)
# Loop through each entry and store them in the dataframe
for item, _ in zip_longest(
soup.find_all(
"div",
attrs={"class": "s-result-item", "data-component-type": "s-search-result"},
),
soup.find_all("div", attrs={"class": "a-row a-size-base a-color-base"}),
):
try:
product_title = item.h2.text
product_link = "https://www.amazon.in/" + item.h2.a["href"]
product_price = item.find("span", attrs={"class": "a-offscreen"}).text
try:
product_rating = item.find("span", attrs={"class": "a-icon-alt"}).text
except AttributeError:
product_rating = "Not available"
try:
product_mrp = (
"₹"
+ item.find(
"span", attrs={"class": "a-price a-text-price"}
).text.split("₹")[1]
)
except AttributeError:
product_mrp = ""
try:
discount = float(
(
(
float(product_mrp.strip("₹").replace(",", ""))
- float(product_price.strip("₹").replace(",", ""))
)
/ float(product_mrp.strip("₹").replace(",", ""))
)
* 100
)
except ValueError:
discount = float("nan")
except AttributeError:
continue
data_frame.loc[str(len(data_frame.index))] = [
product_title,
product_link,
product_price,
product_rating,
product_mrp,
discount,
]
data_frame.loc[
data_frame["Current Price of the product"] > data_frame["MRP of the product"],
"MRP of the product",
] = " "
data_frame.loc[
data_frame["Current Price of the product"] > data_frame["MRP of the product"],
"Discount",
] = " "
data_frame.index += 1
return data_frame
if __name__ == "__main__":
product = "headphones"
get_amazon_product_data(product).to_csv(f"Amazon Product Data for {product}.csv")