mirror of
https://github.com/TheAlgorithms/Python.git
synced 2026-09-28 21:45:27 +08:00
* deps: migrate from httpx to httpx2 (pydantic's maintained fork) Mechanical rename of httpx -> httpx2 (API-compatible fork of httpx 0.28.1): pyproject.toml deps, PEP 723 inline-script headers, and all import/call sites. Excludes uv.lock (the keeper's allow-list rejects .lock files); the lock refresh needs a separate maintainer-merged PR. Refs #15081 * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * deps: drop tweepy + migrate remaining requests refs to httpx2 - maths/allocation_number.py: docstring example uses httpx2, not requests - web_programming/get_imdbtop.py.DISABLED: import httpx2 instead of requests - remove web_programming/get_user_tweets.py.DISABLED (a Twitter API how-to, not an algorithm) and drop the tweepy dependency that was its only user and the last high-level dep pulling in requests - uv.lock intentionally untouched (keeper allow-list) Refs #15081 --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
35 lines
1015 B
Python
35 lines
1015 B
Python
"""
|
|
Scraping jobs given job title and location from indeed website
|
|
"""
|
|
|
|
# /// script
|
|
# requires-python = ">=3.13"
|
|
# dependencies = [
|
|
# "beautifulsoup4",
|
|
# "httpx2",
|
|
# ]
|
|
# ///
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Generator
|
|
|
|
import httpx2
|
|
from bs4 import BeautifulSoup
|
|
|
|
url = "https://www.indeed.co.in/jobs?q=mobile+app+development&l="
|
|
|
|
|
|
def fetch_jobs(location: str = "mumbai") -> Generator[tuple[str, str]]:
|
|
soup = BeautifulSoup(httpx2.get(url + location, timeout=10).content, "html.parser")
|
|
# This attribute finds out all the specifics listed in a job
|
|
for job in soup.find_all("div", attrs={"data-tn-component": "organicJob"}):
|
|
job_title = job.find("a", attrs={"data-tn-element": "jobTitle"}).text.strip()
|
|
company_name = job.find("span", {"class": "company"}).text.strip()
|
|
yield job_title, company_name
|
|
|
|
|
|
if __name__ == "__main__":
|
|
for i, job in enumerate(fetch_jobs("Bangalore"), 1):
|
|
print(f"Job {i:>2} is {job[0]} at {job[1]}")
|