Полуавтоматический отклик на hh.ru: поиск, ИИ-генерация, веб-интерфейс
This commit is contained in:
commit
0ca55f3818
11 files changed
+1740
No files matched your search
+210
@@ -0,0 +1,210 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Поиск вакансий на hh.ru через парсинг страницы поиска (без API)."""
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
import urllib.parse
|
||||
|
||||
import requests
|
||||
|
||||
HEADERS = {
|
||||
"User-Agent": ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0 Safari/537.36"),
|
||||
"Accept-Language": "ru-RU,ru;q=0.9",
|
||||
}
|
||||
|
||||
SEARCH_URL = "https://hh.ru/search/vacancy"
|
||||
|
||||
|
||||
def _clean(text):
|
||||
"""Убираем HTML-теги и лишние пробелы."""
|
||||
text = re.sub(r"<[^>]+>", "", text)
|
||||
text = text.replace("\xa0", " ")
|
||||
return re.sub(r"\s+", " ", text).strip()
|
||||
|
||||
|
||||
def _extract_salary(card_html):
|
||||
"""Зарплата из <data value='...'> внутри карточки."""
|
||||
values = re.findall(r'<data value="(\d+)">', card_html)
|
||||
if not values:
|
||||
return None
|
||||
nums = [int(v) for v in values]
|
||||
if len(nums) >= 2:
|
||||
return {"from": nums[0], "to": nums[1]}
|
||||
return {"from": nums[0], "to": None}
|
||||
|
||||
|
||||
def parse_search_page(html):
|
||||
"""Разбираем HTML страницы поиска в список вакансий."""
|
||||
vacancies = []
|
||||
# Разбиваем страницу на карточки
|
||||
card_parts = re.split(r'data-qa="vacancy-serp__vacancy"', html)
|
||||
for part in card_parts[1:]:
|
||||
# Ссылка и название
|
||||
m = re.search(r'data-qa="serp-item__title"[^>]*href="([^"]+)"[^>]*>(.*?)</a>', part, re.S)
|
||||
if not m:
|
||||
continue
|
||||
href, title_html = m.group(1), m.group(2)
|
||||
title = _clean(title_html)
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Компания
|
||||
m = re.search(r'data-qa="vacancy-serp__vacancy-employer"[^>]*>(.*?)</a>', part, re.S)
|
||||
company = _clean(m.group(1)) if m else None
|
||||
|
||||
# Адрес
|
||||
m = re.search(r'data-qa="vacancy-serp__vacancy-address"[^>]*>(.*?)</span>', part, re.S)
|
||||
address = _clean(m.group(1)) if m else None
|
||||
|
||||
# Опыт работы
|
||||
m = re.search(r'data-qa="vacancy-serp__vacancy-work-experience-([^"]+)"', part)
|
||||
experience = m.group(1) if m else None
|
||||
|
||||
# ID вакансии
|
||||
m = re.search(r"/vacancy/(\d+)", href)
|
||||
vacancy_id = m.group(1) if m else None
|
||||
|
||||
vacancies.append({
|
||||
"id": vacancy_id,
|
||||
"title": title,
|
||||
"company": company,
|
||||
"address": address,
|
||||
"experience": experience,
|
||||
"salary": _extract_salary(part),
|
||||
"url": href.split("?")[0],
|
||||
})
|
||||
return vacancies
|
||||
|
||||
|
||||
def parse_cookies(cookie_str):
|
||||
"""Разбирает cookies в словарь. Поддерживает два формата:
|
||||
1. JSON из расширений (Cookie-Editor): [{"name": ..., "value": ...}, ...]
|
||||
2. Строка вида 'name1=value1; name2=value2'
|
||||
"""
|
||||
cookies = {}
|
||||
if not cookie_str:
|
||||
return cookies
|
||||
s = cookie_str.strip()
|
||||
# Формат JSON
|
||||
if s.startswith("["):
|
||||
try:
|
||||
items = json.loads(s)
|
||||
for item in items:
|
||||
if isinstance(item, dict) and item.get("name"):
|
||||
val = str(item.get("value", ""))
|
||||
if _is_latin1(val):
|
||||
cookies[item["name"]] = val
|
||||
return cookies
|
||||
except Exception:
|
||||
pass
|
||||
# Формат name=value; ...
|
||||
for part in s.split(";"):
|
||||
part = part.strip()
|
||||
if "=" in part:
|
||||
k, v = part.split("=", 1)
|
||||
if _is_latin1(v):
|
||||
cookies[k.strip()] = v.strip()
|
||||
return cookies
|
||||
|
||||
|
||||
def _is_latin1(s):
|
||||
"""HTTP-заголовки требуют latin-1: пропускаем cookies с другими символами."""
|
||||
try:
|
||||
s.encode("latin-1")
|
||||
return True
|
||||
except UnicodeEncodeError:
|
||||
return False
|
||||
|
||||
|
||||
def search_vacancies(query, area=None, pages=1, per_page=20, delay=2.0, cookies=None):
|
||||
"""Ищем вакансии по запросу. Возвращает список словарей."""
|
||||
all_vacancies = []
|
||||
seen = set()
|
||||
for page_num in range(pages):
|
||||
params = {"text": query, "page": page_num, "per_page": per_page}
|
||||
if area:
|
||||
params["area"] = area
|
||||
url = f"{SEARCH_URL}?{urllib.parse.urlencode(params)}"
|
||||
resp = requests.get(url, headers=HEADERS, timeout=30, cookies=cookies)
|
||||
if resp.status_code != 200:
|
||||
print(f" [поиск] HTTP {resp.status_code} для страницы {page_num}")
|
||||
break
|
||||
found = parse_search_page(resp.text)
|
||||
if not found:
|
||||
print(f" [поиск] страница {page_num}: вакансий не найдено (возможно, капча)")
|
||||
break
|
||||
for v in found:
|
||||
if v["id"] and v["id"] not in seen:
|
||||
seen.add(v["id"])
|
||||
all_vacancies.append(v)
|
||||
if len(found) < per_page:
|
||||
break
|
||||
time.sleep(delay) # вежливая пауза между страницами
|
||||
return all_vacancies
|
||||
|
||||
|
||||
def fetch_vacancy_details(vacancy_id, cookies=None):
|
||||
"""Получаем описание вакансии со страницы вакансии."""
|
||||
url = f"https://hh.ru/vacancy/{vacancy_id}"
|
||||
resp = requests.get(url, headers=HEADERS, timeout=30, cookies=cookies)
|
||||
if resp.status_code != 200:
|
||||
return None
|
||||
html = resp.text
|
||||
details = {"id": vacancy_id, "url": url}
|
||||
|
||||
m = re.search(r'data-qa="vacancy-title"[^>]*>(.*?)</h1>', html, re.S)
|
||||
details["title"] = _clean(m.group(1)) if m else None
|
||||
|
||||
m = re.search(r'data-qa="vacancy-company-name"[^>]*>(.*?)</a>', html, re.S)
|
||||
details["company"] = _clean(m.group(1)) if m else None
|
||||
|
||||
m = re.search(r'data-qa="vacancy-salary"[^>]*>(.*?)</div>', html, re.S)
|
||||
details["salary"] = _clean(m.group(1)) if m else None
|
||||
|
||||
m = re.search(r'data-qa="vacancy-description"[^>]*>(.*?)</div>\s*</div>', html, re.S)
|
||||
details["description"] = _clean(m.group(1)) if m else None
|
||||
if details["description"]:
|
||||
details["description"] = details["description"][:3000]
|
||||
|
||||
return details
|
||||
|
||||
|
||||
def fetch_my_resumes(cookies=None):
|
||||
"""Получает список резюме пользователя со страницы hh.ru/applicant/resumes.
|
||||
Возвращает список {"id": hash, "title": ..., "updated": ...}."""
|
||||
url = "https://hh.ru/applicant/resumes"
|
||||
resp = requests.get(url, headers=HEADERS, timeout=30, cookies=cookies)
|
||||
if resp.status_code != 200:
|
||||
return None, f"HTTP {resp.status_code} — проверьте ключ сессии"
|
||||
html = resp.text
|
||||
if "resume" not in html.lower() and "Войти" in html:
|
||||
return None, "Сессия недействительна — страница требует входа"
|
||||
|
||||
resumes = []
|
||||
seen = set()
|
||||
# Блоки резюме: ищем ссылки /resume/{hash} (hash = hex-строка 30-50 символов)
|
||||
for m in re.finditer(r'/resume/([a-f0-9]{30,50})', html):
|
||||
hid = m.group(1)
|
||||
if hid in seen:
|
||||
continue
|
||||
seen.add(hid)
|
||||
# Ищем заголовок резюме рядом (до 1500 символов после ссылки)
|
||||
chunk = html[m.start():m.start() + 1500]
|
||||
title_m = re.search(r'data-qa="title"[^>]*>(.*?)</h3>', chunk, re.S)
|
||||
title = _clean(title_m.group(1)) if title_m else None
|
||||
if not title:
|
||||
title_m = re.search(r'data-qa="resume-title"[^>]*>(.*?)<', chunk, re.S)
|
||||
title = _clean(title_m.group(1)) if title_m else "Без названия"
|
||||
resumes.append({"id": hid, "title": title})
|
||||
return resumes, None
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
query = sys.argv[1] if len(sys.argv) > 1 else "python developer"
|
||||
print(f"Ищу: {query}")
|
||||
result = search_vacancies(query, pages=1)
|
||||
print(f"Найдено: {len(result)}")
|
||||
for v in result[:5]:
|
||||
print(f" {v['id']} | {v['title']} | {v['company']} | {v['salary']}")
|
||||
Reference in new issue
Block a user