This commit is contained in:
queukat 2024-09-08 17:32:07 +03:00
parent 77f5b5ab9e
commit 58584def99
3 changed files with 212 additions and 33 deletions

View file

@ -8,7 +8,7 @@ import traceback
from typing import List, Optional, Any, Tuple
from httpx import HTTPStatusError
from reportlab.lib.pagesizes import letter
from reportlab.lib.pagesizes import A4
from reportlab.pdfgen import canvas
from selenium.common.exceptions import NoSuchElementException, TimeoutException
from selenium.webdriver import ActionChains
@ -62,8 +62,7 @@ class LinkedInEasyApplier:
def check_for_premium_redirect(self, job: Any, max_attempts=3):
"""Проверяет, был ли выполнен редирект на страницу LinkedIn Premium.
В случае редиректа возвращает пользователя на исходную страницу вакансии."""
current_url = self.driver.current_url
attempts = 0
@ -514,11 +513,48 @@ class LinkedInEasyApplier:
file_path_pdf = os.path.join(folder_path, f"Cover_Letter_{timestamp}.pdf")
logger.debug(f"Generated file path for cover letter: {file_path_pdf}")
c = canvas.Canvas(file_path_pdf, pagesize=letter)
_, height = letter
text_object = c.beginText(100, height - 100)
c = canvas.Canvas(file_path_pdf, pagesize=A4)
page_width, page_height = A4
text_object = c.beginText(50, page_height - 50)
text_object.setFont("Helvetica", 12)
text_object.textLines(cover_letter_text)
max_width = page_width - 100
bottom_margin = 50
available_height = page_height - bottom_margin - 50
def split_text_by_width(text, font, font_size, max_width):
wrapped_lines = []
for line in text.splitlines():
if utils.stringWidth(line, font, font_size) > max_width:
words = line.split()
new_line = ""
for word in words:
if utils.stringWidth(new_line + word + " ", font, font_size) <= max_width:
new_line += word + " "
else:
wrapped_lines.append(new_line.strip())
new_line = word + " "
wrapped_lines.append(new_line.strip())
else:
wrapped_lines.append(line)
return wrapped_lines
lines = split_text_by_width(cover_letter_text, "Helvetica", 12, max_width)
for line in lines:
text_height = text_object.getY()
if text_height > bottom_margin:
text_object.textLine(line)
else:
c.drawText(text_object)
c.showPage()
text_object = c.beginText(50, page_height - 50)
text_object.setFont("Helvetica", 12)
text_object.textLine(line)
c.drawText(text_object)
c.save()
logger.debug(f"Cover letter successfully generated and saved to: {file_path_pdf}")
@ -530,6 +566,7 @@ class LinkedInEasyApplier:
logger.error(f"Traceback: {tb_str}")
raise
file_size = os.path.getsize(file_path_pdf)
max_file_size = 2 * 1024 * 1024 # 2 MB
logger.debug(f"Cover letter file size: {file_size} bytes")
@ -701,12 +738,14 @@ class LinkedInEasyApplier:
def _find_and_handle_dropdown_question(self, section: WebElement) -> bool:
try:
# Попытка найти элемент с вопросом через класс
question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element')
question_text = question.find_element(By.TAG_NAME, 'label').text.lower()
logger.debug(f"Processing dropdown or combobox question: {question_text}")
# Если не удалось найти элемент с классом, пробуем искать по атрибуту 'data-test-text-entity-list-form-select'
dropdowns = question.find_elements(By.TAG_NAME, 'select')
if not dropdowns:
dropdowns = section.find_elements(By.CSS_SELECTOR, '[data-test-text-entity-list-form-select]')
if dropdowns:
dropdown = dropdowns[0]
select = Select(dropdown)
@ -714,9 +753,14 @@ class LinkedInEasyApplier:
logger.debug(f"Dropdown options found: {options}")
# Извлечение текста вопроса
question_text = question.find_element(By.TAG_NAME, 'label').text.lower()
logger.debug(f"Processing dropdown or combobox question: {question_text}")
current_selection = select.first_selected_option.text
logger.debug(f"Current selection: {current_selection}")
# Найдем существующий ответ в сохраненных данных
existing_answer = None
for item in self.all_data:
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown':
@ -738,9 +782,15 @@ class LinkedInEasyApplier:
logger.debug(f"Selected new dropdown answer: {answer}")
return True
else:
logger.debug(f"No dropdown found. Logging elements for debugging.")
elements = section.find_elements(By.XPATH, ".//*")
logger.debug(f"Elements found: {[element.tag_name for element in elements]}")
return False
except Exception as e:
logger.warning(f"Failed to handle dropdown or combobox question: {e}")
logger.warning(f"Failed to handle dropdown or combobox question: {e}", exc_info=True)
return False
def _is_numeric_field(self, field: WebElement) -> bool:

View file

@ -5,6 +5,7 @@ import time
from itertools import product
from pathlib import Path
from inputimeout import inputimeout, TimeoutOccurred
from selenium.common.exceptions import NoSuchElementException
from selenium.webdriver.common.by import By
@ -45,13 +46,18 @@ class LinkedInJobManager:
def set_parameters(self, parameters):
logger.debug("Setting parameters for LinkedInJobManager")
self.company_blacklist = parameters.get('companyBlacklist', []) or []
self.company_blacklist = parameters.get('company_blacklist', []) or []
self.title_blacklist = parameters.get('titleBlacklist', []) or []
self.positions = parameters.get('positions', [])
self.locations = parameters.get('locations', [])
self.apply_once_at_company = parameters.get('applyOnceAtCompany', False)
self.base_search_url = self.get_base_search_url(parameters)
self.seen_jobs = []
job_applicants_threshold = parameters.get('job_applicants_threshold', {})
self.min_applicants = job_applicants_threshold.get('min_applicants', 0)
self.max_applicants = job_applicants_threshold.get('max_applicants', float('inf'))
resume_path = parameters.get('uploads', {}).get('resume', None)
self.resume_path = Path(resume_path) if resume_path and Path(resume_path).exists() else None
self.output_file_directory = Path(parameters['outputFileDirectory'])
@ -109,31 +115,79 @@ class LinkedInJobManager:
utils.printyellow("Applying to jobs on this page has been completed!")
time_left = minimum_page_time - time.time()
# Ask user if they want to skip waiting, with timeout
if time_left > 0:
try:
user_input = inputimeout(
prompt=f"Sleeping for {time_left} seconds. Press 'y' to skip waiting. Timeout 10 seconds : ",
timeout=10).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {time_left} seconds as user chose not to skip.")
utils.printyellow(f"Sleeping for {time_left} seconds.")
logger.debug("Sleeping for %d seconds", time_left)
time.sleep(time_left)
minimum_page_time = time.time() + minimum_time
if page_sleep % 5 == 0:
sleep_time = random.randint(5, 34)
try:
user_input = inputimeout(
prompt=f"Sleeping for {sleep_time / 60} minutes. Press 'y' to skip waiting. Timeout 10 seconds : ",
timeout=10).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {sleep_time} seconds.")
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.")
logger.debug("Sleeping for %d seconds", sleep_time)
time.sleep(sleep_time)
page_sleep += 1
except Exception as e:
logger.error("Unexpected error during job search: %s", e)
utils.printred(f"Unexpected error: {e}")
continue
time_left = minimum_page_time - time.time()
if time_left > 0:
try:
user_input = inputimeout(
prompt=f"Sleeping for {time_left} seconds. Press 'y' to skip waiting. Timeout 10 seconds : ",
timeout=10).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {time_left} seconds as user chose not to skip.")
utils.printyellow(f"Sleeping for {time_left} seconds.")
logger.debug("Sleeping for %d seconds", time_left)
time.sleep(time_left)
minimum_page_time = time.time() + minimum_time
if page_sleep % 5 == 0:
sleep_time = random.randint(50, 90)
try:
user_input = inputimeout(
prompt=f"Sleeping for {sleep_time / 60} minutes. Press 'y' to skip waiting: ",
timeout=10).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {sleep_time} seconds.")
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.")
logger.debug("Sleeping for %d seconds", sleep_time)
time.sleep(sleep_time)
page_sleep += 1
@ -183,16 +237,82 @@ class LinkedInJobManager:
pass
job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list")
utils.scroll_slow(self.driver, job_results)
utils.scroll_slow(self.driver, job_results, step=300, reverse=True)
# utils.scroll_slow(self.driver, job_results)
# utils.scroll_slow(self.driver, job_results, step=300, reverse=True)
job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[
0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item')
if not job_list_elements:
utils.printyellow("No job class elements found on page, moving to next page.")
logger.debug("No job class elements found on page, skipping")
return
job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements]
for job in job_list:
try:
logger.debug(f"Starting applicant count search for job: {job.title} at {job.company}")
# Find all job insight elements
job_insight_elements = self.driver.find_elements(By.CLASS_NAME,
"job-details-jobs-unified-top-card__job-insight")
logger.debug(f"Found {len(job_insight_elements)} job insight elements")
# Initialize applicants_count as None
applicants_count = None
# Iterate over each job insight element to find the one containing the word "applicant"
for element in job_insight_elements:
logger.debug(f"Checking element text: {element.text}")
if "applicant" in element.text.lower():
# Found an element containing "applicant"
applicants_text = element.text.strip()
logger.debug(f"Applicants text found: {applicants_text}")
# Extract numeric digits from the text (e.g., "70 applicants" -> "70")
applicants_count = ''.join(filter(str.isdigit, applicants_text))
logger.debug(f"Extracted applicants count: {applicants_count}")
if applicants_count:
if "over" in applicants_text.lower():
applicants_count = int(applicants_count) + 1 # Handle "over X applicants"
logger.debug(f"Applicants count adjusted for 'over': {applicants_count}")
else:
applicants_count = int(applicants_count) # Convert the extracted number to an integer
break
# Check if applicants_count is valid (not None) before performing comparisons
if applicants_count is not None:
# Perform the threshold check for applicants count
if applicants_count < self.min_applicants or applicants_count > self.max_applicants:
utils.printyellow(
f"Skipping {job.title} at {job.company} due to applicants count: {applicants_count}")
logger.debug(f"Skipping {job.title} at {job.company}, applicants count: {applicants_count}")
self.write_to_file(job, "skipped_due_to_applicants")
continue # Skip this job if applicants count is outside the threshold
else:
logger.debug(f"Applicants count {applicants_count} is within the threshold")
else:
# If no applicants count was found, log a warning but continue the process
logger.warning(
f"Applicants count not found for {job.title} at {job.company}, continuing with application.")
except NoSuchElementException:
# Log a warning if the job insight elements are not found, but do not stop the job application process
logger.warning(
f"Applicants count elements not found for {job.title} at {job.company}, continuing with application.")
except ValueError as e:
# Handle errors when parsing the applicants count
logger.error(f"Error parsing applicants count for {job.title} at {job.company}: {e}")
except Exception as e:
# Catch any other exceptions to ensure the process continues
logger.error(
f"Unexpected error during applicants count processing for {job.title} at {job.company}: {e}")
# Continue with the job application process regardless of the applicants count check
logger.debug(f"Continuing with job application for {job.title} at {job.company}")
if self.is_blacklisted(job.title, job.company, job.link):
utils.printyellow(f"Blacklisted {job.title} at {job.company}, skipping...")
logger.debug("Job blacklisted: %s at %s", job.title, job.company)
@ -307,7 +427,6 @@ class LinkedInJobManager:
title_blacklisted = any(word in job_title_words for word in self.title_blacklist)
company_blacklisted = company.strip().lower() in (word.strip().lower() for word in self.company_blacklist)
link_seen = link in self.seen_jobs
is_blacklisted = title_blacklisted or company_blacklisted or link_seen
logger.debug("Job blacklisted status: %s", is_blacklisted)
return is_blacklisted

View file

@ -90,7 +90,13 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse
return
position = start
previous_position = None # Tracking the previous position to avoid duplicate scrolls
while (step > 0 and position < end) or (step < 0 and position > end):
if position == previous_position:
# Avoid re-scrolling to the same position
logger.debug("Stopping scroll as position hasn't changed: %d", position)
break
try:
driver.execute_script(script_scroll_to, scrollable_element, position)
logger.debug("Scrolled to position: %d", position)
@ -98,11 +104,15 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse
logger.error("Error during scrolling: %s", e)
print(f"Error during scrolling: {e}")
previous_position = position
position += step
# Decrease the step but ensure it doesn't reverse direction
step = max(10, abs(step) - 10) * (-1 if reverse else 1)
time.sleep(random.uniform(0.6, 1.5))
# Ensure the final scroll position is correct
driver.execute_script(script_scroll_to, scrollable_element, end)
logger.debug("Scrolled to final position: %d", end)
time.sleep(0.5)