From 58584def99926abbef458b11fd481b554447b781 Mon Sep 17 00:00:00 2001 From: queukat Date: Sun, 8 Sep 2024 17:32:07 +0300 Subject: [PATCH] new func --- src/linkedIn_easy_applier.py | 74 +++++++++++++--- src/linkedIn_job_manager.py | 161 ++++++++++++++++++++++++++++++----- src/utils.py | 10 +++ 3 files changed, 212 insertions(+), 33 deletions(-) diff --git a/src/linkedIn_easy_applier.py b/src/linkedIn_easy_applier.py index 951fea8..eb7322a 100644 --- a/src/linkedIn_easy_applier.py +++ b/src/linkedIn_easy_applier.py @@ -8,7 +8,7 @@ import traceback from typing import List, Optional, Any, Tuple from httpx import HTTPStatusError -from reportlab.lib.pagesizes import letter +from reportlab.lib.pagesizes import A4 from reportlab.pdfgen import canvas from selenium.common.exceptions import NoSuchElementException, TimeoutException from selenium.webdriver import ActionChains @@ -62,8 +62,7 @@ class LinkedInEasyApplier: def check_for_premium_redirect(self, job: Any, max_attempts=3): - """Проверяет, был ли выполнен редирект на страницу LinkedIn Premium. - В случае редиректа возвращает пользователя на исходную страницу вакансии.""" + current_url = self.driver.current_url attempts = 0 @@ -514,11 +513,48 @@ class LinkedInEasyApplier: file_path_pdf = os.path.join(folder_path, f"Cover_Letter_{timestamp}.pdf") logger.debug(f"Generated file path for cover letter: {file_path_pdf}") - c = canvas.Canvas(file_path_pdf, pagesize=letter) - _, height = letter - text_object = c.beginText(100, height - 100) + c = canvas.Canvas(file_path_pdf, pagesize=A4) + page_width, page_height = A4 + text_object = c.beginText(50, page_height - 50) text_object.setFont("Helvetica", 12) - text_object.textLines(cover_letter_text) + + max_width = page_width - 100 + bottom_margin = 50 + available_height = page_height - bottom_margin - 50 + + def split_text_by_width(text, font, font_size, max_width): + wrapped_lines = [] + for line in text.splitlines(): + + if utils.stringWidth(line, font, font_size) > max_width: + words = line.split() + new_line = "" + for word in words: + if utils.stringWidth(new_line + word + " ", font, font_size) <= max_width: + new_line += word + " " + else: + wrapped_lines.append(new_line.strip()) + new_line = word + " " + wrapped_lines.append(new_line.strip()) + else: + wrapped_lines.append(line) + return wrapped_lines + + + lines = split_text_by_width(cover_letter_text, "Helvetica", 12, max_width) + + for line in lines: + text_height = text_object.getY() + if text_height > bottom_margin: + text_object.textLine(line) + else: + + c.drawText(text_object) + c.showPage() + text_object = c.beginText(50, page_height - 50) + text_object.setFont("Helvetica", 12) + text_object.textLine(line) + c.drawText(text_object) c.save() logger.debug(f"Cover letter successfully generated and saved to: {file_path_pdf}") @@ -530,6 +566,7 @@ class LinkedInEasyApplier: logger.error(f"Traceback: {tb_str}") raise + file_size = os.path.getsize(file_path_pdf) max_file_size = 2 * 1024 * 1024 # 2 MB logger.debug(f"Cover letter file size: {file_size} bytes") @@ -701,12 +738,14 @@ class LinkedInEasyApplier: def _find_and_handle_dropdown_question(self, section: WebElement) -> bool: try: - + # Попытка найти элемент с вопросом через класс question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element') - question_text = question.find_element(By.TAG_NAME, 'label').text.lower() - logger.debug(f"Processing dropdown or combobox question: {question_text}") + # Если не удалось найти элемент с классом, пробуем искать по атрибуту 'data-test-text-entity-list-form-select' dropdowns = question.find_elements(By.TAG_NAME, 'select') + if not dropdowns: + dropdowns = section.find_elements(By.CSS_SELECTOR, '[data-test-text-entity-list-form-select]') + if dropdowns: dropdown = dropdowns[0] select = Select(dropdown) @@ -714,9 +753,14 @@ class LinkedInEasyApplier: logger.debug(f"Dropdown options found: {options}") + # Извлечение текста вопроса + question_text = question.find_element(By.TAG_NAME, 'label').text.lower() + logger.debug(f"Processing dropdown or combobox question: {question_text}") + current_selection = select.first_selected_option.text logger.debug(f"Current selection: {current_selection}") + # Найдем существующий ответ в сохраненных данных existing_answer = None for item in self.all_data: if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown': @@ -738,9 +782,15 @@ class LinkedInEasyApplier: logger.debug(f"Selected new dropdown answer: {answer}") return True - return False + else: + + logger.debug(f"No dropdown found. Logging elements for debugging.") + elements = section.find_elements(By.XPATH, ".//*") + logger.debug(f"Elements found: {[element.tag_name for element in elements]}") + return False + except Exception as e: - logger.warning(f"Failed to handle dropdown or combobox question: {e}") + logger.warning(f"Failed to handle dropdown or combobox question: {e}", exc_info=True) return False def _is_numeric_field(self, field: WebElement) -> bool: diff --git a/src/linkedIn_job_manager.py b/src/linkedIn_job_manager.py index 5b6e53b..adda476 100644 --- a/src/linkedIn_job_manager.py +++ b/src/linkedIn_job_manager.py @@ -5,6 +5,7 @@ import time from itertools import product from pathlib import Path +from inputimeout import inputimeout, TimeoutOccurred from selenium.common.exceptions import NoSuchElementException from selenium.webdriver.common.by import By @@ -45,13 +46,18 @@ class LinkedInJobManager: def set_parameters(self, parameters): logger.debug("Setting parameters for LinkedInJobManager") - self.company_blacklist = parameters.get('companyBlacklist', []) or [] + self.company_blacklist = parameters.get('company_blacklist', []) or [] self.title_blacklist = parameters.get('titleBlacklist', []) or [] self.positions = parameters.get('positions', []) self.locations = parameters.get('locations', []) self.apply_once_at_company = parameters.get('applyOnceAtCompany', False) self.base_search_url = self.get_base_search_url(parameters) self.seen_jobs = [] + + job_applicants_threshold = parameters.get('job_applicants_threshold', {}) + self.min_applicants = job_applicants_threshold.get('min_applicants', 0) + self.max_applicants = job_applicants_threshold.get('max_applicants', float('inf')) + resume_path = parameters.get('uploads', {}).get('resume', None) self.resume_path = Path(resume_path) if resume_path and Path(resume_path).exists() else None self.output_file_directory = Path(parameters['outputFileDirectory']) @@ -109,32 +115,80 @@ class LinkedInJobManager: utils.printyellow("Applying to jobs on this page has been completed!") time_left = minimum_page_time - time.time() + + # Ask user if they want to skip waiting, with timeout if time_left > 0: - utils.printyellow(f"Sleeping for {time_left} seconds.") - logger.debug("Sleeping for %d seconds", time_left) - time.sleep(time_left) - minimum_page_time = time.time() + minimum_time + try: + user_input = inputimeout( + prompt=f"Sleeping for {time_left} seconds. Press 'y' to skip waiting. Timeout 10 seconds : ", + timeout=10).strip().lower() + except TimeoutOccurred: + user_input = '' # No input after timeout + if user_input == 'y': + logger.debug("User chose to skip waiting.") + utils.printyellow("User skipped waiting.") + else: + logger.debug(f"Sleeping for {time_left} seconds as user chose not to skip.") + utils.printyellow(f"Sleeping for {time_left} seconds.") + time.sleep(time_left) + + minimum_page_time = time.time() + minimum_time + if page_sleep % 5 == 0: sleep_time = random.randint(5, 34) - utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") - logger.debug("Sleeping for %d seconds", sleep_time) - time.sleep(sleep_time) + try: + user_input = inputimeout( + prompt=f"Sleeping for {sleep_time / 60} minutes. Press 'y' to skip waiting. Timeout 10 seconds : ", + timeout=10).strip().lower() + except TimeoutOccurred: + user_input = '' # No input after timeout + if user_input == 'y': + logger.debug("User chose to skip waiting.") + utils.printyellow("User skipped waiting.") + else: + logger.debug(f"Sleeping for {sleep_time} seconds.") + utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") + time.sleep(sleep_time) page_sleep += 1 except Exception as e: logger.error("Unexpected error during job search: %s", e) utils.printred(f"Unexpected error: {e}") continue + time_left = minimum_page_time - time.time() + if time_left > 0: - utils.printyellow(f"Sleeping for {time_left} seconds.") - logger.debug("Sleeping for %d seconds", time_left) - time.sleep(time_left) - minimum_page_time = time.time() + minimum_time + try: + user_input = inputimeout( + prompt=f"Sleeping for {time_left} seconds. Press 'y' to skip waiting. Timeout 10 seconds : ", + timeout=10).strip().lower() + except TimeoutOccurred: + user_input = '' # No input after timeout + if user_input == 'y': + logger.debug("User chose to skip waiting.") + utils.printyellow("User skipped waiting.") + else: + logger.debug(f"Sleeping for {time_left} seconds as user chose not to skip.") + utils.printyellow(f"Sleeping for {time_left} seconds.") + time.sleep(time_left) + + minimum_page_time = time.time() + minimum_time + if page_sleep % 5 == 0: sleep_time = random.randint(50, 90) - utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") - logger.debug("Sleeping for %d seconds", sleep_time) - time.sleep(sleep_time) + try: + user_input = inputimeout( + prompt=f"Sleeping for {sleep_time / 60} minutes. Press 'y' to skip waiting: ", + timeout=10).strip().lower() + except TimeoutOccurred: + user_input = '' # No input after timeout + if user_input == 'y': + logger.debug("User chose to skip waiting.") + utils.printyellow("User skipped waiting.") + else: + logger.debug(f"Sleeping for {sleep_time} seconds.") + utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") + time.sleep(sleep_time) page_sleep += 1 def get_jobs_from_page(self): @@ -183,16 +237,82 @@ class LinkedInJobManager: pass job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list") - utils.scroll_slow(self.driver, job_results) - utils.scroll_slow(self.driver, job_results, step=300, reverse=True) + # utils.scroll_slow(self.driver, job_results) + # utils.scroll_slow(self.driver, job_results, step=300, reverse=True) + job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[ 0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item') + if not job_list_elements: utils.printyellow("No job class elements found on page, moving to next page.") logger.debug("No job class elements found on page, skipping") return + job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements] + for job in job_list: + + try: + logger.debug(f"Starting applicant count search for job: {job.title} at {job.company}") + + # Find all job insight elements + job_insight_elements = self.driver.find_elements(By.CLASS_NAME, + "job-details-jobs-unified-top-card__job-insight") + logger.debug(f"Found {len(job_insight_elements)} job insight elements") + + # Initialize applicants_count as None + applicants_count = None + + # Iterate over each job insight element to find the one containing the word "applicant" + for element in job_insight_elements: + logger.debug(f"Checking element text: {element.text}") + if "applicant" in element.text.lower(): + # Found an element containing "applicant" + applicants_text = element.text.strip() + logger.debug(f"Applicants text found: {applicants_text}") + + # Extract numeric digits from the text (e.g., "70 applicants" -> "70") + applicants_count = ''.join(filter(str.isdigit, applicants_text)) + logger.debug(f"Extracted applicants count: {applicants_count}") + + if applicants_count: + if "over" in applicants_text.lower(): + applicants_count = int(applicants_count) + 1 # Handle "over X applicants" + logger.debug(f"Applicants count adjusted for 'over': {applicants_count}") + else: + applicants_count = int(applicants_count) # Convert the extracted number to an integer + break + + # Check if applicants_count is valid (not None) before performing comparisons + if applicants_count is not None: + # Perform the threshold check for applicants count + if applicants_count < self.min_applicants or applicants_count > self.max_applicants: + utils.printyellow( + f"Skipping {job.title} at {job.company} due to applicants count: {applicants_count}") + logger.debug(f"Skipping {job.title} at {job.company}, applicants count: {applicants_count}") + self.write_to_file(job, "skipped_due_to_applicants") + continue # Skip this job if applicants count is outside the threshold + else: + logger.debug(f"Applicants count {applicants_count} is within the threshold") + else: + # If no applicants count was found, log a warning but continue the process + logger.warning( + f"Applicants count not found for {job.title} at {job.company}, continuing with application.") + except NoSuchElementException: + # Log a warning if the job insight elements are not found, but do not stop the job application process + logger.warning( + f"Applicants count elements not found for {job.title} at {job.company}, continuing with application.") + except ValueError as e: + # Handle errors when parsing the applicants count + logger.error(f"Error parsing applicants count for {job.title} at {job.company}: {e}") + except Exception as e: + # Catch any other exceptions to ensure the process continues + logger.error( + f"Unexpected error during applicants count processing for {job.title} at {job.company}: {e}") + + # Continue with the job application process regardless of the applicants count check + logger.debug(f"Continuing with job application for {job.title} at {job.company}") + if self.is_blacklisted(job.title, job.company, job.link): utils.printyellow(f"Blacklisted {job.title} at {job.company}, skipping...") logger.debug("Job blacklisted: %s at %s", job.title, job.company) @@ -200,7 +320,7 @@ class LinkedInJobManager: continue if self.is_already_applied_to_job(job.title, job.company, job.link): self.write_to_file(job, "skipped") - continue + continue if self.is_already_applied_to_company(job.company): self.write_to_file(job, "skipped") continue @@ -307,7 +427,6 @@ class LinkedInJobManager: title_blacklisted = any(word in job_title_words for word in self.title_blacklist) company_blacklisted = company.strip().lower() in (word.strip().lower() for word in self.company_blacklist) link_seen = link in self.seen_jobs - is_blacklisted = title_blacklisted or company_blacklisted or link_seen logger.debug("Job blacklisted status: %s", is_blacklisted) return is_blacklisted @@ -322,8 +441,8 @@ class LinkedInJobManager: def is_already_applied_to_company(self, company): if not self.apply_once_at_company: - return False - + return False + output_files = ["success.json"] for file_name in output_files: file_path = self.output_file_directory / file_name diff --git a/src/utils.py b/src/utils.py index 44d022f..f4e4d4a 100644 --- a/src/utils.py +++ b/src/utils.py @@ -90,7 +90,13 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse return position = start + previous_position = None # Tracking the previous position to avoid duplicate scrolls while (step > 0 and position < end) or (step < 0 and position > end): + if position == previous_position: + # Avoid re-scrolling to the same position + logger.debug("Stopping scroll as position hasn't changed: %d", position) + break + try: driver.execute_script(script_scroll_to, scrollable_element, position) logger.debug("Scrolled to position: %d", position) @@ -98,11 +104,15 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse logger.error("Error during scrolling: %s", e) print(f"Error during scrolling: {e}") + previous_position = position position += step + + # Decrease the step but ensure it doesn't reverse direction step = max(10, abs(step) - 10) * (-1 if reverse else 1) time.sleep(random.uniform(0.6, 1.5)) + # Ensure the final scroll position is correct driver.execute_script(script_scroll_to, scrollable_element, end) logger.debug("Scrolled to final position: %d", end) time.sleep(0.5)