diff --git a/src/gpt.py b/src/gpt.py index 77a9992..c22f123 100644 --- a/src/gpt.py +++ b/src/gpt.py @@ -2,19 +2,24 @@ import json import os import re import textwrap +import time from datetime import datetime from abc import ABC, abstractmethod from typing import Dict, List, Union from pathlib import Path +from typing import Dict, List + +import httpx +from Levenshtein import distance from dotenv import load_dotenv from langchain_core.messages.ai import AIMessage from langchain_core.output_parsers import StrOutputParser from langchain_core.prompt_values import StringPromptValue from langchain_core.prompts import ChatPromptTemplate from langchain_openai import ChatOpenAI -from Levenshtein import distance import src.strings as strings +from src.utils import logger load_dotenv() @@ -76,150 +81,272 @@ class AIAdapter: return self.model.invoke(prompt) class LLMLogger: + def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]): + self.llm = llm + logger.debug("LLMLogger successfully initialized with LLM: %s", llm) @staticmethod def log_request(prompts, parsed_reply: Dict[str, Dict]): - calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json") + logger.debug("Starting log_request method") + logger.debug("Prompts received: %s", prompts) + logger.debug("Parsed reply received: %s", parsed_reply) + + try: + calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json") + logger.debug("Logging path determined: %s", calls_log) + except Exception as e: + logger.error("Error determining the log path: %s", str(e)) + raise + if isinstance(prompts, StringPromptValue): + logger.debug("Prompts are of type StringPromptValue") prompts = prompts.text + logger.debug("Prompts converted to text: %s", prompts) elif isinstance(prompts, Dict): - # Convert prompts to a dictionary if they are not in the expected format - prompts = { - f"prompt_{i+1}": prompt.content - for i, prompt in enumerate(prompts.messages) - } + logger.debug("Prompts are of type Dict") + try: + prompts = { + f"prompt_{i + 1}": prompt.content + for i, prompt in enumerate(prompts.messages) + } + logger.debug("Prompts converted to dictionary: %s", prompts) + except Exception as e: + logger.error("Error converting prompts to dictionary: %s", str(e)) + raise else: - prompts = { - f"prompt_{i+1}": prompt.content - for i, prompt in enumerate(prompts.messages) + logger.debug("Prompts are of unknown type, attempting default conversion") + try: + prompts = { + f"prompt_{i + 1}": prompt.content + for i, prompt in enumerate(prompts.messages) + } + logger.debug("Prompts converted to dictionary using default method: %s", prompts) + except Exception as e: + logger.error("Error converting prompts using default method: %s", str(e)) + raise + + try: + current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S") + logger.debug("Current time obtained: %s", current_time) + except Exception as e: + logger.error("Error obtaining current time: %s", str(e)) + raise + + try: + token_usage = parsed_reply["usage_metadata"] + output_tokens = token_usage["output_tokens"] + input_tokens = token_usage["input_tokens"] + total_tokens = token_usage["total_tokens"] + logger.debug("Token usage - Input: %d, Output: %d, Total: %d", input_tokens, output_tokens, total_tokens) + except KeyError as e: + logger.error("KeyError in parsed_reply structure: %s", str(e)) + raise + + try: + model_name = parsed_reply["response_metadata"]["model_name"] + logger.debug("Model name: %s", model_name) + except KeyError as e: + logger.error("KeyError in response_metadata: %s", str(e)) + raise + + try: + prompt_price_per_token = 0.00000015 + completion_price_per_token = 0.0000006 + total_cost = (input_tokens * prompt_price_per_token) + (output_tokens * completion_price_per_token) + logger.debug("Total cost calculated: %f", total_cost) + except Exception as e: + logger.error("Error calculating total cost: %s", str(e)) + raise + + try: + log_entry = { + "model": model_name, + "time": current_time, + "prompts": prompts, + "replies": parsed_reply["content"], + "total_tokens": total_tokens, + "input_tokens": input_tokens, + "output_tokens": output_tokens, + "total_cost": total_cost, } + logger.debug("Log entry created: %s", log_entry) + except KeyError as e: + logger.error("Error creating log entry: missing key %s in parsed_reply", str(e)) + raise - current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S") - - # Extract token usage details from the response - token_usage = parsed_reply["usage_metadata"] - output_tokens = token_usage["output_tokens"] - input_tokens = token_usage["input_tokens"] - total_tokens = token_usage["total_tokens"] - - # Extract model details from the response - model_name = parsed_reply["response_metadata"]["model_name"] - prompt_price_per_token = 0.00000015 - completion_price_per_token = 0.0000006 - - # Calculate the total cost of the API call - total_cost = (input_tokens * prompt_price_per_token) + ( - output_tokens * completion_price_per_token - ) - - # Create a log entry with all relevant information - log_entry = { - "model": model_name, - "time": current_time, - "prompts": prompts, - "replies": parsed_reply["content"], # Response content - "total_tokens": total_tokens, - "input_tokens": input_tokens, - "output_tokens": output_tokens, - "total_cost": total_cost, - } - - # Write the log entry to the log file in JSON format - with open(calls_log, "a", encoding="utf-8") as f: - json_string = json.dumps(log_entry, ensure_ascii=False, indent=4) - f.write(json_string + "\n") + try: + with open(calls_log, "a", encoding="utf-8") as f: + json_string = json.dumps(log_entry, ensure_ascii=False, indent=4) + f.write(json_string + "\n") + logger.debug("Log entry written to file: %s", calls_log) + except Exception as e: + logger.error("Error writing log entry to file: %s", str(e)) + raise class LoggerChatModel: + def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]): + self.llm = llm + logger.debug("LoggerChatModel successfully initialized with LLM: %s", llm) def __call__(self, messages: List[Dict[str, str]]) -> str: - # Call the LLM with the provided messages and log the response. - reply = self.llm.invoke(messages) - parsed_reply = self.parse_llmresult(reply) - LLMLogger.log_request(prompts=messages, parsed_reply=parsed_reply) - return reply + + logger.debug("Entering __call__ method with messages: %s", messages) + while True: + try: + logger.debug("Attempting to call the LLM with messages") + reply = self.llm(messages) + logger.debug("LLM response received: %s", reply) + + parsed_reply = self.parse_llmresult(reply) + logger.debug("Parsed LLM reply: %s", parsed_reply) + + LLMLogger.log_request(prompts=messages, parsed_reply=parsed_reply) + logger.debug("Request successfully logged") + + return reply + + except httpx.HTTPStatusError as e: + logger.error("HTTPStatusError encountered: %s", str(e)) + if e.response.status_code == 429: + retry_after = e.response.headers.get('retry-after') + retry_after_ms = e.response.headers.get('retry-after-ms') + + if retry_after: + wait_time = int(retry_after) + logger.warning( + "Rate limit exceeded. Waiting for %d seconds before retrying (extracted from 'retry-after' header)...", + wait_time) + time.sleep(wait_time) + elif retry_after_ms: + wait_time = int(retry_after_ms) / 1000.0 + logger.warning( + "Rate limit exceeded. Waiting for %f seconds before retrying (extracted from 'retry-after-ms' header)...", + wait_time) + time.sleep(wait_time) + else: + wait_time = 30 + logger.warning( + "'retry-after' header not found. Waiting for %d seconds before retrying (default)...", + wait_time) + time.sleep(wait_time) + else: + logger.error("HTTP error occurred with status code: %d, waiting 30 seconds before retrying", + e.response.status_code) + time.sleep(30) + + except Exception as e: + logger.error("Unexpected error occurred: %s", str(e)) + logger.info("Waiting for 30 seconds before retrying due to an unexpected error.") + time.sleep(30) + continue + def parse_llmresult(self, llmresult: AIMessage) -> Dict[str, Dict]: - # Parse the LLM result into a structured format. - content = llmresult.content - response_metadata = llmresult.response_metadata - id_ = llmresult.id - usage_metadata = llmresult.usage_metadata - parsed_result = { - "content": content, - "response_metadata": { - "model_name": response_metadata.get("model_name", ""), - "system_fingerprint": response_metadata.get("system_fingerprint", ""), - "finish_reason": response_metadata.get("finish_reason", ""), - "logprobs": response_metadata.get("logprobs", None), - }, - "id": id_, - "usage_metadata": { - "input_tokens": usage_metadata.get("input_tokens", 0), - "output_tokens": usage_metadata.get("output_tokens", 0), - "total_tokens": usage_metadata.get("total_tokens", 0), - }, - } - return parsed_result + logger.debug("Parsing LLM result: %s", llmresult) + + try: + content = llmresult.content + response_metadata = llmresult.response_metadata + id_ = llmresult.id + usage_metadata = llmresult.usage_metadata + + parsed_result = { + "content": content, + "response_metadata": { + "model_name": response_metadata.get("model_name", ""), + "system_fingerprint": response_metadata.get("system_fingerprint", ""), + "finish_reason": response_metadata.get("finish_reason", ""), + "logprobs": response_metadata.get("logprobs", None), + }, + "id": id_, + "usage_metadata": { + "input_tokens": usage_metadata.get("input_tokens", 0), + "output_tokens": usage_metadata.get("output_tokens", 0), + "total_tokens": usage_metadata.get("total_tokens", 0), + }, + } + + logger.debug("Parsed LLM result successfully: %s", parsed_result) + return parsed_result + + except KeyError as e: + logger.error("KeyError while parsing LLM result: missing key %s", str(e)) + raise + + except Exception as e: + logger.error("Unexpected error while parsing LLM result: %s", str(e)) + raise class GPTAnswerer: + def __init__(self, config, llm_api_key): self.ai_adapter = AIAdapter(config, llm_api_key) self.llm_cheap = LoggerChatModel(self.ai_adapter) + @property def job_description(self): return self.job.description @staticmethod def find_best_match(text: str, options: list[str]) -> str: + logger.debug("Finding best match for text: '%s' in options: %s", text, options) distances = [ (option, distance(text.lower(), option.lower())) for option in options ] best_option = min(distances, key=lambda x: x[1])[0] + logger.debug("Best match found: %s", best_option) return best_option @staticmethod def _remove_placeholders(text: str) -> str: + logger.debug("Removing placeholders from text: %s", text) text = text.replace("PLACEHOLDER", "") return text.strip() @staticmethod def _preprocess_template_string(template: str) -> str: - # Preprocess a template string to remove unnecessary indentation. + logger.debug("Preprocessing template string") return textwrap.dedent(template) def set_resume(self, resume): + logger.debug("Setting resume: %s", resume) self.resume = resume def set_job(self, job): + logger.debug("Setting job: %s", job) self.job = job self.job.set_summarize_job_description(self.summarize_job_description(self.job.description)) def set_job_application_profile(self, job_application_profile): + logger.debug("Setting job application profile: %s", job_application_profile) self.job_application_profile = job_application_profile - + def summarize_job_description(self, text: str) -> str: + logger.debug("Summarizing job description: %s", text) strings.summarize_prompt_template = self._preprocess_template_string( strings.summarize_prompt_template ) prompt = ChatPromptTemplate.from_template(strings.summarize_prompt_template) chain = prompt | self.llm_cheap | StrOutputParser() output = chain.invoke({"text": text}) + logger.debug("Summary generated: %s", output) return output - + def _create_chain(self, template: str): + logger.debug("Creating chain with template: %s", template) prompt = ChatPromptTemplate.from_template(template) return prompt | self.llm_cheap | StrOutputParser() - + def answer_question_textual_wide_range(self, question: str) -> str: - # Define chains for each section of the resume + logger.debug("Answering textual question: %s", question) chains = { "personal_information": self._create_chain(strings.personal_information_template), "self_identification": self._create_chain(strings.self_identification_template), @@ -326,59 +453,83 @@ class GPTAnswerer: prompt = ChatPromptTemplate.from_template(section_prompt) chain = prompt | self.llm_cheap | StrOutputParser() output = chain.invoke({"question": question}) + match = re.search(r"(Personal information|Self Identification|Legal Authorization|Work Preferences|Education Details|Experience Details|Projects|Availability|Salary Expectations|Certifications|Languages|Interests|Cover letter)", output, re.IGNORECASE) if not match: raise ValueError("Could not extract section name from the response.") section_name = match.group(1).lower().replace(" ", "_") + if section_name == "cover_letter": chain = chains.get(section_name) output = chain.invoke({"resume": self.resume, "job_description": self.job_description}) + logger.debug("Cover letter generated: %s", output) return output - resume_section = getattr(self.resume, section_name, None) or getattr(self.job_application_profile, section_name, None) + resume_section = getattr(self.resume, section_name, None) or getattr(self.job_application_profile, section_name, + None) if resume_section is None: + logger.error("Section '%s' not found in either resume or job_application_profile.", section_name) raise ValueError(f"Section '{section_name}' not found in either resume or job_application_profile.") chain = chains.get(section_name) if chain is None: + logger.error("Chain not defined for section '%s'", section_name) raise ValueError(f"Chain not defined for section '{section_name}'") - return chain.invoke({"resume_section": resume_section, "question": question}) + output = chain.invoke({"resume_section": resume_section, "question": question}) + logger.debug("Question answered: %s", output) + return output def answer_question_numeric(self, question: str, default_experience: int = 3) -> int: + logger.debug("Answering numeric question: %s", question) func_template = self._preprocess_template_string(strings.numeric_question_template) prompt = ChatPromptTemplate.from_template(func_template) chain = prompt | self.llm_cheap | StrOutputParser() - output_str = chain.invoke({"resume_educations": self.resume.education_details,"resume_jobs": self.resume.experience_details,"resume_projects": self.resume.projects , "question": question}) + output_str = chain.invoke( + {"resume_educations": self.resume.education_details, "resume_jobs": self.resume.experience_details, + "resume_projects": self.resume.projects, "question": question}) + logger.debug("Raw output for numeric question: %s", output_str) try: output = self.extract_number_from_string(output_str) + logger.debug("Extracted number: %d", output) except ValueError: + logger.warning("Failed to extract number, using default experience: %d", default_experience) output = default_experience return output def extract_number_from_string(self, output_str): + logger.debug("Extracting number from string: %s", output_str) numbers = re.findall(r"\d+", output_str) if numbers: + logger.debug("Numbers found: %s", numbers) return int(numbers[0]) else: + logger.error("No numbers found in the string") raise ValueError("No numbers found in the string") def answer_question_from_options(self, question: str, options: list[str]) -> str: + logger.debug("Answering question from options: %s", question) func_template = self._preprocess_template_string(strings.options_template) prompt = ChatPromptTemplate.from_template(func_template) chain = prompt | self.llm_cheap | StrOutputParser() output_str = chain.invoke({"resume": self.resume, "question": question, "options": options}) + logger.debug("Raw output for options question: %s", output_str) best_option = self.find_best_match(output_str, options) + logger.debug("Best option determined: %s", best_option) return best_option - + def resume_or_cover(self, phrase: str) -> str: - # Define the prompt template + logger.debug("Determining if phrase refers to resume or cover letter: %s", phrase) prompt_template = """ - Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. Do not provide any additional information or explanations. + Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. + If the phrase contains only one word 'upload', consider it as 'cover'. + If the phrase contains 'upload resume', consider it as 'resume'. + Do not provide any additional information or explanations. phrase: {phrase} """ prompt = ChatPromptTemplate.from_template(prompt_template) chain = prompt | self.llm_cheap | StrOutputParser() response = chain.invoke({"phrase": phrase}) + logger.debug("Response for resume_or_cover: %s", response) if "resume" in response: return "resume" elif "cover" in response: diff --git a/src/job.py b/src/job.py index 31fef22..39b2371 100644 --- a/src/job.py +++ b/src/job.py @@ -1,5 +1,8 @@ from dataclasses import dataclass +from src.utils import logger + + @dataclass class Job: title: str @@ -13,18 +16,22 @@ class Job: recruiter_link: str = "" def set_summarize_job_description(self, summarize_job_description): + logger.debug("Setting summarized job description: %s", summarize_job_description) self.summarize_job_description = summarize_job_description def set_job_description(self, description): + logger.debug("Setting job description: %s", description) self.description = description def set_recruiter_link(self, recruiter_link): + logger.debug("Setting recruiter link: %s", recruiter_link) self.recruiter_link = recruiter_link def formatted_job_information(self): """ Formats the job information as a markdown string. """ + logger.debug("Formatting job information for job: %s at %s", self.title, self.company) job_information = f""" # Job Description ## Job Information @@ -36,4 +43,6 @@ class Job: ## Description {self.description or 'No description provided.'} """ - return job_information.strip() + formatted_information = job_information.strip() + logger.debug("Formatted job information: %s", formatted_information) + return formatted_information diff --git a/src/job_application_profile.py b/src/job_application_profile.py index 89bbdb2..5330c2b 100644 --- a/src/job_application_profile.py +++ b/src/job_application_profile.py @@ -1,7 +1,10 @@ from dataclasses import dataclass -from typing import Dict, List + import yaml +from src.utils import logger + + @dataclass class SelfIdentification: gender: str @@ -10,6 +13,7 @@ class SelfIdentification: disability: str ethnicity: str + @dataclass class LegalAuthorization: eu_work_authorization: str @@ -21,6 +25,7 @@ class LegalAuthorization: legally_allowed_to_work_in_eu: str requires_eu_sponsorship: str + @dataclass class WorkPreferences: remote_work: str @@ -30,14 +35,17 @@ class WorkPreferences: willing_to_undergo_drug_tests: str willing_to_undergo_background_checks: str + @dataclass class Availability: notice_period: str + @dataclass class SalaryExpectations: salary_range_usd: str + @dataclass class JobApplicationProfile: self_identification: SelfIdentification @@ -47,86 +55,123 @@ class JobApplicationProfile: salary_expectations: SalaryExpectations def __init__(self, yaml_str: str): + logger.debug("Initializing JobApplicationProfile with provided YAML string") try: data = yaml.safe_load(yaml_str) + logger.debug("YAML data successfully parsed: %s", data) except yaml.YAMLError as e: + logger.error("Error parsing YAML file: %s", e) raise ValueError("Error parsing YAML file.") from e except Exception as e: + logger.error("Unexpected error occurred while parsing the YAML file: %s", e) raise RuntimeError("An unexpected error occurred while parsing the YAML file.") from e if not isinstance(data, dict): + logger.error("YAML data must be a dictionary, received: %s", type(data)) raise TypeError("YAML data must be a dictionary.") # Process self_identification try: + logger.debug("Processing self_identification") self.self_identification = SelfIdentification(**data['self_identification']) + logger.debug("self_identification processed: %s", self.self_identification) except KeyError as e: + logger.error("Required field %s is missing in self_identification data.", e) raise KeyError(f"Required field {e} is missing in self_identification data.") from e except TypeError as e: + logger.error("Error in self_identification data: %s", e) raise TypeError(f"Error in self_identification data: {e}") from e except AttributeError as e: + logger.error("Attribute error in self_identification processing: %s", e) raise AttributeError("Attribute error in self_identification processing.") from e except Exception as e: + logger.error("An unexpected error occurred while processing self_identification: %s", e) raise RuntimeError("An unexpected error occurred while processing self_identification.") from e # Process legal_authorization try: + logger.debug("Processing legal_authorization") self.legal_authorization = LegalAuthorization(**data['legal_authorization']) + logger.debug("legal_authorization processed: %s", self.legal_authorization) except KeyError as e: + logger.error("Required field %s is missing in legal_authorization data.", e) raise KeyError(f"Required field {e} is missing in legal_authorization data.") from e except TypeError as e: + logger.error("Error in legal_authorization data: %s", e) raise TypeError(f"Error in legal_authorization data: {e}") from e except AttributeError as e: + logger.error("Attribute error in legal_authorization processing: %s", e) raise AttributeError("Attribute error in legal_authorization processing.") from e except Exception as e: + logger.error("An unexpected error occurred while processing legal_authorization: %s", e) raise RuntimeError("An unexpected error occurred while processing legal_authorization.") from e # Process work_preferences try: + logger.debug("Processing work_preferences") self.work_preferences = WorkPreferences(**data['work_preferences']) + logger.debug("work_preferences processed: %s", self.work_preferences) except KeyError as e: + logger.error("Required field %s is missing in work_preferences data.", e) raise KeyError(f"Required field {e} is missing in work_preferences data.") from e except TypeError as e: + logger.error("Error in work_preferences data: %s", e) raise TypeError(f"Error in work_preferences data: {e}") from e except AttributeError as e: + logger.error("Attribute error in work_preferences processing: %s", e) raise AttributeError("Attribute error in work_preferences processing.") from e except Exception as e: + logger.error("An unexpected error occurred while processing work_preferences: %s", e) raise RuntimeError("An unexpected error occurred while processing work_preferences.") from e # Process availability try: + logger.debug("Processing availability") self.availability = Availability(**data['availability']) + logger.debug("availability processed: %s", self.availability) except KeyError as e: + logger.error("Required field %s is missing in availability data.", e) raise KeyError(f"Required field {e} is missing in availability data.") from e except TypeError as e: + logger.error("Error in availability data: %s", e) raise TypeError(f"Error in availability data: {e}") from e except AttributeError as e: + logger.error("Attribute error in availability processing: %s", e) raise AttributeError("Attribute error in availability processing.") from e except Exception as e: + logger.error("An unexpected error occurred while processing availability: %s", e) raise RuntimeError("An unexpected error occurred while processing availability.") from e # Process salary_expectations try: + logger.debug("Processing salary_expectations") self.salary_expectations = SalaryExpectations(**data['salary_expectations']) + logger.debug("salary_expectations processed: %s", self.salary_expectations) except KeyError as e: + logger.error("Required field %s is missing in salary_expectations data.", e) raise KeyError(f"Required field {e} is missing in salary_expectations data.") from e except TypeError as e: + logger.error("Error in salary_expectations data: %s", e) raise TypeError(f"Error in salary_expectations data: {e}") from e except AttributeError as e: + logger.error("Attribute error in salary_expectations processing: %s", e) raise AttributeError("Attribute error in salary_expectations processing.") from e except Exception as e: + logger.error("An unexpected error occurred while processing salary_expectations: %s", e) raise RuntimeError("An unexpected error occurred while processing salary_expectations.") from e - # Process additional fields - - + logger.debug("JobApplicationProfile initialization completed successfully.") def __str__(self): + logger.debug("Generating string representation of JobApplicationProfile") + def format_dataclass(obj): return "\n".join(f"{field.name}: {getattr(obj, field.name)}" for field in obj.__dataclass_fields__.values()) - return (f"Self Identification:\n{format_dataclass(self.self_identification)}\n\n" - f"Legal Authorization:\n{format_dataclass(self.legal_authorization)}\n\n" - f"Work Preferences:\n{format_dataclass(self.work_preferences)}\n\n" - f"Availability: {self.availability.notice_period}\n\n" - f"Salary Expectations: {self.salary_expectations.salary_range_usd}\n\n") + formatted_str = (f"Self Identification:\n{format_dataclass(self.self_identification)}\n\n" + f"Legal Authorization:\n{format_dataclass(self.legal_authorization)}\n\n" + f"Work Preferences:\n{format_dataclass(self.work_preferences)}\n\n" + f"Availability: {self.availability.notice_period}\n\n" + f"Salary Expectations: {self.salary_expectations.salary_range_usd}\n\n") + logger.debug("String representation generated: %s", formatted_str) + return formatted_str diff --git a/src/linkedIn_authenticator.py b/src/linkedIn_authenticator.py index 6a0c02a..6c49dfc 100644 --- a/src/linkedIn_authenticator.py +++ b/src/linkedIn_authenticator.py @@ -1,29 +1,43 @@ +import random import time + from selenium.common.exceptions import NoSuchElementException, TimeoutException from selenium.webdriver.common.by import By -from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC +from selenium.webdriver.support.ui import WebDriverWait + +from src.utils import logger + class LinkedInAuthenticator: - + def __init__(self, driver=None): self.driver = driver self.email = "" self.password = "" + logger.debug("LinkedInAuthenticator initialized with driver: %s", driver) def set_secrets(self, email, password): self.email = email self.password = password + logger.debug("Secrets set with email: %s", email) def start(self): - print("Starting Chrome browser to log in to LinkedIn.") - self.driver.get('https://www.linkedin.com') + logger.info("Starting Chrome browser to log in to LinkedIn.") + self.driver.get('https://www.linkedin.com/feed') self.wait_for_page_load() - if not self.is_logged_in(): + + time.sleep(3) + + if self.is_logged_in(): + logger.info("User is already logged in. Skipping login process.") + return + else: + logger.info("User is not logged in. Proceeding with login.") self.handle_login() def handle_login(self): - print("Navigating to the LinkedIn login page...") + logger.info("Navigating to the LinkedIn login page...") self.driver.get("https://www.linkedin.com/login") if 'feed' in self.driver.current_url: print("User is already logged in.") @@ -31,50 +45,98 @@ class LinkedInAuthenticator: try: self.enter_credentials() self.submit_login_form() - except NoSuchElementException: - print("Could not log in to LinkedIn. Please check your credentials.") - time.sleep(35) #TODO fix better + except NoSuchElementException as e: + logger.error("Could not log in to LinkedIn. Element not found: %s", e) + time.sleep(random.uniform(3, 5)) self.handle_security_check() def enter_credentials(self): try: + logger.debug("Entering credentials...") email_field = WebDriverWait(self.driver, 10).until( EC.presence_of_element_located((By.ID, "username")) ) email_field.send_keys(self.email) + logger.debug("Email entered: %s", self.email) password_field = self.driver.find_element(By.ID, "password") password_field.send_keys(self.password) + logger.debug("Password entered.") except TimeoutException: + logger.error("Login form not found. Aborting login.") print("Login form not found. Aborting login.") def submit_login_form(self): try: + logger.debug("Submitting login form...") login_button = self.driver.find_element(By.XPATH, '//button[@type="submit"]') login_button.click() + logger.debug("Login form submitted.") except NoSuchElementException: + logger.error("Login button not found. Please verify the page structure.") print("Login button not found. Please verify the page structure.") def handle_security_check(self): try: + logger.debug("Handling security check...") WebDriverWait(self.driver, 10).until( EC.url_contains('https://www.linkedin.com/checkpoint/challengesV2/') ) + logger.warning("Security checkpoint detected. Please complete the challenge.") print("Security checkpoint detected. Please complete the challenge.") WebDriverWait(self.driver, 300).until( EC.url_contains('https://www.linkedin.com/feed/') ) + logger.info("Security check completed") print("Security check completed") except TimeoutException: + logger.error("Security check not completed within the timeout.") print("Security check not completed. Please try again later.") def is_logged_in(self): - self.driver.get('https://www.linkedin.com/') - return self.driver.current_url == 'https://www.linkedin.com/feed/' + # target_url = 'https://www.linkedin.com/feed' + # + # # Navigate to the target URL if not already there + # if self.driver.current_url != target_url: + # logger.debug("Navigating to target URL: %s", target_url) + # self.driver.get(target_url) + + try: + # Increase the wait time for the page elements to load + logger.debug("Checking if user is logged in...") + WebDriverWait(self.driver, 10).until( + EC.presence_of_element_located((By.CLASS_NAME, 'share-box-feed-entry__trigger')) + ) + + # Check for the presence of the "Start a post" button + buttons = self.driver.find_elements(By.CLASS_NAME, 'share-box-feed-entry__trigger') + logger.debug("Found %d 'Start a post' buttons", len(buttons)) + + for i, button in enumerate(buttons): + logger.debug("Button %d text: %s", i + 1, button.text.strip()) + + if any(button.text.strip().lower() == 'start a post' for button in buttons): + logger.info("Found 'Start a post' button indicating user is logged in.") + return True + + profile_img_elements = self.driver.find_elements(By.XPATH, "//img[contains(@alt, 'Photo of')]") + if profile_img_elements: + logger.info("Profile image found. Assuming user is logged in.") + return True + + logger.info("Did not find 'Start a post' button or profile image. User might not be logged in.") + return False + + except TimeoutException: + logger.error("Page elements took too long to load or were not found.") + return False def wait_for_page_load(self, timeout=10): try: + logger.debug("Waiting for page to load with timeout: %s seconds", timeout) WebDriverWait(self.driver, timeout).until( lambda d: d.execute_script('return document.readyState') == 'complete' ) + logger.debug("Page load completed.") except TimeoutException: + logger.error("Page load timed out.") print("Page load timed out.") diff --git a/src/linkedIn_bot_facade.py b/src/linkedIn_bot_facade.py index 33dc06a..2f1732c 100644 --- a/src/linkedIn_bot_facade.py +++ b/src/linkedIn_bot_facade.py @@ -1,8 +1,13 @@ +from src.utils import logger + + class LinkedInBotState: def __init__(self): + logger.debug("Initializing LinkedInBotState") self.reset() def reset(self): + logger.debug("Resetting LinkedInBotState") self.credentials_set = False self.api_key_set = False self.job_application_profile_set = False @@ -11,12 +16,17 @@ class LinkedInBotState: self.logged_in = False def validate_state(self, required_keys): + logger.debug("Validating LinkedInBotState with required keys: %s", required_keys) for key in required_keys: if not getattr(self, key): + logger.error("State validation failed: %s is not set", key) raise ValueError(f"{key.replace('_', ' ').capitalize()} must be set before proceeding.") + logger.debug("State validation passed") + class LinkedInBotFacade: def __init__(self, login_component, apply_component): + logger.debug("Initializing LinkedInBotFacade") self.login_component = login_component self.apply_component = apply_component self.state = LinkedInBotState() @@ -27,47 +37,65 @@ class LinkedInBotFacade: self.parameters = None def set_job_application_profile_and_resume(self, job_application_profile, resume): + logger.debug("Setting job application profile and resume") self._validate_non_empty(job_application_profile, "Job application profile") self._validate_non_empty(resume, "Resume") self.job_application_profile = job_application_profile self.resume = resume self.state.job_application_profile_set = True + logger.debug("Job application profile and resume set successfully") def set_secrets(self, email, password): + logger.debug("Setting secrets: email and password") self._validate_non_empty(email, "Email") self._validate_non_empty(password, "Password") self.email = email self.password = password self.state.credentials_set = True + logger.debug("Secrets set successfully") def set_gpt_answerer_and_resume_generator(self, gpt_answerer_component, resume_generator_manager): + logger.debug("Setting GPT answerer and resume generator") self._ensure_job_profile_and_resume_set() gpt_answerer_component.set_job_application_profile(self.job_application_profile) gpt_answerer_component.set_resume(self.resume) self.apply_component.set_gpt_answerer(gpt_answerer_component) self.apply_component.set_resume_generator_manager(resume_generator_manager) self.state.gpt_answerer_set = True + logger.debug("GPT answerer and resume generator set successfully") def set_parameters(self, parameters): + logger.debug("Setting parameters") self._validate_non_empty(parameters, "Parameters") self.parameters = parameters self.apply_component.set_parameters(parameters) self.state.parameters_set = True + logger.debug("Parameters set successfully") def start_login(self): + logger.debug("Starting login process") self.state.validate_state(['credentials_set']) self.login_component.set_secrets(self.email, self.password) self.login_component.start() self.state.logged_in = True + logger.debug("Login process completed successfully") def start_apply(self): + logger.debug("Starting apply process") self.state.validate_state(['logged_in', 'job_application_profile_set', 'gpt_answerer_set', 'parameters_set']) self.apply_component.start_applying() + logger.debug("Apply process started successfully") def _validate_non_empty(self, value, name): + logger.debug("Validating that %s is not empty", name) if not value: + logger.error("Validation failed: %s is empty", name) raise ValueError(f"{name} cannot be empty.") + logger.debug("Validation passed for %s", name) def _ensure_job_profile_and_resume_set(self): + logger.debug("Ensuring job profile and resume are set") if not self.state.job_application_profile_set: + logger.error("Job application profile and resume are not set") raise ValueError("Job application profile and resume must be set before proceeding.") + logger.debug("Job profile and resume are set") diff --git a/src/linkedIn_easy_applier.py b/src/linkedIn_easy_applier.py index c9b9625..951fea8 100644 --- a/src/linkedIn_easy_applier.py +++ b/src/linkedIn_easy_applier.py @@ -3,24 +3,29 @@ import json import os import random import re -import tempfile import time import traceback -from datetime import date from typing import List, Optional, Any, Tuple + +from httpx import HTTPStatusError from reportlab.lib.pagesizes import letter from reportlab.pdfgen import canvas -from selenium.common.exceptions import NoSuchElementException +from selenium.common.exceptions import NoSuchElementException, TimeoutException +from selenium.webdriver import ActionChains from selenium.webdriver.common.by import By from selenium.webdriver.common.keys import Keys from selenium.webdriver.remote.webelement import WebElement from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support.ui import Select, WebDriverWait -from selenium.webdriver import ActionChains + import src.utils as utils +from src.utils import logger + class LinkedInEasyApplier: - def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: List[Tuple[str, str, str]], gpt_answerer: Any, resume_generator_manager): + def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: List[Tuple[str, str, str]], + gpt_answerer: Any, resume_generator_manager): + logger.debug("Initializing LinkedInEasyApplier") if resume_dir is None or not os.path.exists(resume_dir): resume_dir = None self.driver = driver @@ -30,106 +35,248 @@ class LinkedInEasyApplier: self.resume_generator_manager = resume_generator_manager self.all_data = self._load_questions_from_json() + logger.debug("LinkedInEasyApplier initialized successfully") + + def _load_questions_from_json(self) -> List[dict]: output_file = 'answers.json' + logger.debug("Loading questions from JSON file: %s", output_file) try: - try: - with open(output_file, 'r') as f: - try: - data = json.load(f) - if not isinstance(data, list): - raise ValueError("JSON file format is incorrect. Expected a list of questions.") - except json.JSONDecodeError: - data = [] - except FileNotFoundError: - data = [] + with open(output_file, 'r') as f: + try: + data = json.load(f) + if not isinstance(data, list): + raise ValueError("JSON file format is incorrect. Expected a list of questions.") + except json.JSONDecodeError: + logger.error("JSON decoding failed") + data = [] + logger.debug("Questions loaded successfully from JSON") return data + except FileNotFoundError: + logger.warning("JSON file not found, returning empty list") + return [] except Exception: tb_str = traceback.format_exc() + logger.error("Error loading questions data from JSON file: %s", tb_str) raise Exception(f"Error loading questions data from JSON file: \nTraceback:\n{tb_str}") + + def check_for_premium_redirect(self, job: Any, max_attempts=3): + """Проверяет, был ли выполнен редирект на страницу LinkedIn Premium. + В случае редиректа возвращает пользователя на исходную страницу вакансии.""" + current_url = self.driver.current_url + attempts = 0 + + while "linkedin.com/premium" in current_url and attempts < max_attempts: + logger.warning("Redirected to LinkedIn Premium page. Attempting to return to job page.") + attempts += 1 + + self.driver.get(job.link) + time.sleep(2) + current_url = self.driver.current_url + + if "linkedin.com/premium" in current_url: + logger.error("Failed to return to job page after %d attempts. Cannot apply for the job.", max_attempts) + raise Exception( + f"Redirected to LinkedIn Premium page and failed to return after {max_attempts} attempts. Job application aborted.") + + def job_apply(self, job: Any): - self.driver.get(job.link) - time.sleep(random.uniform(3, 5)) + logger.debug("Starting job application for job: %s", job) + try: - easy_apply_button = self._find_easy_apply_button() - job.set_job_description(self._get_job_description()) - job.set_recruiter_link(self._get_job_recruiter()) + self.driver.get(job.link) + logger.debug("Navigated to job link: %s", job.link) + except Exception as e: + logger.error("Failed to navigate to job link: %s, error: %s", job.link, str(e)) + raise + + time.sleep(random.uniform(3, 5)) + self.check_for_premium_redirect(job) + + try: + + self.driver.execute_script("document.activeElement.blur();") + logger.debug("Focus removed from the active element") + + self.check_for_premium_redirect(job) + + easy_apply_button = self._find_easy_apply_button(job) + + self.check_for_premium_redirect(job) + + logger.debug("Retrieving job description") + job_description = self._get_job_description() + job.set_job_description(job_description) + logger.debug("Job description set: %s", job_description[:100]) + + logger.debug("Retrieving recruiter link") + recruiter_link = self._get_job_recruiter() + job.set_recruiter_link(recruiter_link) + logger.debug("Recruiter link set: %s", recruiter_link) + + logger.debug("Attempting to click 'Easy Apply' button") actions = ActionChains(self.driver) actions.move_to_element(easy_apply_button).click().perform() - self.gpt_answerer.set_job(job) - self._fill_application_form(job) - except Exception: - tb_str = traceback.format_exc() - self._discard_application() - raise Exception(f"Failed to apply to job! Original exception: \nTraceback:\n{tb_str}") + logger.debug("'Easy Apply' button clicked successfully") - def _find_easy_apply_button(self) -> WebElement: + logger.debug("Passing job information to GPT Answerer") + self.gpt_answerer.set_job(job) + + logger.debug("Filling out application form") + self._fill_application_form(job) + logger.debug("Job application process completed successfully for job: %s", job) + + except Exception as e: + + tb_str = traceback.format_exc() + logger.error("Failed to apply to job: %s. Error traceback: %s", job, tb_str) + + logger.debug("Discarding application due to failure") + self._discard_application() + + raise Exception(f"Failed to apply to job! Original exception:\nTraceback:\n{tb_str}") + + def _find_easy_apply_button(self, job: Any) -> WebElement: + logger.debug("Searching for 'Easy Apply' button") attempt = 0 + + search_methods = [ + { + 'description': "find all 'Easy Apply' buttons using find_elements", + 'find_elements': True, + 'xpath': '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]' + }, + { + 'description': "'aria-label' containing 'Easy Apply to'", + 'xpath': '//button[contains(@aria-label, "Easy Apply to")]' + }, + { + 'description': "button text search", + 'xpath': '//button[contains(text(), "Easy Apply") or contains(text(), "Apply now")]' + } + ] + while attempt < 2: + + self.check_for_premium_redirect(job) self._scroll_page() - buttons = WebDriverWait(self.driver, 10).until( - EC.presence_of_all_elements_located( - (By.XPATH, '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]') - ) - ) - for index, _ in enumerate(buttons): + + for method in search_methods: try: - button = WebDriverWait(self.driver, 10).until( - EC.element_to_be_clickable( - (By.XPATH, f'(//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")])[{index + 1}]') + logger.debug(f"Attempting search using {method['description']}") + + if method.get('find_elements'): + # Поиск всех кнопок "Easy Apply" + buttons = self.driver.find_elements(By.XPATH, method['xpath']) + if buttons: + for index, button in enumerate(buttons): + try: + + WebDriverWait(self.driver, 10).until(EC.visibility_of(button)) + WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button)) + logger.debug(f"Found 'Easy Apply' button {index + 1}, attempting to click") + return button + except Exception as e: + logger.warning(f"Button {index + 1} found but not clickable: {e}") + else: + raise TimeoutException("No 'Easy Apply' buttons found") + else: + + button = WebDriverWait(self.driver, 10).until( + EC.presence_of_element_located((By.XPATH, method['xpath'])) ) - ) - return button + WebDriverWait(self.driver, 10).until(EC.visibility_of(button)) + WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button)) + logger.debug("Found 'Easy Apply' button, attempting to click") + return button + + except TimeoutException: + logger.warning(f"Timeout during search using {method['description']}") except Exception as e: - pass + logger.warning( + f"Failed to click 'Easy Apply' button using {method['description']} on attempt {attempt + 1}: {e}") + + self.check_for_premium_redirect(job) + if attempt == 0: + logger.debug("Refreshing page to retry finding 'Easy Apply' button") self.driver.refresh() - time.sleep(3) + time.sleep(random.randint(3, 5)) attempt += 1 + + page_source = self.driver.page_source + logger.error("No clickable 'Easy Apply' button found after 2 attempts. Page source:\n%s", page_source) raise Exception("No clickable 'Easy Apply' button found") - + + def _get_job_description(self) -> str: + logger.debug("Getting job description") try: - see_more_button = self.driver.find_element(By.XPATH, '//button[@aria-label="Click to see more description"]') - actions = ActionChains(self.driver) - actions.move_to_element(see_more_button).click().perform() - time.sleep(2) + try: + see_more_button = self.driver.find_element(By.XPATH, + '//button[@aria-label="Click to see more description"]') + actions = ActionChains(self.driver) + actions.move_to_element(see_more_button).click().perform() + time.sleep(2) + except NoSuchElementException: + logger.debug("See more button not found, skipping") + description = self.driver.find_element(By.CLASS_NAME, 'jobs-description-content__text').text + logger.debug("Job description retrieved successfully") return description except NoSuchElementException: tb_str = traceback.format_exc() - raise Exception("Job description 'See more' button not found: \nTraceback:\n{tb_str}") + logger.error("Job description not found: %s", tb_str) + raise Exception(f"Job description not found: \nTraceback:\n{tb_str}") except Exception: tb_str = traceback.format_exc() + logger.error("Error getting Job description: %s", tb_str) raise Exception(f"Error getting Job description: \nTraceback:\n{tb_str}") def _get_job_recruiter(self): + logger.debug("Getting job recruiter information") try: hiring_team_section = WebDriverWait(self.driver, 10).until( EC.presence_of_element_located((By.XPATH, '//h2[text()="Meet the hiring team"]')) ) - recruiter_element = hiring_team_section.find_element(By.XPATH, './/following::a[contains(@href, "linkedin.com/in/")]') - recruiter_link = recruiter_element.get_attribute('href') - return recruiter_link + logger.debug("Hiring team section found") + + recruiter_elements = hiring_team_section.find_elements(By.XPATH, + './/following::a[contains(@href, "linkedin.com/in/")]') + + if recruiter_elements: + recruiter_element = recruiter_elements[0] + recruiter_link = recruiter_element.get_attribute('href') + logger.debug("Job recruiter link retrieved successfully: %s", recruiter_link) + return recruiter_link + else: + logger.debug("No recruiter link found in the hiring team section") + return "" except Exception as e: + logger.warning("Failed to retrieve recruiter information: %s", e) return "" def _scroll_page(self) -> None: + logger.debug("Scrolling the page") scrollable_element = self.driver.find_element(By.TAG_NAME, 'html') utils.scroll_slow(self.driver, scrollable_element, step=300, reverse=False) utils.scroll_slow(self.driver, scrollable_element, step=300, reverse=True) def _fill_application_form(self, job): + logger.debug("Filling out application form for job: %s", job) while True: self.fill_up(job) if self._next_or_submit(): + logger.debug("Application form submitted") break def _next_or_submit(self): + logger.debug("Clicking 'Next' or 'Submit' button") next_button = self.driver.find_element(By.CLASS_NAME, "artdeco-button--primary") button_text = next_button.text.lower() if 'submit application' in button_text: + logger.debug("Submit button found, submitting application") self._unfollow_company() time.sleep(random.uniform(1.5, 2.5)) next_button.click() @@ -142,104 +289,304 @@ class LinkedInEasyApplier: def _unfollow_company(self) -> None: try: + logger.debug("Unfollowing company") follow_checkbox = self.driver.find_element( By.XPATH, "//label[contains(.,'to stay up to date with their page.')]") follow_checkbox.click() except Exception as e: - pass + logger.warning("Failed to unfollow company: %s", e) def _check_for_errors(self) -> None: + logger.debug("Checking for form errors") error_elements = self.driver.find_elements(By.CLASS_NAME, 'artdeco-inline-feedback--error') if error_elements: + logger.error("Form submission failed with errors: %s", [e.text for e in error_elements]) raise Exception(f"Failed answering or file upload. {str([e.text for e in error_elements])}") def _discard_application(self) -> None: + logger.debug("Discarding application") try: self.driver.find_element(By.CLASS_NAME, 'artdeco-modal__dismiss').click() time.sleep(random.uniform(3, 5)) self.driver.find_elements(By.CLASS_NAME, 'artdeco-modal__confirm-dialog-btn')[0].click() time.sleep(random.uniform(3, 5)) except Exception as e: - pass + logger.warning("Failed to discard application: %s", e) def fill_up(self, job) -> None: - easy_apply_content = self.driver.find_element(By.CLASS_NAME, 'jobs-easy-apply-content') - pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4') - for element in pb4_elements: - self._process_form_element(element, job) - + logger.debug("Filling up form sections for job: %s", job) + + try: + easy_apply_content = WebDriverWait(self.driver, 10).until( + EC.presence_of_element_located((By.CLASS_NAME, 'jobs-easy-apply-content')) + ) + + pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4') + for element in pb4_elements: + self._process_form_element(element, job) + except Exception as e: + logger.error(f"Failed to find form elements: {e}") + def _process_form_element(self, element: WebElement, job) -> None: + logger.debug("Processing form element") if self._is_upload_field(element): self._handle_upload_fields(element, job) else: self._fill_additional_questions() + def _handle_dropdown_fields(self, element: WebElement) -> None: + logger.debug("Handling dropdown fields") + + dropdown = element.find_element(By.TAG_NAME, 'select') + select = Select(dropdown) + + options = [option.text for option in select.options] + logger.debug(f"Dropdown options found: {options}") + + parent_element = dropdown.find_element(By.XPATH, '../..') + + label_elements = parent_element.find_elements(By.TAG_NAME, 'label') + if label_elements: + question_text = label_elements[0].text.lower() + else: + question_text = "unknown" + + logger.debug(f"Detected question text: {question_text}") + + existing_answer = None + for item in self.all_data: + if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown': + existing_answer = item['answer'] + break + + if existing_answer: + logger.debug(f"Found existing answer for question '{question_text}': {existing_answer}") + else: + + logger.debug(f"No existing answer found, querying model for: {question_text}") + existing_answer = self.gpt_answerer.answer_question_from_options(question_text, options) + logger.debug(f"Model provided answer: {existing_answer}") + self._save_questions_to_json({'type': 'dropdown', 'question': question_text, 'answer': existing_answer}) + + if existing_answer in options: + select.select_by_visible_text(existing_answer) + logger.debug(f"Selected option: {existing_answer}") + else: + logger.error(f"Answer '{existing_answer}' is not a valid option in the dropdown") + raise Exception(f"Invalid option selected: {existing_answer}") + def _is_upload_field(self, element: WebElement) -> bool: - return bool(element.find_elements(By.XPATH, ".//input[@type='file']")) + is_upload = bool(element.find_elements(By.XPATH, ".//input[@type='file']")) + logger.debug("Element is upload field: %s", is_upload) + return is_upload def _handle_upload_fields(self, element: WebElement, job) -> None: + logger.debug("Handling upload fields") + + try: + show_more_button = self.driver.find_element(By.XPATH, + "//button[contains(@aria-label, 'Show more resumes')]") + show_more_button.click() + logger.debug("Clicked 'Show more resumes' button") + except NoSuchElementException: + logger.debug("'Show more resumes' button not found, continuing...") + file_upload_elements = self.driver.find_elements(By.XPATH, "//input[@type='file']") for element in file_upload_elements: parent = element.find_element(By.XPATH, "..") self.driver.execute_script("arguments[0].classList.remove('hidden')", element) + output = self.gpt_answerer.resume_or_cover(parent.text.lower()) if 'resume' in output: + logger.debug("Uploading resume") if self.resume_path is not None and self.resume_path.resolve().is_file(): element.send_keys(str(self.resume_path.resolve())) + logger.debug(f"Resume uploaded from path: {self.resume_path.resolve()}") else: + logger.debug("Resume path not found or invalid, generating new resume") self._create_and_upload_resume(element, job) elif 'cover' in output: - self._create_and_upload_cover_letter(element) + logger.debug("Uploading cover letter") + self._create_and_upload_cover_letter(element, job) + + logger.debug("Finished handling upload fields") def _create_and_upload_resume(self, element, job): + logger.debug("Starting the process of creating and uploading resume.") folder_path = 'generated_cv' - os.makedirs(folder_path, exist_ok=True) + try: - file_path_pdf = os.path.join(folder_path, f"CV_{random.randint(0, 9999)}.pdf") - with open(file_path_pdf, "xb") as f: - f.write(base64.b64decode(self.resume_generator_manager.pdf_base64(job_description_text=job.description))) + if not os.path.exists(folder_path): + logger.debug(f"Creating directory at path: {folder_path}") + os.makedirs(folder_path, exist_ok=True) + except Exception as e: + logger.error(f"Failed to create directory: {folder_path}. Error: {e}") + raise + + while True: + try: + timestamp = int(time.time()) + file_path_pdf = os.path.join(folder_path, f"CV_{timestamp}.pdf") + logger.debug(f"Generated file path for resume: {file_path_pdf}") + + logger.debug(f"Generating resume for job: {job.title} at {job.company}") + resume_pdf_base64 = self.resume_generator_manager.pdf_base64(job_description_text=job.description) + with open(file_path_pdf, "xb") as f: + f.write(base64.b64decode(resume_pdf_base64)) + logger.debug(f"Resume successfully generated and saved to: {file_path_pdf}") + + break + except HTTPStatusError as e: + if e.response.status_code == 429: + + retry_after = e.response.headers.get('retry-after') + retry_after_ms = e.response.headers.get('retry-after-ms') + + if retry_after: + wait_time = int(retry_after) + logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...") + elif retry_after_ms: + wait_time = int(retry_after_ms) / 1000.0 + logger.warning(f"Rate limit exceeded, waiting {wait_time} milliseconds before retrying...") + else: + wait_time = 20 + logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...") + + time.sleep(wait_time) + else: + logger.error(f"HTTP error: {e}") + raise + + except Exception as e: + logger.error(f"Failed to generate resume: {e}") + tb_str = traceback.format_exc() + logger.error(f"Traceback: {tb_str}") + if "RateLimitError" in str(e): + logger.warning("Rate limit error encountered, retrying...") + time.sleep(20) + else: + raise + + file_size = os.path.getsize(file_path_pdf) + max_file_size = 2 * 1024 * 1024 # 2 MB + logger.debug(f"Resume file size: {file_size} bytes") + if file_size > max_file_size: + logger.error(f"Resume file size exceeds 2 MB: {file_size} bytes") + raise ValueError("Resume file size exceeds the maximum limit of 2 MB.") + + allowed_extensions = {'.pdf', '.doc', '.docx'} + file_extension = os.path.splitext(file_path_pdf)[1].lower() + logger.debug(f"Resume file extension: {file_extension}") + if file_extension not in allowed_extensions: + logger.error(f"Invalid resume file format: {file_extension}") + raise ValueError("Resume file format is not allowed. Only PDF, DOC, and DOCX formats are supported.") + + try: + logger.debug(f"Uploading resume from path: {file_path_pdf}") element.send_keys(os.path.abspath(file_path_pdf)) job.pdf_path = os.path.abspath(file_path_pdf) time.sleep(2) - except Exception: + logger.debug(f"Resume created and uploaded successfully: {file_path_pdf}") + except Exception as e: tb_str = traceback.format_exc() + logger.error(f"Resume upload failed: {tb_str}") raise Exception(f"Upload failed: \nTraceback:\n{tb_str}") - def _create_and_upload_cover_letter(self, element: WebElement) -> None: - cover_letter = self.gpt_answerer.answer_question_textual_wide_range("Write a cover letter") - with tempfile.NamedTemporaryFile(delete=False, suffix='.pdf') as temp_pdf_file: - letter_path = temp_pdf_file.name - c = canvas.Canvas(letter_path, pagesize=letter) - _, height = letter - text_object = c.beginText(100, height - 100) - text_object.setFont("Helvetica", 12) - text_object.textLines(cover_letter) - c.drawText(text_object) - c.save() - element.send_keys(letter_path) + def _create_and_upload_cover_letter(self, element: WebElement, job) -> None: + logger.debug("Starting the process of creating and uploading cover letter.") + + cover_letter_text = self.gpt_answerer.answer_question_textual_wide_range("Write a cover letter") + + folder_path = 'generated_cv' + + try: + + if not os.path.exists(folder_path): + logger.debug(f"Creating directory at path: {folder_path}") + os.makedirs(folder_path, exist_ok=True) + except Exception as e: + logger.error(f"Failed to create directory: {folder_path}. Error: {e}") + raise + + while True: + try: + timestamp = int(time.time()) + file_path_pdf = os.path.join(folder_path, f"Cover_Letter_{timestamp}.pdf") + logger.debug(f"Generated file path for cover letter: {file_path_pdf}") + + c = canvas.Canvas(file_path_pdf, pagesize=letter) + _, height = letter + text_object = c.beginText(100, height - 100) + text_object.setFont("Helvetica", 12) + text_object.textLines(cover_letter_text) + c.drawText(text_object) + c.save() + logger.debug(f"Cover letter successfully generated and saved to: {file_path_pdf}") + + break + except Exception as e: + logger.error(f"Failed to generate cover letter: {e}") + tb_str = traceback.format_exc() + logger.error(f"Traceback: {tb_str}") + raise + + file_size = os.path.getsize(file_path_pdf) + max_file_size = 2 * 1024 * 1024 # 2 MB + logger.debug(f"Cover letter file size: {file_size} bytes") + if file_size > max_file_size: + logger.error(f"Cover letter file size exceeds 2 MB: {file_size} bytes") + raise ValueError("Cover letter file size exceeds the maximum limit of 2 MB.") + + allowed_extensions = {'.pdf', '.doc', '.docx'} + file_extension = os.path.splitext(file_path_pdf)[1].lower() + logger.debug(f"Cover letter file extension: {file_extension}") + if file_extension not in allowed_extensions: + logger.error(f"Invalid cover letter file format: {file_extension}") + raise ValueError("Cover letter file format is not allowed. Only PDF, DOC, and DOCX formats are supported.") + + try: + + logger.debug(f"Uploading cover letter from path: {file_path_pdf}") + element.send_keys(os.path.abspath(file_path_pdf)) + job.cover_letter_path = os.path.abspath(file_path_pdf) + time.sleep(2) + logger.debug(f"Cover letter created and uploaded successfully: {file_path_pdf}") + except Exception as e: + tb_str = traceback.format_exc() + logger.error(f"Cover letter upload failed: {tb_str}") + raise Exception(f"Upload failed: \nTraceback:\n{tb_str}") def _fill_additional_questions(self) -> None: + logger.debug("Filling additional questions") form_sections = self.driver.find_elements(By.CLASS_NAME, 'jobs-easy-apply-form-section__grouping') for section in form_sections: self._process_form_section(section) - def _process_form_section(self, section: WebElement) -> None: + logger.debug("Processing form section") if self._handle_terms_of_service(section): + logger.debug("Handled terms of service") return if self._find_and_handle_radio_question(section): + logger.debug("Handled radio question") return if self._find_and_handle_textbox_question(section): + logger.debug("Handled textbox question") return if self._find_and_handle_date_question(section): + logger.debug("Handled date question") return + if self._find_and_handle_dropdown_question(section): + logger.debug("Handled dropdown question") return def _handle_terms_of_service(self, element: WebElement) -> bool: checkbox = element.find_elements(By.TAG_NAME, 'label') - if checkbox and any(term in checkbox[0].text.lower() for term in ['terms of service', 'privacy policy', 'terms of use']): + if checkbox and any( + term in checkbox[0].text.lower() for term in ['terms of service', 'privacy policy', 'terms of use']): checkbox[0].click() + logger.debug("Clicked terms of service checkbox") return True return False @@ -249,39 +596,81 @@ class LinkedInEasyApplier: if radios: question_text = section.text.lower() options = [radio.text.lower() for radio in radios] + existing_answer = None for item in self.all_data: if self._sanitize_text(question_text) in item['question'] and item['type'] == 'radio': existing_answer = item - self._select_radio(radios, existing_answer['answer']) - return True + + break + if existing_answer: + self._select_radio(radios, existing_answer['answer']) + logger.debug("Selected existing radio answer") + return True + answer = self.gpt_answerer.answer_question_from_options(question_text, options) self._save_questions_to_json({'type': 'radio', 'question': question_text, 'answer': answer}) self._select_radio(radios, answer) + logger.debug("Selected new radio answer") return True return False def _find_and_handle_textbox_question(self, section: WebElement) -> bool: + logger.debug("Searching for text fields in the section.") text_fields = section.find_elements(By.TAG_NAME, 'input') + section.find_elements(By.TAG_NAME, 'textarea') + if text_fields: text_field = text_fields[0] - question_text = section.find_element(By.TAG_NAME, 'label').text.lower() + question_text = section.find_element(By.TAG_NAME, 'label').text.lower().strip() + logger.debug(f"Found text field with label: {question_text}") + is_numeric = self._is_numeric_field(text_field) - if is_numeric: - question_type = 'numeric' - answer = self.gpt_answerer.answer_question_numeric(question_text) - else: - question_type = 'textbox' - answer = self.gpt_answerer.answer_question_textual_wide_range(question_text) + logger.debug(f"Is the field numeric? {'Yes' if is_numeric else 'No'}") + existing_answer = None + question_type = 'numeric' if is_numeric else 'textbox' + for item in self.all_data: - if 'cover' not in item['question'] and item['question'] == self._sanitize_text(question_text) and item['type'] == question_type: + + + logger.debug( + f"Comparing sanitized stored question: '{self._sanitize_text(item['question'])}' and type: '{item.get('type')}' with current question: '{self._sanitize_text(question_text)}' and type: '{question_type}'") + + if self._sanitize_text(item['question']) == self._sanitize_text(question_text) and item.get( + 'type') == question_type: existing_answer = item - self._enter_text(text_field, existing_answer['answer']) - return True + logger.debug(f"Found existing answer in the data: {existing_answer['answer']}") + break + + if existing_answer: + self._enter_text(text_field, existing_answer['answer']) + logger.debug("Entered existing answer into the textbox.") + + time.sleep(1) + text_field.send_keys(Keys.ARROW_DOWN) + text_field.send_keys(Keys.ENTER) + logger.debug("Selected first option from the dropdown.") + return True + + if is_numeric: + answer = self.gpt_answerer.answer_question_numeric(question_text) + logger.debug(f"Generated numeric answer: {answer}") + else: + answer = self.gpt_answerer.answer_question_textual_wide_range(question_text) + logger.debug(f"Generated textual answer: {answer}") + + self._save_questions_to_json({'type': question_type, 'question': question_text, 'answer': answer}) self._enter_text(text_field, answer) + logger.debug("Entered new answer into the textbox and saved it to JSON.") + + time.sleep(1) + text_field.send_keys(Keys.ARROW_DOWN) + text_field.send_keys(Keys.ENTER) + logger.debug("Selected first option from the dropdown.") return True + + logger.debug("No text fields found in the section.") return False def _find_and_handle_date_question(self, section: WebElement) -> bool: @@ -294,49 +683,80 @@ class LinkedInEasyApplier: existing_answer = None for item in self.all_data: - if self._sanitize_text(question_text) in item['question'] and item['type'] == 'date': + if self._sanitize_text(question_text) in item['question'] and item['type'] == 'date': existing_answer = item - self._enter_text(date_field, existing_answer['answer']) - return True + + break + if existing_answer: + self._enter_text(date_field, existing_answer['answer']) + logger.debug("Entered existing date answer") + return True + self._save_questions_to_json({'type': 'date', 'question': question_text, 'answer': answer_text}) self._enter_text(date_field, answer_text) + logger.debug("Entered new date answer") return True return False def _find_and_handle_dropdown_question(self, section: WebElement) -> bool: try: + question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element') question_text = question.find_element(By.TAG_NAME, 'label').text.lower() - dropdown = question.find_element(By.TAG_NAME, 'select') - if dropdown: + logger.debug(f"Processing dropdown or combobox question: {question_text}") + + dropdowns = question.find_elements(By.TAG_NAME, 'select') + if dropdowns: + dropdown = dropdowns[0] select = Select(dropdown) options = [option.text for option in select.options] + + logger.debug(f"Dropdown options found: {options}") + + current_selection = select.first_selected_option.text + logger.debug(f"Current selection: {current_selection}") + existing_answer = None for item in self.all_data: - if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown': - existing_answer = item - self._select_dropdown_option(dropdown, existing_answer['answer']) - return True + if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown': + existing_answer = item['answer'] + break + + if existing_answer: + logger.debug(f"Found existing answer for question '{question_text}': {existing_answer}") + if current_selection != existing_answer: + logger.debug(f"Updating selection to: {existing_answer}") + self._select_dropdown_option(dropdown, existing_answer) + return True + + logger.debug(f"No existing answer found, querying model for: {question_text}") + answer = self.gpt_answerer.answer_question_from_options(question_text, options) self._save_questions_to_json({'type': 'dropdown', 'question': question_text, 'answer': answer}) self._select_dropdown_option(dropdown, answer) + logger.debug(f"Selected new dropdown answer: {answer}") return True - except Exception: + + return False + except Exception as e: + logger.warning(f"Failed to handle dropdown or combobox question: {e}") return False def _is_numeric_field(self, field: WebElement) -> bool: field_type = field.get_attribute('type').lower() - if 'numeric' in field_type: - return True - class_attribute = field.get_attribute("id") - return class_attribute and 'numeric' in class_attribute + field_id = field.get_attribute("id").lower() + is_numeric = 'numeric' in field_id or field_type == 'number' or ('text' == field_type and 'numeric' in field_id) + logger.debug("Field type: %s, Field ID: %s, Is numeric: %s", field_type, field_id, is_numeric) + return is_numeric def _enter_text(self, element: WebElement, text: str) -> None: + logger.debug("Entering text: %s", text) element.clear() element.send_keys(text) def _select_radio(self, radios: List[WebElement], answer: str) -> None: + logger.debug("Selecting radio option: %s", answer) for radio in radios: if answer in radio.text.lower(): radio.find_element(By.TAG_NAME, 'label').click() @@ -344,12 +764,14 @@ class LinkedInEasyApplier: radios[-1].find_element(By.TAG_NAME, 'label').click() def _select_dropdown_option(self, element: WebElement, text: str) -> None: + logger.debug("Selecting dropdown option: %s", text) select = Select(element) select.select_by_visible_text(text) def _save_questions_to_json(self, question_data: dict) -> None: output_file = 'answers.json' question_data['question'] = self._sanitize_text(question_data['question']) + logger.debug("Saving question data to JSON: %s", question_data) try: try: with open(output_file, 'r') as f: @@ -358,22 +780,22 @@ class LinkedInEasyApplier: if not isinstance(data, list): raise ValueError("JSON file format is incorrect. Expected a list of questions.") except json.JSONDecodeError: + logger.error("JSON decoding failed") data = [] except FileNotFoundError: + logger.warning("JSON file not found, creating new file") data = [] data.append(question_data) with open(output_file, 'w') as f: json.dump(data, f, indent=4) + logger.debug("Question data saved successfully to JSON") except Exception: tb_str = traceback.format_exc() + logger.error("Error saving questions data to JSON file: %s", tb_str) raise Exception(f"Error saving questions data to JSON file: \nTraceback:\n{tb_str}") def _sanitize_text(self, text: str) -> str: - sanitized_text = text.lower() - sanitized_text = sanitized_text.strip() - sanitized_text = sanitized_text.replace('"', '') - sanitized_text = sanitized_text.replace('\\', '') - sanitized_text = re.sub(r'[\x00-\x1F\x7F]', '', sanitized_text) - sanitized_text = sanitized_text.replace('\n', ' ').replace('\r', '') - sanitized_text = sanitized_text.rstrip(',') + sanitized_text = text.lower().strip().replace('"', '').replace('\\', '') + sanitized_text = re.sub(r'[\x00-\x1F\x7F]', '', sanitized_text).replace('\n', ' ').replace('\r', '').rstrip(',') + logger.debug("Sanitized text: %s", sanitized_text) return sanitized_text diff --git a/src/linkedIn_job_manager.py b/src/linkedIn_job_manager.py index e42c087..5b6e53b 100644 --- a/src/linkedIn_job_manager.py +++ b/src/linkedIn_job_manager.py @@ -1,37 +1,50 @@ +import json import os import random import time -import traceback from itertools import product from pathlib import Path + from selenium.common.exceptions import NoSuchElementException from selenium.webdriver.common.by import By + import src.utils as utils from src.job import Job from src.linkedIn_easy_applier import LinkedInEasyApplier -import json +from src.utils import logger class EnvironmentKeys: def __init__(self): + logger.debug("Initializing EnvironmentKeys") self.skip_apply = self._read_env_key_bool("SKIP_APPLY") self.disable_description_filter = self._read_env_key_bool("DISABLE_DESCRIPTION_FILTER") + logger.debug("EnvironmentKeys initialized: skip_apply=%s, disable_description_filter=%s", + self.skip_apply, self.disable_description_filter) @staticmethod def _read_env_key(key: str) -> str: - return os.getenv(key, "") + value = os.getenv(key, "") + logger.debug("Read environment key %s: %s", key, value) + return value @staticmethod def _read_env_key_bool(key: str) -> bool: - return os.getenv(key) == "True" + value = os.getenv(key) == "True" + logger.debug("Read environment key %s as bool: %s", key, value) + return value + class LinkedInJobManager: def __init__(self, driver): + logger.debug("Initializing LinkedInJobManager") self.driver = driver self.set_old_answers = set() self.easy_applier_component = None + logger.debug("LinkedInJobManager initialized successfully") def set_parameters(self, parameters): + logger.debug("Setting parameters for LinkedInJobManager") self.company_blacklist = parameters.get('companyBlacklist', []) or [] self.title_blacklist = parameters.get('titleBlacklist', []) or [] self.positions = parameters.get('positions', []) @@ -40,22 +53,23 @@ class LinkedInJobManager: self.base_search_url = self.get_base_search_url(parameters) self.seen_jobs = [] resume_path = parameters.get('uploads', {}).get('resume', None) - if resume_path is not None and Path(resume_path).exists(): - self.resume_path = Path(resume_path) - else: - self.resume_path = None + self.resume_path = Path(resume_path) if resume_path and Path(resume_path).exists() else None self.output_file_directory = Path(parameters['outputFileDirectory']) self.env_config = EnvironmentKeys() - #self.old_question() + logger.debug("Parameters set successfully") def set_gpt_answerer(self, gpt_answerer): + logger.debug("Setting GPT answerer") self.gpt_answerer = gpt_answerer def set_resume_generator_manager(self, resume_generator_manager): + logger.debug("Setting resume generator manager") self.resume_generator_manager = resume_generator_manager def start_applying(self): - self.easy_applier_component = LinkedInEasyApplier(self.driver, self.resume_path, self.set_old_answers, self.gpt_answerer, self.resume_generator_manager) + logger.debug("Starting job application process") + self.easy_applier_component = LinkedInEasyApplier(self.driver, self.resume_path, self.set_old_answers, + self.gpt_answerer, self.resume_generator_manager) searches = list(product(self.positions, self.locations)) random.shuffle(searches) page_sleep = 0 @@ -75,51 +89,113 @@ class LinkedInJobManager: self.next_job_page(position, location_url, job_page_number) time.sleep(random.uniform(1.5, 3.5)) utils.printyellow("Starting the application process for this page...") - self.apply_jobs() + + try: + jobs = self.get_jobs_from_page() + if not jobs: + utils.printyellow("No more jobs found on this page. Exiting loop.") + break + except Exception as e: + logger.error(f"Failed to retrieve jobs: {e}") + break + + try: + self.apply_jobs() + except Exception as e: + logger.error("Error during job application: %s", e) + utils.printred(f"Error during job application: {e}") + continue + utils.printyellow("Applying to jobs on this page has been completed!") time_left = minimum_page_time - time.time() if time_left > 0: utils.printyellow(f"Sleeping for {time_left} seconds.") + logger.debug("Sleeping for %d seconds", time_left) time.sleep(time_left) minimum_page_time = time.time() + minimum_time if page_sleep % 5 == 0: sleep_time = random.randint(5, 34) utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") + logger.debug("Sleeping for %d seconds", sleep_time) time.sleep(sleep_time) page_sleep += 1 - except Exception: - traceback.format_exc() - pass + except Exception as e: + logger.error("Unexpected error during job search: %s", e) + utils.printred(f"Unexpected error: {e}") + continue time_left = minimum_page_time - time.time() if time_left > 0: utils.printyellow(f"Sleeping for {time_left} seconds.") + logger.debug("Sleeping for %d seconds", time_left) time.sleep(time_left) minimum_page_time = time.time() + minimum_time if page_sleep % 5 == 0: sleep_time = random.randint(50, 90) utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") + logger.debug("Sleeping for %d seconds", sleep_time) time.sleep(sleep_time) page_sleep += 1 + def get_jobs_from_page(self): + + try: + + no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand') + if 'No matching jobs found' in no_jobs_element.text or 'unfortunately, things aren' in self.driver.page_source.lower(): + utils.printyellow("No matching jobs found on this page.") + logger.debug("No matching jobs found on this page, skipping.") + return [] + + except NoSuchElementException: + pass + + try: + job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list") + utils.scroll_slow(self.driver, job_results) + utils.scroll_slow(self.driver, job_results, step=300, reverse=True) + + job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[ + 0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item') + if not job_list_elements: + utils.printyellow("No job class elements found on page.") + logger.debug("No job class elements found on page, skipping.") + return [] + + return job_list_elements + + except NoSuchElementException: + logger.debug("No job results found on the page.") + return [] + + except Exception as e: + logger.error(f"Error while fetching job elements: {e}") + return [] + def apply_jobs(self): try: no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand') if 'No matching jobs found' in no_jobs_element.text or 'unfortunately, things aren' in self.driver.page_source.lower(): - raise Exception("No more jobs on this page") + utils.printyellow("No matching jobs found on this page, moving to next.") + logger.debug("No matching jobs found on this page, skipping") + return except NoSuchElementException: pass - + job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list") utils.scroll_slow(self.driver, job_results) utils.scroll_slow(self.driver, job_results, step=300, reverse=True) - job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item') + job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[ + 0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item') if not job_list_elements: - raise Exception("No job class elements found on page") - job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements] + utils.printyellow("No job class elements found on page, moving to next page.") + logger.debug("No job class elements found on page, skipping") + return + job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements] for job in job_list: if self.is_blacklisted(job.title, job.company, job.link): utils.printyellow(f"Blacklisted {job.title} at {job.company}, skipping...") + logger.debug("Job blacklisted: %s at %s", job.title, job.company) self.write_to_file(job, "skipped") continue if self.is_already_applied_to_job(job.title, job.company, job.link): @@ -132,12 +208,15 @@ class LinkedInJobManager: if job.apply_method not in {"Continue", "Applied", "Apply"}: self.easy_applier_component.job_apply(job) self.write_to_file(job, "success") + logger.debug("Applied to job: %s at %s", job.title, job.company) except Exception as e: - utils.printred(traceback.format_exc()) + logger.error("Failed to apply for %s at %s: %s", job.title, job.company, e) + utils.printred(f"Failed to apply for {job.title} at {job.company}: {e}") self.write_to_file(job, "failed") continue - + def write_to_file(self, job, file_name): + logger.debug("Writing job application result to file: %s", file_name) pdf_path = Path(job.pdf_path).resolve() pdf_path = pdf_path.as_uri() data = { @@ -152,22 +231,27 @@ class LinkedInJobManager: if not file_path.exists(): with open(file_path, 'w', encoding='utf-8') as f: json.dump([data], f, indent=4) + logger.debug("Job data written to new file: %s", file_path) else: with open(file_path, 'r+', encoding='utf-8') as f: try: existing_data = json.load(f) except json.JSONDecodeError: + logger.error("JSON decode error in file: %s", file_path) existing_data = [] existing_data.append(data) f.seek(0) json.dump(existing_data, f, indent=4) f.truncate() + logger.debug("Job data appended to existing file: %s", file_path) def get_base_search_url(self, parameters): + logger.debug("Constructing base search URL") url_parts = [] if parameters['remote']: url_parts.append("f_CF=f_WRA") - experience_levels = [str(i+1) for i, (level, v) in enumerate(parameters.get('experienceLevel', {}).items()) if v] + experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experienceLevel', {}).items()) if + v] if experience_levels: url_parts.append(f"f_E={','.join(experience_levels)}") url_parts.append(f"distance={parameters['distance']}") @@ -183,35 +267,51 @@ class LinkedInJobManager: date_param = next((v for k, v in date_mapping.items() if parameters.get('date', {}).get(k)), "") url_parts.append("f_LF=f_AL") # Easy Apply base_url = "&".join(url_parts) - return f"?{base_url}{date_param}" - + full_url = f"?{base_url}{date_param}" + logger.debug("Base search URL constructed: %s", full_url) + return full_url + def next_job_page(self, position, location, job_page): - self.driver.get(f"https://www.linkedin.com/jobs/search/{self.base_search_url}&keywords={position}{location}&start={job_page * 25}") - + logger.debug("Navigating to next job page: %s in %s, page %d", position, location, job_page) + self.driver.get( + f"https://www.linkedin.com/jobs/search/{self.base_search_url}&keywords={position}{location}&start={job_page * 25}") + def extract_job_information_from_tile(self, job_tile): + logger.debug("Extracting job information from tile") job_title, company, job_location, apply_method, link = "", "", "", "", "" try: job_title = job_tile.find_element(By.CLASS_NAME, 'job-card-list__title').text link = job_tile.find_element(By.CLASS_NAME, 'job-card-list__title').get_attribute('href').split('?')[0] company = job_tile.find_element(By.CLASS_NAME, 'job-card-container__primary-description').text - except: - pass + logger.debug("Job information extracted: %s at %s", job_title, company) + except NoSuchElementException: + utils.printyellow("Some job information (title, link, or company) is missing.") + logger.warning("Some job information (title, link, or company) is missing.") try: job_location = job_tile.find_element(By.CLASS_NAME, 'job-card-container__metadata-item').text - except: - pass + except NoSuchElementException: + utils.printyellow("Job location is missing.") + logger.warning("Job location is missing.") try: apply_method = job_tile.find_element(By.CLASS_NAME, 'job-card-container__apply-method').text - except: + except NoSuchElementException: apply_method = "Applied" + utils.printyellow("Apply method not found, assuming 'Applied'.") + logger.warning("Apply method not found, assuming 'Applied'.") return job_title, company, job_location, link, apply_method - + def is_blacklisted(self, job_title, company, link): + logger.debug("Checking if job is blacklisted: %s at %s", job_title, company) job_title_words = job_title.lower().split(' ') title_blacklisted = any(word in job_title_words for word in self.title_blacklist) company_blacklisted = company.strip().lower() in (word.strip().lower() for word in self.company_blacklist) link_seen = link in self.seen_jobs + + is_blacklisted = title_blacklisted or company_blacklisted or link_seen + logger.debug("Job blacklisted status: %s", is_blacklisted) + return is_blacklisted + return title_blacklisted or company_blacklisted or link_seen def is_already_applied_to_job(self, job_title, company, link): @@ -237,4 +337,5 @@ class LinkedInJobManager: return True except json.JSONDecodeError: continue - return False \ No newline at end of file + return False + diff --git a/src/strings.py b/src/strings.py index f54abc1..16cb84e 100644 --- a/src/strings.py +++ b/src/strings.py @@ -181,7 +181,7 @@ Answer the following question based on the provided language skills. - Answer questions directly. - If it seems likely that you have the experience, even if not explicitly defined, answer as if you have the experience. - If unsure, respond with "I have no experience with that, but I learn fast" or "Not yet, but willing to learn." -- Keep the answer under 140 characters. +- Keep the answer under 140 characters. Do not add any additional languages what is not in my experience ## Example My resume: Fluent in Italian and English. @@ -238,7 +238,6 @@ This comprehensive overview will serve as a guideline for the recruitment proces # Job Description Summary""" - coverletter_template = """ Compose a brief and impactful cover letter based on the provided job description and resume. The letter should be no longer than three paragraphs and should be written in a professional, yet conversational tone. Avoid using any placeholders, and ensure that the letter flows naturally and is tailored to the job. @@ -371,7 +370,6 @@ Options: [1-2, 3-5, 6-10, 10+] ## """ - try_to_fix_template = """\ The objective is to fix the text of a form input on a web page. diff --git a/src/utils.py b/src/utils.py index ea7c07b..44d022f 100644 --- a/src/utils.py +++ b/src/utils.py @@ -1,103 +1,171 @@ +import logging import os import random import time from selenium import webdriver +log_file = "app_log.log" + +logging.basicConfig( + level=logging.DEBUG, + format='%(asctime)s - %(name)s - %(levelname)s - %(message)s', + handlers=[ + logging.FileHandler(log_file, mode='a', encoding='utf-8'), + logging.StreamHandler() + ], + force=True # This will reset the root logger's handlers and apply the new configuration +) + +logger = logging.getLogger(__name__) +file_handler = logging.FileHandler(log_file, mode='a', encoding='utf-8') +formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s') +file_handler.setFormatter(formatter) +logger.addHandler(file_handler) +logger.setLevel(logging.DEBUG) + chromeProfilePath = os.path.join(os.getcwd(), "chrome_profile", "linkedin_profile") + def ensure_chrome_profile(): + logger.debug("Ensuring Chrome profile exists at path: %s", chromeProfilePath) profile_dir = os.path.dirname(chromeProfilePath) if not os.path.exists(profile_dir): os.makedirs(profile_dir) + logger.debug("Created directory for Chrome profile: %s", profile_dir) if not os.path.exists(chromeProfilePath): os.makedirs(chromeProfilePath) + logger.debug("Created Chrome profile directory: %s", chromeProfilePath) return chromeProfilePath + def is_scrollable(element): scroll_height = element.get_attribute("scrollHeight") client_height = element.get_attribute("clientHeight") - return int(scroll_height) > int(client_height) + scrollable = int(scroll_height) > int(client_height) + logger.debug("Element scrollable check: scrollHeight=%s, clientHeight=%s, scrollable=%s", scroll_height, + client_height, scrollable) + return scrollable + + +def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse=False): + logger.debug("Starting slow scroll: start=%d, end=%d, step=%d, reverse=%s", start, end, step, reverse) -def scroll_slow(driver, scrollable_element, start=0, end=3600, step=100, reverse=False): if reverse: start, end = end, start step = -step + if step == 0: + logger.error("Step value cannot be zero.") raise ValueError("Step cannot be zero.") + + max_scroll_height = int(scrollable_element.get_attribute("scrollHeight")) + current_scroll_position = int(scrollable_element.get_attribute("scrollTop")) + logger.debug("Max scroll height of the element: %d", max_scroll_height) + logger.debug("Current scroll position: %d", current_scroll_position) + + if reverse: + + if current_scroll_position < start: + start = current_scroll_position + logger.debug("Adjusted start position for upward scroll: %d", start) + else: + + if end > max_scroll_height: + logger.warning("End value exceeds the scroll height. Adjusting end to %d", max_scroll_height) + end = max_scroll_height + script_scroll_to = "arguments[0].scrollTop = arguments[1];" + try: if scrollable_element.is_displayed(): if not is_scrollable(scrollable_element): + logger.warning("The element is not scrollable.") print("The element is not scrollable.") return + if (step > 0 and start >= end) or (step < 0 and start <= end): + logger.warning("No scrolling will occur due to incorrect start/end values.") print("No scrolling will occur due to incorrect start/end values.") - return - for position in range(start, end, step): + return + + position = start + while (step > 0 and position < end) or (step < 0 and position > end): try: driver.execute_script(script_scroll_to, scrollable_element, position) + logger.debug("Scrolled to position: %d", position) except Exception as e: + logger.error("Error during scrolling: %s", e) print(f"Error during scrolling: {e}") - time.sleep(random.uniform(1.0, 2.6)) + + position += step + step = max(10, abs(step) - 10) * (-1 if reverse else 1) + + time.sleep(random.uniform(0.6, 1.5)) + driver.execute_script(script_scroll_to, scrollable_element, end) - time.sleep(1) + logger.debug("Scrolled to final position: %d", end) + time.sleep(0.5) else: + logger.warning("The element is not visible.") print("The element is not visible.") except Exception as e: + logger.error("Exception occurred during scrolling: %s", e) print(f"Exception occurred: {e}") -def chromeBrowserOptions(): + +def chrome_browser_options(): + logger.debug("Setting Chrome browser options") ensure_chrome_profile() options = webdriver.ChromeOptions() - options.add_argument("--start-maximized") # Avvia il browser a schermo intero - options.add_argument("--no-sandbox") # Disabilita la sandboxing per migliorare le prestazioni - options.add_argument("--disable-dev-shm-usage") # Utilizza una directory temporanea per la memoria condivisa - options.add_argument("--ignore-certificate-errors") # Ignora gli errori dei certificati SSL - options.add_argument("--disable-extensions") # Disabilita le estensioni del browser - options.add_argument("--disable-gpu") # Disabilita l'accelerazione GPU - options.add_argument("window-size=1200x800") # Imposta la dimensione della finestra del browser - options.add_argument("--disable-background-timer-throttling") # Disabilita il throttling dei timer in background - options.add_argument("--disable-backgrounding-occluded-windows") # Disabilita la sospensione delle finestre occluse - options.add_argument("--disable-translate") # Disabilita il traduttore automatico - options.add_argument("--disable-popup-blocking") # Disabilita il blocco dei popup - options.add_argument("--no-first-run") # Disabilita la configurazione iniziale del browser - options.add_argument("--no-default-browser-check") # Disabilita il controllo del browser predefinito - options.add_argument("--disable-logging") # Disabilita il logging - options.add_argument("--disable-autofill") # Disabilita l'autocompletamento dei moduli - options.add_argument("--disable-plugins") # Disabilita i plugin del browser - options.add_argument("--disable-animations") # Disabilita le animazioni - options.add_argument("--disable-cache") # Disabilita la cache - options.add_experimental_option("excludeSwitches", ["enable-automation", "enable-logging"]) # Esclude switch della modalità automatica e logging + options.add_argument("--start-maximized") + options.add_argument("--no-sandbox") + options.add_argument("--disable-dev-shm-usage") + options.add_argument("--ignore-certificate-errors") + options.add_argument("--disable-extensions") + options.add_argument("--disable-gpu") + options.add_argument("window-size=1200x800") + options.add_argument("--disable-background-timer-throttling") + options.add_argument("--disable-backgrounding-occluded-windows") + options.add_argument("--disable-translate") + options.add_argument("--disable-popup-blocking") + options.add_argument("--no-first-run") + options.add_argument("--no-default-browser-check") + options.add_argument("--disable-logging") + options.add_argument("--disable-autofill") + options.add_argument("--disable-plugins") + options.add_argument("--disable-animations") + options.add_argument("--disable-cache") + options.add_experimental_option("excludeSwitches", ["enable-automation", "enable-logging"]) - # Preferenze per contenuti prefs = { - "profile.default_content_setting_values.images": 2, # Disabilita il caricamento delle immagini - "profile.managed_default_content_settings.stylesheets": 2, # Disabilita il caricamento dei fogli di stile + "profile.default_content_setting_values.images": 2, + "profile.managed_default_content_settings.stylesheets": 2, } options.add_experimental_option("prefs", prefs) if len(chromeProfilePath) > 0: - initialPath = os.path.dirname(chromeProfilePath) - profileDir = os.path.basename(chromeProfilePath) - options.add_argument('--user-data-dir=' + initialPath) - options.add_argument("--profile-directory=" + profileDir) + initial_path = os.path.dirname(chromeProfilePath) + profile_dir = os.path.basename(chromeProfilePath) + options.add_argument('--user-data-dir=' + initial_path) + options.add_argument("--profile-directory=" + profile_dir) + logger.debug("Using Chrome profile directory: %s", chromeProfilePath) else: options.add_argument("--incognito") + logger.debug("Using Chrome in incognito mode") return options def printred(text): - # Codice colore ANSI per il rosso - RED = "\033[91m" - RESET = "\033[0m" - # Stampa il testo in rosso - print(f"{RED}{text}{RESET}") + red = "\033[91m" + reset = "\033[0m" + logger.debug("Printing text in red: %s", text) + print(f"{red}{text}{reset}") + def printyellow(text): - # Codice colore ANSI per il giallo - YELLOW = "\033[93m" - RESET = "\033[0m" - # Stampa il testo in giallo - print(f"{YELLOW}{text}{RESET}") \ No newline at end of file + yellow = "\033[93m" + reset = "\033[0m" + logger.debug("Printing text in yellow: %s", text) + print(f"{yellow}{text}{reset}")