Merge pull request #295 from queukat/v3

add logs, fix finding easy apply button, fix prompts for achievements
This commit is contained in:
Federico 2024-09-08 00:05:23 +02:00 committed by GitHub
commit ad1b17c031
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 1175 additions and 291 deletions

View file

@ -2,19 +2,24 @@ import json
import os import os
import re import re
import textwrap import textwrap
import time
from datetime import datetime from datetime import datetime
from abc import ABC, abstractmethod from abc import ABC, abstractmethod
from typing import Dict, List, Union from typing import Dict, List, Union
from pathlib import Path from pathlib import Path
from typing import Dict, List
import httpx
from Levenshtein import distance
from dotenv import load_dotenv from dotenv import load_dotenv
from langchain_core.messages.ai import AIMessage from langchain_core.messages.ai import AIMessage
from langchain_core.output_parsers import StrOutputParser from langchain_core.output_parsers import StrOutputParser
from langchain_core.prompt_values import StringPromptValue from langchain_core.prompt_values import StringPromptValue
from langchain_core.prompts import ChatPromptTemplate from langchain_core.prompts import ChatPromptTemplate
from langchain_openai import ChatOpenAI from langchain_openai import ChatOpenAI
from Levenshtein import distance
import src.strings as strings import src.strings as strings
from src.utils import logger
load_dotenv() load_dotenv()
@ -76,150 +81,272 @@ class AIAdapter:
return self.model.invoke(prompt) return self.model.invoke(prompt)
class LLMLogger: class LLMLogger:
def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]): def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]):
self.llm = llm self.llm = llm
logger.debug("LLMLogger successfully initialized with LLM: %s", llm)
@staticmethod @staticmethod
def log_request(prompts, parsed_reply: Dict[str, Dict]): def log_request(prompts, parsed_reply: Dict[str, Dict]):
calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json") logger.debug("Starting log_request method")
logger.debug("Prompts received: %s", prompts)
logger.debug("Parsed reply received: %s", parsed_reply)
try:
calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json")
logger.debug("Logging path determined: %s", calls_log)
except Exception as e:
logger.error("Error determining the log path: %s", str(e))
raise
if isinstance(prompts, StringPromptValue): if isinstance(prompts, StringPromptValue):
logger.debug("Prompts are of type StringPromptValue")
prompts = prompts.text prompts = prompts.text
logger.debug("Prompts converted to text: %s", prompts)
elif isinstance(prompts, Dict): elif isinstance(prompts, Dict):
# Convert prompts to a dictionary if they are not in the expected format logger.debug("Prompts are of type Dict")
prompts = { try:
f"prompt_{i+1}": prompt.content prompts = {
for i, prompt in enumerate(prompts.messages) f"prompt_{i + 1}": prompt.content
} for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts converted to dictionary: %s", prompts)
except Exception as e:
logger.error("Error converting prompts to dictionary: %s", str(e))
raise
else: else:
prompts = { logger.debug("Prompts are of unknown type, attempting default conversion")
f"prompt_{i+1}": prompt.content try:
for i, prompt in enumerate(prompts.messages) prompts = {
f"prompt_{i + 1}": prompt.content
for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts converted to dictionary using default method: %s", prompts)
except Exception as e:
logger.error("Error converting prompts using default method: %s", str(e))
raise
try:
current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
logger.debug("Current time obtained: %s", current_time)
except Exception as e:
logger.error("Error obtaining current time: %s", str(e))
raise
try:
token_usage = parsed_reply["usage_metadata"]
output_tokens = token_usage["output_tokens"]
input_tokens = token_usage["input_tokens"]
total_tokens = token_usage["total_tokens"]
logger.debug("Token usage - Input: %d, Output: %d, Total: %d", input_tokens, output_tokens, total_tokens)
except KeyError as e:
logger.error("KeyError in parsed_reply structure: %s", str(e))
raise
try:
model_name = parsed_reply["response_metadata"]["model_name"]
logger.debug("Model name: %s", model_name)
except KeyError as e:
logger.error("KeyError in response_metadata: %s", str(e))
raise
try:
prompt_price_per_token = 0.00000015
completion_price_per_token = 0.0000006
total_cost = (input_tokens * prompt_price_per_token) + (output_tokens * completion_price_per_token)
logger.debug("Total cost calculated: %f", total_cost)
except Exception as e:
logger.error("Error calculating total cost: %s", str(e))
raise
try:
log_entry = {
"model": model_name,
"time": current_time,
"prompts": prompts,
"replies": parsed_reply["content"],
"total_tokens": total_tokens,
"input_tokens": input_tokens,
"output_tokens": output_tokens,
"total_cost": total_cost,
} }
logger.debug("Log entry created: %s", log_entry)
except KeyError as e:
logger.error("Error creating log entry: missing key %s in parsed_reply", str(e))
raise
current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S") try:
with open(calls_log, "a", encoding="utf-8") as f:
# Extract token usage details from the response json_string = json.dumps(log_entry, ensure_ascii=False, indent=4)
token_usage = parsed_reply["usage_metadata"] f.write(json_string + "\n")
output_tokens = token_usage["output_tokens"] logger.debug("Log entry written to file: %s", calls_log)
input_tokens = token_usage["input_tokens"] except Exception as e:
total_tokens = token_usage["total_tokens"] logger.error("Error writing log entry to file: %s", str(e))
raise
# Extract model details from the response
model_name = parsed_reply["response_metadata"]["model_name"]
prompt_price_per_token = 0.00000015
completion_price_per_token = 0.0000006
# Calculate the total cost of the API call
total_cost = (input_tokens * prompt_price_per_token) + (
output_tokens * completion_price_per_token
)
# Create a log entry with all relevant information
log_entry = {
"model": model_name,
"time": current_time,
"prompts": prompts,
"replies": parsed_reply["content"], # Response content
"total_tokens": total_tokens,
"input_tokens": input_tokens,
"output_tokens": output_tokens,
"total_cost": total_cost,
}
# Write the log entry to the log file in JSON format
with open(calls_log, "a", encoding="utf-8") as f:
json_string = json.dumps(log_entry, ensure_ascii=False, indent=4)
f.write(json_string + "\n")
class LoggerChatModel: class LoggerChatModel:
def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]): def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]):
self.llm = llm self.llm = llm
logger.debug("LoggerChatModel successfully initialized with LLM: %s", llm)
def __call__(self, messages: List[Dict[str, str]]) -> str: def __call__(self, messages: List[Dict[str, str]]) -> str:
# Call the LLM with the provided messages and log the response.
reply = self.llm.invoke(messages) logger.debug("Entering __call__ method with messages: %s", messages)
parsed_reply = self.parse_llmresult(reply) while True:
LLMLogger.log_request(prompts=messages, parsed_reply=parsed_reply) try:
return reply logger.debug("Attempting to call the LLM with messages")
reply = self.llm(messages)
logger.debug("LLM response received: %s", reply)
parsed_reply = self.parse_llmresult(reply)
logger.debug("Parsed LLM reply: %s", parsed_reply)
LLMLogger.log_request(prompts=messages, parsed_reply=parsed_reply)
logger.debug("Request successfully logged")
return reply
except httpx.HTTPStatusError as e:
logger.error("HTTPStatusError encountered: %s", str(e))
if e.response.status_code == 429:
retry_after = e.response.headers.get('retry-after')
retry_after_ms = e.response.headers.get('retry-after-ms')
if retry_after:
wait_time = int(retry_after)
logger.warning(
"Rate limit exceeded. Waiting for %d seconds before retrying (extracted from 'retry-after' header)...",
wait_time)
time.sleep(wait_time)
elif retry_after_ms:
wait_time = int(retry_after_ms) / 1000.0
logger.warning(
"Rate limit exceeded. Waiting for %f seconds before retrying (extracted from 'retry-after-ms' header)...",
wait_time)
time.sleep(wait_time)
else:
wait_time = 30
logger.warning(
"'retry-after' header not found. Waiting for %d seconds before retrying (default)...",
wait_time)
time.sleep(wait_time)
else:
logger.error("HTTP error occurred with status code: %d, waiting 30 seconds before retrying",
e.response.status_code)
time.sleep(30)
except Exception as e:
logger.error("Unexpected error occurred: %s", str(e))
logger.info("Waiting for 30 seconds before retrying due to an unexpected error.")
time.sleep(30)
continue
def parse_llmresult(self, llmresult: AIMessage) -> Dict[str, Dict]: def parse_llmresult(self, llmresult: AIMessage) -> Dict[str, Dict]:
# Parse the LLM result into a structured format. logger.debug("Parsing LLM result: %s", llmresult)
content = llmresult.content
response_metadata = llmresult.response_metadata try:
id_ = llmresult.id content = llmresult.content
usage_metadata = llmresult.usage_metadata response_metadata = llmresult.response_metadata
parsed_result = { id_ = llmresult.id
"content": content, usage_metadata = llmresult.usage_metadata
"response_metadata": {
"model_name": response_metadata.get("model_name", ""), parsed_result = {
"system_fingerprint": response_metadata.get("system_fingerprint", ""), "content": content,
"finish_reason": response_metadata.get("finish_reason", ""), "response_metadata": {
"logprobs": response_metadata.get("logprobs", None), "model_name": response_metadata.get("model_name", ""),
}, "system_fingerprint": response_metadata.get("system_fingerprint", ""),
"id": id_, "finish_reason": response_metadata.get("finish_reason", ""),
"usage_metadata": { "logprobs": response_metadata.get("logprobs", None),
"input_tokens": usage_metadata.get("input_tokens", 0), },
"output_tokens": usage_metadata.get("output_tokens", 0), "id": id_,
"total_tokens": usage_metadata.get("total_tokens", 0), "usage_metadata": {
}, "input_tokens": usage_metadata.get("input_tokens", 0),
} "output_tokens": usage_metadata.get("output_tokens", 0),
return parsed_result "total_tokens": usage_metadata.get("total_tokens", 0),
},
}
logger.debug("Parsed LLM result successfully: %s", parsed_result)
return parsed_result
except KeyError as e:
logger.error("KeyError while parsing LLM result: missing key %s", str(e))
raise
except Exception as e:
logger.error("Unexpected error while parsing LLM result: %s", str(e))
raise
class GPTAnswerer: class GPTAnswerer:
def __init__(self, config, llm_api_key): def __init__(self, config, llm_api_key):
self.ai_adapter = AIAdapter(config, llm_api_key) self.ai_adapter = AIAdapter(config, llm_api_key)
self.llm_cheap = LoggerChatModel(self.ai_adapter) self.llm_cheap = LoggerChatModel(self.ai_adapter)
@property @property
def job_description(self): def job_description(self):
return self.job.description return self.job.description
@staticmethod @staticmethod
def find_best_match(text: str, options: list[str]) -> str: def find_best_match(text: str, options: list[str]) -> str:
logger.debug("Finding best match for text: '%s' in options: %s", text, options)
distances = [ distances = [
(option, distance(text.lower(), option.lower())) for option in options (option, distance(text.lower(), option.lower())) for option in options
] ]
best_option = min(distances, key=lambda x: x[1])[0] best_option = min(distances, key=lambda x: x[1])[0]
logger.debug("Best match found: %s", best_option)
return best_option return best_option
@staticmethod @staticmethod
def _remove_placeholders(text: str) -> str: def _remove_placeholders(text: str) -> str:
logger.debug("Removing placeholders from text: %s", text)
text = text.replace("PLACEHOLDER", "") text = text.replace("PLACEHOLDER", "")
return text.strip() return text.strip()
@staticmethod @staticmethod
def _preprocess_template_string(template: str) -> str: def _preprocess_template_string(template: str) -> str:
# Preprocess a template string to remove unnecessary indentation. logger.debug("Preprocessing template string")
return textwrap.dedent(template) return textwrap.dedent(template)
def set_resume(self, resume): def set_resume(self, resume):
logger.debug("Setting resume: %s", resume)
self.resume = resume self.resume = resume
def set_job(self, job): def set_job(self, job):
logger.debug("Setting job: %s", job)
self.job = job self.job = job
self.job.set_summarize_job_description(self.summarize_job_description(self.job.description)) self.job.set_summarize_job_description(self.summarize_job_description(self.job.description))
def set_job_application_profile(self, job_application_profile): def set_job_application_profile(self, job_application_profile):
logger.debug("Setting job application profile: %s", job_application_profile)
self.job_application_profile = job_application_profile self.job_application_profile = job_application_profile
def summarize_job_description(self, text: str) -> str: def summarize_job_description(self, text: str) -> str:
logger.debug("Summarizing job description: %s", text)
strings.summarize_prompt_template = self._preprocess_template_string( strings.summarize_prompt_template = self._preprocess_template_string(
strings.summarize_prompt_template strings.summarize_prompt_template
) )
prompt = ChatPromptTemplate.from_template(strings.summarize_prompt_template) prompt = ChatPromptTemplate.from_template(strings.summarize_prompt_template)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output = chain.invoke({"text": text}) output = chain.invoke({"text": text})
logger.debug("Summary generated: %s", output)
return output return output
def _create_chain(self, template: str): def _create_chain(self, template: str):
logger.debug("Creating chain with template: %s", template)
prompt = ChatPromptTemplate.from_template(template) prompt = ChatPromptTemplate.from_template(template)
return prompt | self.llm_cheap | StrOutputParser() return prompt | self.llm_cheap | StrOutputParser()
def answer_question_textual_wide_range(self, question: str) -> str: def answer_question_textual_wide_range(self, question: str) -> str:
# Define chains for each section of the resume logger.debug("Answering textual question: %s", question)
chains = { chains = {
"personal_information": self._create_chain(strings.personal_information_template), "personal_information": self._create_chain(strings.personal_information_template),
"self_identification": self._create_chain(strings.self_identification_template), "self_identification": self._create_chain(strings.self_identification_template),
@ -326,59 +453,83 @@ class GPTAnswerer:
prompt = ChatPromptTemplate.from_template(section_prompt) prompt = ChatPromptTemplate.from_template(section_prompt)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output = chain.invoke({"question": question}) output = chain.invoke({"question": question})
match = re.search(r"(Personal information|Self Identification|Legal Authorization|Work Preferences|Education Details|Experience Details|Projects|Availability|Salary Expectations|Certifications|Languages|Interests|Cover letter)", output, re.IGNORECASE) match = re.search(r"(Personal information|Self Identification|Legal Authorization|Work Preferences|Education Details|Experience Details|Projects|Availability|Salary Expectations|Certifications|Languages|Interests|Cover letter)", output, re.IGNORECASE)
if not match: if not match:
raise ValueError("Could not extract section name from the response.") raise ValueError("Could not extract section name from the response.")
section_name = match.group(1).lower().replace(" ", "_") section_name = match.group(1).lower().replace(" ", "_")
if section_name == "cover_letter": if section_name == "cover_letter":
chain = chains.get(section_name) chain = chains.get(section_name)
output = chain.invoke({"resume": self.resume, "job_description": self.job_description}) output = chain.invoke({"resume": self.resume, "job_description": self.job_description})
logger.debug("Cover letter generated: %s", output)
return output return output
resume_section = getattr(self.resume, section_name, None) or getattr(self.job_application_profile, section_name, None) resume_section = getattr(self.resume, section_name, None) or getattr(self.job_application_profile, section_name,
None)
if resume_section is None: if resume_section is None:
logger.error("Section '%s' not found in either resume or job_application_profile.", section_name)
raise ValueError(f"Section '{section_name}' not found in either resume or job_application_profile.") raise ValueError(f"Section '{section_name}' not found in either resume or job_application_profile.")
chain = chains.get(section_name) chain = chains.get(section_name)
if chain is None: if chain is None:
logger.error("Chain not defined for section '%s'", section_name)
raise ValueError(f"Chain not defined for section '{section_name}'") raise ValueError(f"Chain not defined for section '{section_name}'")
return chain.invoke({"resume_section": resume_section, "question": question}) output = chain.invoke({"resume_section": resume_section, "question": question})
logger.debug("Question answered: %s", output)
return output
def answer_question_numeric(self, question: str, default_experience: int = 3) -> int: def answer_question_numeric(self, question: str, default_experience: int = 3) -> int:
logger.debug("Answering numeric question: %s", question)
func_template = self._preprocess_template_string(strings.numeric_question_template) func_template = self._preprocess_template_string(strings.numeric_question_template)
prompt = ChatPromptTemplate.from_template(func_template) prompt = ChatPromptTemplate.from_template(func_template)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output_str = chain.invoke({"resume_educations": self.resume.education_details,"resume_jobs": self.resume.experience_details,"resume_projects": self.resume.projects , "question": question}) output_str = chain.invoke(
{"resume_educations": self.resume.education_details, "resume_jobs": self.resume.experience_details,
"resume_projects": self.resume.projects, "question": question})
logger.debug("Raw output for numeric question: %s", output_str)
try: try:
output = self.extract_number_from_string(output_str) output = self.extract_number_from_string(output_str)
logger.debug("Extracted number: %d", output)
except ValueError: except ValueError:
logger.warning("Failed to extract number, using default experience: %d", default_experience)
output = default_experience output = default_experience
return output return output
def extract_number_from_string(self, output_str): def extract_number_from_string(self, output_str):
logger.debug("Extracting number from string: %s", output_str)
numbers = re.findall(r"\d+", output_str) numbers = re.findall(r"\d+", output_str)
if numbers: if numbers:
logger.debug("Numbers found: %s", numbers)
return int(numbers[0]) return int(numbers[0])
else: else:
logger.error("No numbers found in the string")
raise ValueError("No numbers found in the string") raise ValueError("No numbers found in the string")
def answer_question_from_options(self, question: str, options: list[str]) -> str: def answer_question_from_options(self, question: str, options: list[str]) -> str:
logger.debug("Answering question from options: %s", question)
func_template = self._preprocess_template_string(strings.options_template) func_template = self._preprocess_template_string(strings.options_template)
prompt = ChatPromptTemplate.from_template(func_template) prompt = ChatPromptTemplate.from_template(func_template)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output_str = chain.invoke({"resume": self.resume, "question": question, "options": options}) output_str = chain.invoke({"resume": self.resume, "question": question, "options": options})
logger.debug("Raw output for options question: %s", output_str)
best_option = self.find_best_match(output_str, options) best_option = self.find_best_match(output_str, options)
logger.debug("Best option determined: %s", best_option)
return best_option return best_option
def resume_or_cover(self, phrase: str) -> str: def resume_or_cover(self, phrase: str) -> str:
# Define the prompt template logger.debug("Determining if phrase refers to resume or cover letter: %s", phrase)
prompt_template = """ prompt_template = """
Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. Do not provide any additional information or explanations. Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter.
If the phrase contains only one word 'upload', consider it as 'cover'.
If the phrase contains 'upload resume', consider it as 'resume'.
Do not provide any additional information or explanations.
phrase: {phrase} phrase: {phrase}
""" """
prompt = ChatPromptTemplate.from_template(prompt_template) prompt = ChatPromptTemplate.from_template(prompt_template)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
response = chain.invoke({"phrase": phrase}) response = chain.invoke({"phrase": phrase})
logger.debug("Response for resume_or_cover: %s", response)
if "resume" in response: if "resume" in response:
return "resume" return "resume"
elif "cover" in response: elif "cover" in response:

View file

@ -1,5 +1,8 @@
from dataclasses import dataclass from dataclasses import dataclass
from src.utils import logger
@dataclass @dataclass
class Job: class Job:
title: str title: str
@ -13,18 +16,22 @@ class Job:
recruiter_link: str = "" recruiter_link: str = ""
def set_summarize_job_description(self, summarize_job_description): def set_summarize_job_description(self, summarize_job_description):
logger.debug("Setting summarized job description: %s", summarize_job_description)
self.summarize_job_description = summarize_job_description self.summarize_job_description = summarize_job_description
def set_job_description(self, description): def set_job_description(self, description):
logger.debug("Setting job description: %s", description)
self.description = description self.description = description
def set_recruiter_link(self, recruiter_link): def set_recruiter_link(self, recruiter_link):
logger.debug("Setting recruiter link: %s", recruiter_link)
self.recruiter_link = recruiter_link self.recruiter_link = recruiter_link
def formatted_job_information(self): def formatted_job_information(self):
""" """
Formats the job information as a markdown string. Formats the job information as a markdown string.
""" """
logger.debug("Formatting job information for job: %s at %s", self.title, self.company)
job_information = f""" job_information = f"""
# Job Description # Job Description
## Job Information ## Job Information
@ -36,4 +43,6 @@ class Job:
## Description ## Description
{self.description or 'No description provided.'} {self.description or 'No description provided.'}
""" """
return job_information.strip() formatted_information = job_information.strip()
logger.debug("Formatted job information: %s", formatted_information)
return formatted_information

View file

@ -1,7 +1,10 @@
from dataclasses import dataclass from dataclasses import dataclass
from typing import Dict, List
import yaml import yaml
from src.utils import logger
@dataclass @dataclass
class SelfIdentification: class SelfIdentification:
gender: str gender: str
@ -10,6 +13,7 @@ class SelfIdentification:
disability: str disability: str
ethnicity: str ethnicity: str
@dataclass @dataclass
class LegalAuthorization: class LegalAuthorization:
eu_work_authorization: str eu_work_authorization: str
@ -21,6 +25,7 @@ class LegalAuthorization:
legally_allowed_to_work_in_eu: str legally_allowed_to_work_in_eu: str
requires_eu_sponsorship: str requires_eu_sponsorship: str
@dataclass @dataclass
class WorkPreferences: class WorkPreferences:
remote_work: str remote_work: str
@ -30,14 +35,17 @@ class WorkPreferences:
willing_to_undergo_drug_tests: str willing_to_undergo_drug_tests: str
willing_to_undergo_background_checks: str willing_to_undergo_background_checks: str
@dataclass @dataclass
class Availability: class Availability:
notice_period: str notice_period: str
@dataclass @dataclass
class SalaryExpectations: class SalaryExpectations:
salary_range_usd: str salary_range_usd: str
@dataclass @dataclass
class JobApplicationProfile: class JobApplicationProfile:
self_identification: SelfIdentification self_identification: SelfIdentification
@ -47,86 +55,123 @@ class JobApplicationProfile:
salary_expectations: SalaryExpectations salary_expectations: SalaryExpectations
def __init__(self, yaml_str: str): def __init__(self, yaml_str: str):
logger.debug("Initializing JobApplicationProfile with provided YAML string")
try: try:
data = yaml.safe_load(yaml_str) data = yaml.safe_load(yaml_str)
logger.debug("YAML data successfully parsed: %s", data)
except yaml.YAMLError as e: except yaml.YAMLError as e:
logger.error("Error parsing YAML file: %s", e)
raise ValueError("Error parsing YAML file.") from e raise ValueError("Error parsing YAML file.") from e
except Exception as e: except Exception as e:
logger.error("Unexpected error occurred while parsing the YAML file: %s", e)
raise RuntimeError("An unexpected error occurred while parsing the YAML file.") from e raise RuntimeError("An unexpected error occurred while parsing the YAML file.") from e
if not isinstance(data, dict): if not isinstance(data, dict):
logger.error("YAML data must be a dictionary, received: %s", type(data))
raise TypeError("YAML data must be a dictionary.") raise TypeError("YAML data must be a dictionary.")
# Process self_identification # Process self_identification
try: try:
logger.debug("Processing self_identification")
self.self_identification = SelfIdentification(**data['self_identification']) self.self_identification = SelfIdentification(**data['self_identification'])
logger.debug("self_identification processed: %s", self.self_identification)
except KeyError as e: except KeyError as e:
logger.error("Required field %s is missing in self_identification data.", e)
raise KeyError(f"Required field {e} is missing in self_identification data.") from e raise KeyError(f"Required field {e} is missing in self_identification data.") from e
except TypeError as e: except TypeError as e:
logger.error("Error in self_identification data: %s", e)
raise TypeError(f"Error in self_identification data: {e}") from e raise TypeError(f"Error in self_identification data: {e}") from e
except AttributeError as e: except AttributeError as e:
logger.error("Attribute error in self_identification processing: %s", e)
raise AttributeError("Attribute error in self_identification processing.") from e raise AttributeError("Attribute error in self_identification processing.") from e
except Exception as e: except Exception as e:
logger.error("An unexpected error occurred while processing self_identification: %s", e)
raise RuntimeError("An unexpected error occurred while processing self_identification.") from e raise RuntimeError("An unexpected error occurred while processing self_identification.") from e
# Process legal_authorization # Process legal_authorization
try: try:
logger.debug("Processing legal_authorization")
self.legal_authorization = LegalAuthorization(**data['legal_authorization']) self.legal_authorization = LegalAuthorization(**data['legal_authorization'])
logger.debug("legal_authorization processed: %s", self.legal_authorization)
except KeyError as e: except KeyError as e:
logger.error("Required field %s is missing in legal_authorization data.", e)
raise KeyError(f"Required field {e} is missing in legal_authorization data.") from e raise KeyError(f"Required field {e} is missing in legal_authorization data.") from e
except TypeError as e: except TypeError as e:
logger.error("Error in legal_authorization data: %s", e)
raise TypeError(f"Error in legal_authorization data: {e}") from e raise TypeError(f"Error in legal_authorization data: {e}") from e
except AttributeError as e: except AttributeError as e:
logger.error("Attribute error in legal_authorization processing: %s", e)
raise AttributeError("Attribute error in legal_authorization processing.") from e raise AttributeError("Attribute error in legal_authorization processing.") from e
except Exception as e: except Exception as e:
logger.error("An unexpected error occurred while processing legal_authorization: %s", e)
raise RuntimeError("An unexpected error occurred while processing legal_authorization.") from e raise RuntimeError("An unexpected error occurred while processing legal_authorization.") from e
# Process work_preferences # Process work_preferences
try: try:
logger.debug("Processing work_preferences")
self.work_preferences = WorkPreferences(**data['work_preferences']) self.work_preferences = WorkPreferences(**data['work_preferences'])
logger.debug("work_preferences processed: %s", self.work_preferences)
except KeyError as e: except KeyError as e:
logger.error("Required field %s is missing in work_preferences data.", e)
raise KeyError(f"Required field {e} is missing in work_preferences data.") from e raise KeyError(f"Required field {e} is missing in work_preferences data.") from e
except TypeError as e: except TypeError as e:
logger.error("Error in work_preferences data: %s", e)
raise TypeError(f"Error in work_preferences data: {e}") from e raise TypeError(f"Error in work_preferences data: {e}") from e
except AttributeError as e: except AttributeError as e:
logger.error("Attribute error in work_preferences processing: %s", e)
raise AttributeError("Attribute error in work_preferences processing.") from e raise AttributeError("Attribute error in work_preferences processing.") from e
except Exception as e: except Exception as e:
logger.error("An unexpected error occurred while processing work_preferences: %s", e)
raise RuntimeError("An unexpected error occurred while processing work_preferences.") from e raise RuntimeError("An unexpected error occurred while processing work_preferences.") from e
# Process availability # Process availability
try: try:
logger.debug("Processing availability")
self.availability = Availability(**data['availability']) self.availability = Availability(**data['availability'])
logger.debug("availability processed: %s", self.availability)
except KeyError as e: except KeyError as e:
logger.error("Required field %s is missing in availability data.", e)
raise KeyError(f"Required field {e} is missing in availability data.") from e raise KeyError(f"Required field {e} is missing in availability data.") from e
except TypeError as e: except TypeError as e:
logger.error("Error in availability data: %s", e)
raise TypeError(f"Error in availability data: {e}") from e raise TypeError(f"Error in availability data: {e}") from e
except AttributeError as e: except AttributeError as e:
logger.error("Attribute error in availability processing: %s", e)
raise AttributeError("Attribute error in availability processing.") from e raise AttributeError("Attribute error in availability processing.") from e
except Exception as e: except Exception as e:
logger.error("An unexpected error occurred while processing availability: %s", e)
raise RuntimeError("An unexpected error occurred while processing availability.") from e raise RuntimeError("An unexpected error occurred while processing availability.") from e
# Process salary_expectations # Process salary_expectations
try: try:
logger.debug("Processing salary_expectations")
self.salary_expectations = SalaryExpectations(**data['salary_expectations']) self.salary_expectations = SalaryExpectations(**data['salary_expectations'])
logger.debug("salary_expectations processed: %s", self.salary_expectations)
except KeyError as e: except KeyError as e:
logger.error("Required field %s is missing in salary_expectations data.", e)
raise KeyError(f"Required field {e} is missing in salary_expectations data.") from e raise KeyError(f"Required field {e} is missing in salary_expectations data.") from e
except TypeError as e: except TypeError as e:
logger.error("Error in salary_expectations data: %s", e)
raise TypeError(f"Error in salary_expectations data: {e}") from e raise TypeError(f"Error in salary_expectations data: {e}") from e
except AttributeError as e: except AttributeError as e:
logger.error("Attribute error in salary_expectations processing: %s", e)
raise AttributeError("Attribute error in salary_expectations processing.") from e raise AttributeError("Attribute error in salary_expectations processing.") from e
except Exception as e: except Exception as e:
logger.error("An unexpected error occurred while processing salary_expectations: %s", e)
raise RuntimeError("An unexpected error occurred while processing salary_expectations.") from e raise RuntimeError("An unexpected error occurred while processing salary_expectations.") from e
# Process additional fields logger.debug("JobApplicationProfile initialization completed successfully.")
def __str__(self): def __str__(self):
logger.debug("Generating string representation of JobApplicationProfile")
def format_dataclass(obj): def format_dataclass(obj):
return "\n".join(f"{field.name}: {getattr(obj, field.name)}" for field in obj.__dataclass_fields__.values()) return "\n".join(f"{field.name}: {getattr(obj, field.name)}" for field in obj.__dataclass_fields__.values())
return (f"Self Identification:\n{format_dataclass(self.self_identification)}\n\n" formatted_str = (f"Self Identification:\n{format_dataclass(self.self_identification)}\n\n"
f"Legal Authorization:\n{format_dataclass(self.legal_authorization)}\n\n" f"Legal Authorization:\n{format_dataclass(self.legal_authorization)}\n\n"
f"Work Preferences:\n{format_dataclass(self.work_preferences)}\n\n" f"Work Preferences:\n{format_dataclass(self.work_preferences)}\n\n"
f"Availability: {self.availability.notice_period}\n\n" f"Availability: {self.availability.notice_period}\n\n"
f"Salary Expectations: {self.salary_expectations.salary_range_usd}\n\n") f"Salary Expectations: {self.salary_expectations.salary_range_usd}\n\n")
logger.debug("String representation generated: %s", formatted_str)
return formatted_str

View file

@ -1,29 +1,43 @@
import random
import time import time
from selenium.common.exceptions import NoSuchElementException, TimeoutException from selenium.common.exceptions import NoSuchElementException, TimeoutException
from selenium.webdriver.common.by import By from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.ui import WebDriverWait
from src.utils import logger
class LinkedInAuthenticator: class LinkedInAuthenticator:
def __init__(self, driver=None): def __init__(self, driver=None):
self.driver = driver self.driver = driver
self.email = "" self.email = ""
self.password = "" self.password = ""
logger.debug("LinkedInAuthenticator initialized with driver: %s", driver)
def set_secrets(self, email, password): def set_secrets(self, email, password):
self.email = email self.email = email
self.password = password self.password = password
logger.debug("Secrets set with email: %s", email)
def start(self): def start(self):
print("Starting Chrome browser to log in to LinkedIn.") logger.info("Starting Chrome browser to log in to LinkedIn.")
self.driver.get('https://www.linkedin.com') self.driver.get('https://www.linkedin.com/feed')
self.wait_for_page_load() self.wait_for_page_load()
if not self.is_logged_in():
time.sleep(3)
if self.is_logged_in():
logger.info("User is already logged in. Skipping login process.")
return
else:
logger.info("User is not logged in. Proceeding with login.")
self.handle_login() self.handle_login()
def handle_login(self): def handle_login(self):
print("Navigating to the LinkedIn login page...") logger.info("Navigating to the LinkedIn login page...")
self.driver.get("https://www.linkedin.com/login") self.driver.get("https://www.linkedin.com/login")
if 'feed' in self.driver.current_url: if 'feed' in self.driver.current_url:
print("User is already logged in.") print("User is already logged in.")
@ -31,50 +45,98 @@ class LinkedInAuthenticator:
try: try:
self.enter_credentials() self.enter_credentials()
self.submit_login_form() self.submit_login_form()
except NoSuchElementException: except NoSuchElementException as e:
print("Could not log in to LinkedIn. Please check your credentials.") logger.error("Could not log in to LinkedIn. Element not found: %s", e)
time.sleep(35) #TODO fix better time.sleep(random.uniform(3, 5))
self.handle_security_check() self.handle_security_check()
def enter_credentials(self): def enter_credentials(self):
try: try:
logger.debug("Entering credentials...")
email_field = WebDriverWait(self.driver, 10).until( email_field = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.ID, "username")) EC.presence_of_element_located((By.ID, "username"))
) )
email_field.send_keys(self.email) email_field.send_keys(self.email)
logger.debug("Email entered: %s", self.email)
password_field = self.driver.find_element(By.ID, "password") password_field = self.driver.find_element(By.ID, "password")
password_field.send_keys(self.password) password_field.send_keys(self.password)
logger.debug("Password entered.")
except TimeoutException: except TimeoutException:
logger.error("Login form not found. Aborting login.")
print("Login form not found. Aborting login.") print("Login form not found. Aborting login.")
def submit_login_form(self): def submit_login_form(self):
try: try:
logger.debug("Submitting login form...")
login_button = self.driver.find_element(By.XPATH, '//button[@type="submit"]') login_button = self.driver.find_element(By.XPATH, '//button[@type="submit"]')
login_button.click() login_button.click()
logger.debug("Login form submitted.")
except NoSuchElementException: except NoSuchElementException:
logger.error("Login button not found. Please verify the page structure.")
print("Login button not found. Please verify the page structure.") print("Login button not found. Please verify the page structure.")
def handle_security_check(self): def handle_security_check(self):
try: try:
logger.debug("Handling security check...")
WebDriverWait(self.driver, 10).until( WebDriverWait(self.driver, 10).until(
EC.url_contains('https://www.linkedin.com/checkpoint/challengesV2/') EC.url_contains('https://www.linkedin.com/checkpoint/challengesV2/')
) )
logger.warning("Security checkpoint detected. Please complete the challenge.")
print("Security checkpoint detected. Please complete the challenge.") print("Security checkpoint detected. Please complete the challenge.")
WebDriverWait(self.driver, 300).until( WebDriverWait(self.driver, 300).until(
EC.url_contains('https://www.linkedin.com/feed/') EC.url_contains('https://www.linkedin.com/feed/')
) )
logger.info("Security check completed")
print("Security check completed") print("Security check completed")
except TimeoutException: except TimeoutException:
logger.error("Security check not completed within the timeout.")
print("Security check not completed. Please try again later.") print("Security check not completed. Please try again later.")
def is_logged_in(self): def is_logged_in(self):
self.driver.get('https://www.linkedin.com/') # target_url = 'https://www.linkedin.com/feed'
return self.driver.current_url == 'https://www.linkedin.com/feed/' #
# # Navigate to the target URL if not already there
# if self.driver.current_url != target_url:
# logger.debug("Navigating to target URL: %s", target_url)
# self.driver.get(target_url)
try:
# Increase the wait time for the page elements to load
logger.debug("Checking if user is logged in...")
WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.CLASS_NAME, 'share-box-feed-entry__trigger'))
)
# Check for the presence of the "Start a post" button
buttons = self.driver.find_elements(By.CLASS_NAME, 'share-box-feed-entry__trigger')
logger.debug("Found %d 'Start a post' buttons", len(buttons))
for i, button in enumerate(buttons):
logger.debug("Button %d text: %s", i + 1, button.text.strip())
if any(button.text.strip().lower() == 'start a post' for button in buttons):
logger.info("Found 'Start a post' button indicating user is logged in.")
return True
profile_img_elements = self.driver.find_elements(By.XPATH, "//img[contains(@alt, 'Photo of')]")
if profile_img_elements:
logger.info("Profile image found. Assuming user is logged in.")
return True
logger.info("Did not find 'Start a post' button or profile image. User might not be logged in.")
return False
except TimeoutException:
logger.error("Page elements took too long to load or were not found.")
return False
def wait_for_page_load(self, timeout=10): def wait_for_page_load(self, timeout=10):
try: try:
logger.debug("Waiting for page to load with timeout: %s seconds", timeout)
WebDriverWait(self.driver, timeout).until( WebDriverWait(self.driver, timeout).until(
lambda d: d.execute_script('return document.readyState') == 'complete' lambda d: d.execute_script('return document.readyState') == 'complete'
) )
logger.debug("Page load completed.")
except TimeoutException: except TimeoutException:
logger.error("Page load timed out.")
print("Page load timed out.") print("Page load timed out.")

View file

@ -1,8 +1,13 @@
from src.utils import logger
class LinkedInBotState: class LinkedInBotState:
def __init__(self): def __init__(self):
logger.debug("Initializing LinkedInBotState")
self.reset() self.reset()
def reset(self): def reset(self):
logger.debug("Resetting LinkedInBotState")
self.credentials_set = False self.credentials_set = False
self.api_key_set = False self.api_key_set = False
self.job_application_profile_set = False self.job_application_profile_set = False
@ -11,12 +16,17 @@ class LinkedInBotState:
self.logged_in = False self.logged_in = False
def validate_state(self, required_keys): def validate_state(self, required_keys):
logger.debug("Validating LinkedInBotState with required keys: %s", required_keys)
for key in required_keys: for key in required_keys:
if not getattr(self, key): if not getattr(self, key):
logger.error("State validation failed: %s is not set", key)
raise ValueError(f"{key.replace('_', ' ').capitalize()} must be set before proceeding.") raise ValueError(f"{key.replace('_', ' ').capitalize()} must be set before proceeding.")
logger.debug("State validation passed")
class LinkedInBotFacade: class LinkedInBotFacade:
def __init__(self, login_component, apply_component): def __init__(self, login_component, apply_component):
logger.debug("Initializing LinkedInBotFacade")
self.login_component = login_component self.login_component = login_component
self.apply_component = apply_component self.apply_component = apply_component
self.state = LinkedInBotState() self.state = LinkedInBotState()
@ -27,47 +37,65 @@ class LinkedInBotFacade:
self.parameters = None self.parameters = None
def set_job_application_profile_and_resume(self, job_application_profile, resume): def set_job_application_profile_and_resume(self, job_application_profile, resume):
logger.debug("Setting job application profile and resume")
self._validate_non_empty(job_application_profile, "Job application profile") self._validate_non_empty(job_application_profile, "Job application profile")
self._validate_non_empty(resume, "Resume") self._validate_non_empty(resume, "Resume")
self.job_application_profile = job_application_profile self.job_application_profile = job_application_profile
self.resume = resume self.resume = resume
self.state.job_application_profile_set = True self.state.job_application_profile_set = True
logger.debug("Job application profile and resume set successfully")
def set_secrets(self, email, password): def set_secrets(self, email, password):
logger.debug("Setting secrets: email and password")
self._validate_non_empty(email, "Email") self._validate_non_empty(email, "Email")
self._validate_non_empty(password, "Password") self._validate_non_empty(password, "Password")
self.email = email self.email = email
self.password = password self.password = password
self.state.credentials_set = True self.state.credentials_set = True
logger.debug("Secrets set successfully")
def set_gpt_answerer_and_resume_generator(self, gpt_answerer_component, resume_generator_manager): def set_gpt_answerer_and_resume_generator(self, gpt_answerer_component, resume_generator_manager):
logger.debug("Setting GPT answerer and resume generator")
self._ensure_job_profile_and_resume_set() self._ensure_job_profile_and_resume_set()
gpt_answerer_component.set_job_application_profile(self.job_application_profile) gpt_answerer_component.set_job_application_profile(self.job_application_profile)
gpt_answerer_component.set_resume(self.resume) gpt_answerer_component.set_resume(self.resume)
self.apply_component.set_gpt_answerer(gpt_answerer_component) self.apply_component.set_gpt_answerer(gpt_answerer_component)
self.apply_component.set_resume_generator_manager(resume_generator_manager) self.apply_component.set_resume_generator_manager(resume_generator_manager)
self.state.gpt_answerer_set = True self.state.gpt_answerer_set = True
logger.debug("GPT answerer and resume generator set successfully")
def set_parameters(self, parameters): def set_parameters(self, parameters):
logger.debug("Setting parameters")
self._validate_non_empty(parameters, "Parameters") self._validate_non_empty(parameters, "Parameters")
self.parameters = parameters self.parameters = parameters
self.apply_component.set_parameters(parameters) self.apply_component.set_parameters(parameters)
self.state.parameters_set = True self.state.parameters_set = True
logger.debug("Parameters set successfully")
def start_login(self): def start_login(self):
logger.debug("Starting login process")
self.state.validate_state(['credentials_set']) self.state.validate_state(['credentials_set'])
self.login_component.set_secrets(self.email, self.password) self.login_component.set_secrets(self.email, self.password)
self.login_component.start() self.login_component.start()
self.state.logged_in = True self.state.logged_in = True
logger.debug("Login process completed successfully")
def start_apply(self): def start_apply(self):
logger.debug("Starting apply process")
self.state.validate_state(['logged_in', 'job_application_profile_set', 'gpt_answerer_set', 'parameters_set']) self.state.validate_state(['logged_in', 'job_application_profile_set', 'gpt_answerer_set', 'parameters_set'])
self.apply_component.start_applying() self.apply_component.start_applying()
logger.debug("Apply process started successfully")
def _validate_non_empty(self, value, name): def _validate_non_empty(self, value, name):
logger.debug("Validating that %s is not empty", name)
if not value: if not value:
logger.error("Validation failed: %s is empty", name)
raise ValueError(f"{name} cannot be empty.") raise ValueError(f"{name} cannot be empty.")
logger.debug("Validation passed for %s", name)
def _ensure_job_profile_and_resume_set(self): def _ensure_job_profile_and_resume_set(self):
logger.debug("Ensuring job profile and resume are set")
if not self.state.job_application_profile_set: if not self.state.job_application_profile_set:
logger.error("Job application profile and resume are not set")
raise ValueError("Job application profile and resume must be set before proceeding.") raise ValueError("Job application profile and resume must be set before proceeding.")
logger.debug("Job profile and resume are set")

View file

@ -3,24 +3,29 @@ import json
import os import os
import random import random
import re import re
import tempfile
import time import time
import traceback import traceback
from datetime import date
from typing import List, Optional, Any, Tuple from typing import List, Optional, Any, Tuple
from httpx import HTTPStatusError
from reportlab.lib.pagesizes import letter from reportlab.lib.pagesizes import letter
from reportlab.pdfgen import canvas from reportlab.pdfgen import canvas
from selenium.common.exceptions import NoSuchElementException from selenium.common.exceptions import NoSuchElementException, TimeoutException
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys from selenium.webdriver.common.keys import Keys
from selenium.webdriver.remote.webelement import WebElement from selenium.webdriver.remote.webelement import WebElement
from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.ui import Select, WebDriverWait from selenium.webdriver.support.ui import Select, WebDriverWait
from selenium.webdriver import ActionChains
import src.utils as utils import src.utils as utils
from src.utils import logger
class LinkedInEasyApplier: class LinkedInEasyApplier:
def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: List[Tuple[str, str, str]], gpt_answerer: Any, resume_generator_manager): def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: List[Tuple[str, str, str]],
gpt_answerer: Any, resume_generator_manager):
logger.debug("Initializing LinkedInEasyApplier")
if resume_dir is None or not os.path.exists(resume_dir): if resume_dir is None or not os.path.exists(resume_dir):
resume_dir = None resume_dir = None
self.driver = driver self.driver = driver
@ -30,106 +35,248 @@ class LinkedInEasyApplier:
self.resume_generator_manager = resume_generator_manager self.resume_generator_manager = resume_generator_manager
self.all_data = self._load_questions_from_json() self.all_data = self._load_questions_from_json()
logger.debug("LinkedInEasyApplier initialized successfully")
def _load_questions_from_json(self) -> List[dict]: def _load_questions_from_json(self) -> List[dict]:
output_file = 'answers.json' output_file = 'answers.json'
logger.debug("Loading questions from JSON file: %s", output_file)
try: try:
try: with open(output_file, 'r') as f:
with open(output_file, 'r') as f: try:
try: data = json.load(f)
data = json.load(f) if not isinstance(data, list):
if not isinstance(data, list): raise ValueError("JSON file format is incorrect. Expected a list of questions.")
raise ValueError("JSON file format is incorrect. Expected a list of questions.") except json.JSONDecodeError:
except json.JSONDecodeError: logger.error("JSON decoding failed")
data = [] data = []
except FileNotFoundError: logger.debug("Questions loaded successfully from JSON")
data = []
return data return data
except FileNotFoundError:
logger.warning("JSON file not found, returning empty list")
return []
except Exception: except Exception:
tb_str = traceback.format_exc() tb_str = traceback.format_exc()
logger.error("Error loading questions data from JSON file: %s", tb_str)
raise Exception(f"Error loading questions data from JSON file: \nTraceback:\n{tb_str}") raise Exception(f"Error loading questions data from JSON file: \nTraceback:\n{tb_str}")
def check_for_premium_redirect(self, job: Any, max_attempts=3):
"""Проверяет, был ли выполнен редирект на страницу LinkedIn Premium.
В случае редиректа возвращает пользователя на исходную страницу вакансии."""
current_url = self.driver.current_url
attempts = 0
while "linkedin.com/premium" in current_url and attempts < max_attempts:
logger.warning("Redirected to LinkedIn Premium page. Attempting to return to job page.")
attempts += 1
self.driver.get(job.link)
time.sleep(2)
current_url = self.driver.current_url
if "linkedin.com/premium" in current_url:
logger.error("Failed to return to job page after %d attempts. Cannot apply for the job.", max_attempts)
raise Exception(
f"Redirected to LinkedIn Premium page and failed to return after {max_attempts} attempts. Job application aborted.")
def job_apply(self, job: Any): def job_apply(self, job: Any):
self.driver.get(job.link) logger.debug("Starting job application for job: %s", job)
time.sleep(random.uniform(3, 5))
try: try:
easy_apply_button = self._find_easy_apply_button() self.driver.get(job.link)
job.set_job_description(self._get_job_description()) logger.debug("Navigated to job link: %s", job.link)
job.set_recruiter_link(self._get_job_recruiter()) except Exception as e:
logger.error("Failed to navigate to job link: %s, error: %s", job.link, str(e))
raise
time.sleep(random.uniform(3, 5))
self.check_for_premium_redirect(job)
try:
self.driver.execute_script("document.activeElement.blur();")
logger.debug("Focus removed from the active element")
self.check_for_premium_redirect(job)
easy_apply_button = self._find_easy_apply_button(job)
self.check_for_premium_redirect(job)
logger.debug("Retrieving job description")
job_description = self._get_job_description()
job.set_job_description(job_description)
logger.debug("Job description set: %s", job_description[:100])
logger.debug("Retrieving recruiter link")
recruiter_link = self._get_job_recruiter()
job.set_recruiter_link(recruiter_link)
logger.debug("Recruiter link set: %s", recruiter_link)
logger.debug("Attempting to click 'Easy Apply' button")
actions = ActionChains(self.driver) actions = ActionChains(self.driver)
actions.move_to_element(easy_apply_button).click().perform() actions.move_to_element(easy_apply_button).click().perform()
self.gpt_answerer.set_job(job) logger.debug("'Easy Apply' button clicked successfully")
self._fill_application_form(job)
except Exception:
tb_str = traceback.format_exc()
self._discard_application()
raise Exception(f"Failed to apply to job! Original exception: \nTraceback:\n{tb_str}")
def _find_easy_apply_button(self) -> WebElement: logger.debug("Passing job information to GPT Answerer")
self.gpt_answerer.set_job(job)
logger.debug("Filling out application form")
self._fill_application_form(job)
logger.debug("Job application process completed successfully for job: %s", job)
except Exception as e:
tb_str = traceback.format_exc()
logger.error("Failed to apply to job: %s. Error traceback: %s", job, tb_str)
logger.debug("Discarding application due to failure")
self._discard_application()
raise Exception(f"Failed to apply to job! Original exception:\nTraceback:\n{tb_str}")
def _find_easy_apply_button(self, job: Any) -> WebElement:
logger.debug("Searching for 'Easy Apply' button")
attempt = 0 attempt = 0
search_methods = [
{
'description': "find all 'Easy Apply' buttons using find_elements",
'find_elements': True,
'xpath': '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]'
},
{
'description': "'aria-label' containing 'Easy Apply to'",
'xpath': '//button[contains(@aria-label, "Easy Apply to")]'
},
{
'description': "button text search",
'xpath': '//button[contains(text(), "Easy Apply") or contains(text(), "Apply now")]'
}
]
while attempt < 2: while attempt < 2:
self.check_for_premium_redirect(job)
self._scroll_page() self._scroll_page()
buttons = WebDriverWait(self.driver, 10).until(
EC.presence_of_all_elements_located( for method in search_methods:
(By.XPATH, '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]')
)
)
for index, _ in enumerate(buttons):
try: try:
button = WebDriverWait(self.driver, 10).until( logger.debug(f"Attempting search using {method['description']}")
EC.element_to_be_clickable(
(By.XPATH, f'(//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")])[{index + 1}]') if method.get('find_elements'):
# Поиск всех кнопок "Easy Apply"
buttons = self.driver.find_elements(By.XPATH, method['xpath'])
if buttons:
for index, button in enumerate(buttons):
try:
WebDriverWait(self.driver, 10).until(EC.visibility_of(button))
WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button))
logger.debug(f"Found 'Easy Apply' button {index + 1}, attempting to click")
return button
except Exception as e:
logger.warning(f"Button {index + 1} found but not clickable: {e}")
else:
raise TimeoutException("No 'Easy Apply' buttons found")
else:
button = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.XPATH, method['xpath']))
) )
) WebDriverWait(self.driver, 10).until(EC.visibility_of(button))
return button WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button))
logger.debug("Found 'Easy Apply' button, attempting to click")
return button
except TimeoutException:
logger.warning(f"Timeout during search using {method['description']}")
except Exception as e: except Exception as e:
pass logger.warning(
f"Failed to click 'Easy Apply' button using {method['description']} on attempt {attempt + 1}: {e}")
self.check_for_premium_redirect(job)
if attempt == 0: if attempt == 0:
logger.debug("Refreshing page to retry finding 'Easy Apply' button")
self.driver.refresh() self.driver.refresh()
time.sleep(3) time.sleep(random.randint(3, 5))
attempt += 1 attempt += 1
page_source = self.driver.page_source
logger.error("No clickable 'Easy Apply' button found after 2 attempts. Page source:\n%s", page_source)
raise Exception("No clickable 'Easy Apply' button found") raise Exception("No clickable 'Easy Apply' button found")
def _get_job_description(self) -> str: def _get_job_description(self) -> str:
logger.debug("Getting job description")
try: try:
see_more_button = self.driver.find_element(By.XPATH, '//button[@aria-label="Click to see more description"]') try:
actions = ActionChains(self.driver) see_more_button = self.driver.find_element(By.XPATH,
actions.move_to_element(see_more_button).click().perform() '//button[@aria-label="Click to see more description"]')
time.sleep(2) actions = ActionChains(self.driver)
actions.move_to_element(see_more_button).click().perform()
time.sleep(2)
except NoSuchElementException:
logger.debug("See more button not found, skipping")
description = self.driver.find_element(By.CLASS_NAME, 'jobs-description-content__text').text description = self.driver.find_element(By.CLASS_NAME, 'jobs-description-content__text').text
logger.debug("Job description retrieved successfully")
return description return description
except NoSuchElementException: except NoSuchElementException:
tb_str = traceback.format_exc() tb_str = traceback.format_exc()
raise Exception("Job description 'See more' button not found: \nTraceback:\n{tb_str}") logger.error("Job description not found: %s", tb_str)
raise Exception(f"Job description not found: \nTraceback:\n{tb_str}")
except Exception: except Exception:
tb_str = traceback.format_exc() tb_str = traceback.format_exc()
logger.error("Error getting Job description: %s", tb_str)
raise Exception(f"Error getting Job description: \nTraceback:\n{tb_str}") raise Exception(f"Error getting Job description: \nTraceback:\n{tb_str}")
def _get_job_recruiter(self): def _get_job_recruiter(self):
logger.debug("Getting job recruiter information")
try: try:
hiring_team_section = WebDriverWait(self.driver, 10).until( hiring_team_section = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.XPATH, '//h2[text()="Meet the hiring team"]')) EC.presence_of_element_located((By.XPATH, '//h2[text()="Meet the hiring team"]'))
) )
recruiter_element = hiring_team_section.find_element(By.XPATH, './/following::a[contains(@href, "linkedin.com/in/")]') logger.debug("Hiring team section found")
recruiter_link = recruiter_element.get_attribute('href')
return recruiter_link recruiter_elements = hiring_team_section.find_elements(By.XPATH,
'.//following::a[contains(@href, "linkedin.com/in/")]')
if recruiter_elements:
recruiter_element = recruiter_elements[0]
recruiter_link = recruiter_element.get_attribute('href')
logger.debug("Job recruiter link retrieved successfully: %s", recruiter_link)
return recruiter_link
else:
logger.debug("No recruiter link found in the hiring team section")
return ""
except Exception as e: except Exception as e:
logger.warning("Failed to retrieve recruiter information: %s", e)
return "" return ""
def _scroll_page(self) -> None: def _scroll_page(self) -> None:
logger.debug("Scrolling the page")
scrollable_element = self.driver.find_element(By.TAG_NAME, 'html') scrollable_element = self.driver.find_element(By.TAG_NAME, 'html')
utils.scroll_slow(self.driver, scrollable_element, step=300, reverse=False) utils.scroll_slow(self.driver, scrollable_element, step=300, reverse=False)
utils.scroll_slow(self.driver, scrollable_element, step=300, reverse=True) utils.scroll_slow(self.driver, scrollable_element, step=300, reverse=True)
def _fill_application_form(self, job): def _fill_application_form(self, job):
logger.debug("Filling out application form for job: %s", job)
while True: while True:
self.fill_up(job) self.fill_up(job)
if self._next_or_submit(): if self._next_or_submit():
logger.debug("Application form submitted")
break break
def _next_or_submit(self): def _next_or_submit(self):
logger.debug("Clicking 'Next' or 'Submit' button")
next_button = self.driver.find_element(By.CLASS_NAME, "artdeco-button--primary") next_button = self.driver.find_element(By.CLASS_NAME, "artdeco-button--primary")
button_text = next_button.text.lower() button_text = next_button.text.lower()
if 'submit application' in button_text: if 'submit application' in button_text:
logger.debug("Submit button found, submitting application")
self._unfollow_company() self._unfollow_company()
time.sleep(random.uniform(1.5, 2.5)) time.sleep(random.uniform(1.5, 2.5))
next_button.click() next_button.click()
@ -142,104 +289,304 @@ class LinkedInEasyApplier:
def _unfollow_company(self) -> None: def _unfollow_company(self) -> None:
try: try:
logger.debug("Unfollowing company")
follow_checkbox = self.driver.find_element( follow_checkbox = self.driver.find_element(
By.XPATH, "//label[contains(.,'to stay up to date with their page.')]") By.XPATH, "//label[contains(.,'to stay up to date with their page.')]")
follow_checkbox.click() follow_checkbox.click()
except Exception as e: except Exception as e:
pass logger.warning("Failed to unfollow company: %s", e)
def _check_for_errors(self) -> None: def _check_for_errors(self) -> None:
logger.debug("Checking for form errors")
error_elements = self.driver.find_elements(By.CLASS_NAME, 'artdeco-inline-feedback--error') error_elements = self.driver.find_elements(By.CLASS_NAME, 'artdeco-inline-feedback--error')
if error_elements: if error_elements:
logger.error("Form submission failed with errors: %s", [e.text for e in error_elements])
raise Exception(f"Failed answering or file upload. {str([e.text for e in error_elements])}") raise Exception(f"Failed answering or file upload. {str([e.text for e in error_elements])}")
def _discard_application(self) -> None: def _discard_application(self) -> None:
logger.debug("Discarding application")
try: try:
self.driver.find_element(By.CLASS_NAME, 'artdeco-modal__dismiss').click() self.driver.find_element(By.CLASS_NAME, 'artdeco-modal__dismiss').click()
time.sleep(random.uniform(3, 5)) time.sleep(random.uniform(3, 5))
self.driver.find_elements(By.CLASS_NAME, 'artdeco-modal__confirm-dialog-btn')[0].click() self.driver.find_elements(By.CLASS_NAME, 'artdeco-modal__confirm-dialog-btn')[0].click()
time.sleep(random.uniform(3, 5)) time.sleep(random.uniform(3, 5))
except Exception as e: except Exception as e:
pass logger.warning("Failed to discard application: %s", e)
def fill_up(self, job) -> None: def fill_up(self, job) -> None:
easy_apply_content = self.driver.find_element(By.CLASS_NAME, 'jobs-easy-apply-content') logger.debug("Filling up form sections for job: %s", job)
pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4')
for element in pb4_elements: try:
self._process_form_element(element, job) easy_apply_content = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.CLASS_NAME, 'jobs-easy-apply-content'))
)
pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4')
for element in pb4_elements:
self._process_form_element(element, job)
except Exception as e:
logger.error(f"Failed to find form elements: {e}")
def _process_form_element(self, element: WebElement, job) -> None: def _process_form_element(self, element: WebElement, job) -> None:
logger.debug("Processing form element")
if self._is_upload_field(element): if self._is_upload_field(element):
self._handle_upload_fields(element, job) self._handle_upload_fields(element, job)
else: else:
self._fill_additional_questions() self._fill_additional_questions()
def _handle_dropdown_fields(self, element: WebElement) -> None:
logger.debug("Handling dropdown fields")
dropdown = element.find_element(By.TAG_NAME, 'select')
select = Select(dropdown)
options = [option.text for option in select.options]
logger.debug(f"Dropdown options found: {options}")
parent_element = dropdown.find_element(By.XPATH, '../..')
label_elements = parent_element.find_elements(By.TAG_NAME, 'label')
if label_elements:
question_text = label_elements[0].text.lower()
else:
question_text = "unknown"
logger.debug(f"Detected question text: {question_text}")
existing_answer = None
for item in self.all_data:
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown':
existing_answer = item['answer']
break
if existing_answer:
logger.debug(f"Found existing answer for question '{question_text}': {existing_answer}")
else:
logger.debug(f"No existing answer found, querying model for: {question_text}")
existing_answer = self.gpt_answerer.answer_question_from_options(question_text, options)
logger.debug(f"Model provided answer: {existing_answer}")
self._save_questions_to_json({'type': 'dropdown', 'question': question_text, 'answer': existing_answer})
if existing_answer in options:
select.select_by_visible_text(existing_answer)
logger.debug(f"Selected option: {existing_answer}")
else:
logger.error(f"Answer '{existing_answer}' is not a valid option in the dropdown")
raise Exception(f"Invalid option selected: {existing_answer}")
def _is_upload_field(self, element: WebElement) -> bool: def _is_upload_field(self, element: WebElement) -> bool:
return bool(element.find_elements(By.XPATH, ".//input[@type='file']")) is_upload = bool(element.find_elements(By.XPATH, ".//input[@type='file']"))
logger.debug("Element is upload field: %s", is_upload)
return is_upload
def _handle_upload_fields(self, element: WebElement, job) -> None: def _handle_upload_fields(self, element: WebElement, job) -> None:
logger.debug("Handling upload fields")
try:
show_more_button = self.driver.find_element(By.XPATH,
"//button[contains(@aria-label, 'Show more resumes')]")
show_more_button.click()
logger.debug("Clicked 'Show more resumes' button")
except NoSuchElementException:
logger.debug("'Show more resumes' button not found, continuing...")
file_upload_elements = self.driver.find_elements(By.XPATH, "//input[@type='file']") file_upload_elements = self.driver.find_elements(By.XPATH, "//input[@type='file']")
for element in file_upload_elements: for element in file_upload_elements:
parent = element.find_element(By.XPATH, "..") parent = element.find_element(By.XPATH, "..")
self.driver.execute_script("arguments[0].classList.remove('hidden')", element) self.driver.execute_script("arguments[0].classList.remove('hidden')", element)
output = self.gpt_answerer.resume_or_cover(parent.text.lower()) output = self.gpt_answerer.resume_or_cover(parent.text.lower())
if 'resume' in output: if 'resume' in output:
logger.debug("Uploading resume")
if self.resume_path is not None and self.resume_path.resolve().is_file(): if self.resume_path is not None and self.resume_path.resolve().is_file():
element.send_keys(str(self.resume_path.resolve())) element.send_keys(str(self.resume_path.resolve()))
logger.debug(f"Resume uploaded from path: {self.resume_path.resolve()}")
else: else:
logger.debug("Resume path not found or invalid, generating new resume")
self._create_and_upload_resume(element, job) self._create_and_upload_resume(element, job)
elif 'cover' in output: elif 'cover' in output:
self._create_and_upload_cover_letter(element) logger.debug("Uploading cover letter")
self._create_and_upload_cover_letter(element, job)
logger.debug("Finished handling upload fields")
def _create_and_upload_resume(self, element, job): def _create_and_upload_resume(self, element, job):
logger.debug("Starting the process of creating and uploading resume.")
folder_path = 'generated_cv' folder_path = 'generated_cv'
os.makedirs(folder_path, exist_ok=True)
try: try:
file_path_pdf = os.path.join(folder_path, f"CV_{random.randint(0, 9999)}.pdf") if not os.path.exists(folder_path):
with open(file_path_pdf, "xb") as f: logger.debug(f"Creating directory at path: {folder_path}")
f.write(base64.b64decode(self.resume_generator_manager.pdf_base64(job_description_text=job.description))) os.makedirs(folder_path, exist_ok=True)
except Exception as e:
logger.error(f"Failed to create directory: {folder_path}. Error: {e}")
raise
while True:
try:
timestamp = int(time.time())
file_path_pdf = os.path.join(folder_path, f"CV_{timestamp}.pdf")
logger.debug(f"Generated file path for resume: {file_path_pdf}")
logger.debug(f"Generating resume for job: {job.title} at {job.company}")
resume_pdf_base64 = self.resume_generator_manager.pdf_base64(job_description_text=job.description)
with open(file_path_pdf, "xb") as f:
f.write(base64.b64decode(resume_pdf_base64))
logger.debug(f"Resume successfully generated and saved to: {file_path_pdf}")
break
except HTTPStatusError as e:
if e.response.status_code == 429:
retry_after = e.response.headers.get('retry-after')
retry_after_ms = e.response.headers.get('retry-after-ms')
if retry_after:
wait_time = int(retry_after)
logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...")
elif retry_after_ms:
wait_time = int(retry_after_ms) / 1000.0
logger.warning(f"Rate limit exceeded, waiting {wait_time} milliseconds before retrying...")
else:
wait_time = 20
logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...")
time.sleep(wait_time)
else:
logger.error(f"HTTP error: {e}")
raise
except Exception as e:
logger.error(f"Failed to generate resume: {e}")
tb_str = traceback.format_exc()
logger.error(f"Traceback: {tb_str}")
if "RateLimitError" in str(e):
logger.warning("Rate limit error encountered, retrying...")
time.sleep(20)
else:
raise
file_size = os.path.getsize(file_path_pdf)
max_file_size = 2 * 1024 * 1024 # 2 MB
logger.debug(f"Resume file size: {file_size} bytes")
if file_size > max_file_size:
logger.error(f"Resume file size exceeds 2 MB: {file_size} bytes")
raise ValueError("Resume file size exceeds the maximum limit of 2 MB.")
allowed_extensions = {'.pdf', '.doc', '.docx'}
file_extension = os.path.splitext(file_path_pdf)[1].lower()
logger.debug(f"Resume file extension: {file_extension}")
if file_extension not in allowed_extensions:
logger.error(f"Invalid resume file format: {file_extension}")
raise ValueError("Resume file format is not allowed. Only PDF, DOC, and DOCX formats are supported.")
try:
logger.debug(f"Uploading resume from path: {file_path_pdf}")
element.send_keys(os.path.abspath(file_path_pdf)) element.send_keys(os.path.abspath(file_path_pdf))
job.pdf_path = os.path.abspath(file_path_pdf) job.pdf_path = os.path.abspath(file_path_pdf)
time.sleep(2) time.sleep(2)
except Exception: logger.debug(f"Resume created and uploaded successfully: {file_path_pdf}")
except Exception as e:
tb_str = traceback.format_exc() tb_str = traceback.format_exc()
logger.error(f"Resume upload failed: {tb_str}")
raise Exception(f"Upload failed: \nTraceback:\n{tb_str}") raise Exception(f"Upload failed: \nTraceback:\n{tb_str}")
def _create_and_upload_cover_letter(self, element: WebElement) -> None: def _create_and_upload_cover_letter(self, element: WebElement, job) -> None:
cover_letter = self.gpt_answerer.answer_question_textual_wide_range("Write a cover letter") logger.debug("Starting the process of creating and uploading cover letter.")
with tempfile.NamedTemporaryFile(delete=False, suffix='.pdf') as temp_pdf_file:
letter_path = temp_pdf_file.name cover_letter_text = self.gpt_answerer.answer_question_textual_wide_range("Write a cover letter")
c = canvas.Canvas(letter_path, pagesize=letter)
_, height = letter folder_path = 'generated_cv'
text_object = c.beginText(100, height - 100)
text_object.setFont("Helvetica", 12) try:
text_object.textLines(cover_letter)
c.drawText(text_object) if not os.path.exists(folder_path):
c.save() logger.debug(f"Creating directory at path: {folder_path}")
element.send_keys(letter_path) os.makedirs(folder_path, exist_ok=True)
except Exception as e:
logger.error(f"Failed to create directory: {folder_path}. Error: {e}")
raise
while True:
try:
timestamp = int(time.time())
file_path_pdf = os.path.join(folder_path, f"Cover_Letter_{timestamp}.pdf")
logger.debug(f"Generated file path for cover letter: {file_path_pdf}")
c = canvas.Canvas(file_path_pdf, pagesize=letter)
_, height = letter
text_object = c.beginText(100, height - 100)
text_object.setFont("Helvetica", 12)
text_object.textLines(cover_letter_text)
c.drawText(text_object)
c.save()
logger.debug(f"Cover letter successfully generated and saved to: {file_path_pdf}")
break
except Exception as e:
logger.error(f"Failed to generate cover letter: {e}")
tb_str = traceback.format_exc()
logger.error(f"Traceback: {tb_str}")
raise
file_size = os.path.getsize(file_path_pdf)
max_file_size = 2 * 1024 * 1024 # 2 MB
logger.debug(f"Cover letter file size: {file_size} bytes")
if file_size > max_file_size:
logger.error(f"Cover letter file size exceeds 2 MB: {file_size} bytes")
raise ValueError("Cover letter file size exceeds the maximum limit of 2 MB.")
allowed_extensions = {'.pdf', '.doc', '.docx'}
file_extension = os.path.splitext(file_path_pdf)[1].lower()
logger.debug(f"Cover letter file extension: {file_extension}")
if file_extension not in allowed_extensions:
logger.error(f"Invalid cover letter file format: {file_extension}")
raise ValueError("Cover letter file format is not allowed. Only PDF, DOC, and DOCX formats are supported.")
try:
logger.debug(f"Uploading cover letter from path: {file_path_pdf}")
element.send_keys(os.path.abspath(file_path_pdf))
job.cover_letter_path = os.path.abspath(file_path_pdf)
time.sleep(2)
logger.debug(f"Cover letter created and uploaded successfully: {file_path_pdf}")
except Exception as e:
tb_str = traceback.format_exc()
logger.error(f"Cover letter upload failed: {tb_str}")
raise Exception(f"Upload failed: \nTraceback:\n{tb_str}")
def _fill_additional_questions(self) -> None: def _fill_additional_questions(self) -> None:
logger.debug("Filling additional questions")
form_sections = self.driver.find_elements(By.CLASS_NAME, 'jobs-easy-apply-form-section__grouping') form_sections = self.driver.find_elements(By.CLASS_NAME, 'jobs-easy-apply-form-section__grouping')
for section in form_sections: for section in form_sections:
self._process_form_section(section) self._process_form_section(section)
def _process_form_section(self, section: WebElement) -> None: def _process_form_section(self, section: WebElement) -> None:
logger.debug("Processing form section")
if self._handle_terms_of_service(section): if self._handle_terms_of_service(section):
logger.debug("Handled terms of service")
return return
if self._find_and_handle_radio_question(section): if self._find_and_handle_radio_question(section):
logger.debug("Handled radio question")
return return
if self._find_and_handle_textbox_question(section): if self._find_and_handle_textbox_question(section):
logger.debug("Handled textbox question")
return return
if self._find_and_handle_date_question(section): if self._find_and_handle_date_question(section):
logger.debug("Handled date question")
return return
if self._find_and_handle_dropdown_question(section): if self._find_and_handle_dropdown_question(section):
logger.debug("Handled dropdown question")
return return
def _handle_terms_of_service(self, element: WebElement) -> bool: def _handle_terms_of_service(self, element: WebElement) -> bool:
checkbox = element.find_elements(By.TAG_NAME, 'label') checkbox = element.find_elements(By.TAG_NAME, 'label')
if checkbox and any(term in checkbox[0].text.lower() for term in ['terms of service', 'privacy policy', 'terms of use']): if checkbox and any(
term in checkbox[0].text.lower() for term in ['terms of service', 'privacy policy', 'terms of use']):
checkbox[0].click() checkbox[0].click()
logger.debug("Clicked terms of service checkbox")
return True return True
return False return False
@ -249,39 +596,81 @@ class LinkedInEasyApplier:
if radios: if radios:
question_text = section.text.lower() question_text = section.text.lower()
options = [radio.text.lower() for radio in radios] options = [radio.text.lower() for radio in radios]
existing_answer = None existing_answer = None
for item in self.all_data: for item in self.all_data:
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'radio': if self._sanitize_text(question_text) in item['question'] and item['type'] == 'radio':
existing_answer = item existing_answer = item
self._select_radio(radios, existing_answer['answer'])
return True break
if existing_answer:
self._select_radio(radios, existing_answer['answer'])
logger.debug("Selected existing radio answer")
return True
answer = self.gpt_answerer.answer_question_from_options(question_text, options) answer = self.gpt_answerer.answer_question_from_options(question_text, options)
self._save_questions_to_json({'type': 'radio', 'question': question_text, 'answer': answer}) self._save_questions_to_json({'type': 'radio', 'question': question_text, 'answer': answer})
self._select_radio(radios, answer) self._select_radio(radios, answer)
logger.debug("Selected new radio answer")
return True return True
return False return False
def _find_and_handle_textbox_question(self, section: WebElement) -> bool: def _find_and_handle_textbox_question(self, section: WebElement) -> bool:
logger.debug("Searching for text fields in the section.")
text_fields = section.find_elements(By.TAG_NAME, 'input') + section.find_elements(By.TAG_NAME, 'textarea') text_fields = section.find_elements(By.TAG_NAME, 'input') + section.find_elements(By.TAG_NAME, 'textarea')
if text_fields: if text_fields:
text_field = text_fields[0] text_field = text_fields[0]
question_text = section.find_element(By.TAG_NAME, 'label').text.lower() question_text = section.find_element(By.TAG_NAME, 'label').text.lower().strip()
logger.debug(f"Found text field with label: {question_text}")
is_numeric = self._is_numeric_field(text_field) is_numeric = self._is_numeric_field(text_field)
if is_numeric: logger.debug(f"Is the field numeric? {'Yes' if is_numeric else 'No'}")
question_type = 'numeric'
answer = self.gpt_answerer.answer_question_numeric(question_text)
else:
question_type = 'textbox'
answer = self.gpt_answerer.answer_question_textual_wide_range(question_text)
existing_answer = None existing_answer = None
question_type = 'numeric' if is_numeric else 'textbox'
for item in self.all_data: for item in self.all_data:
if 'cover' not in item['question'] and item['question'] == self._sanitize_text(question_text) and item['type'] == question_type:
logger.debug(
f"Comparing sanitized stored question: '{self._sanitize_text(item['question'])}' and type: '{item.get('type')}' with current question: '{self._sanitize_text(question_text)}' and type: '{question_type}'")
if self._sanitize_text(item['question']) == self._sanitize_text(question_text) and item.get(
'type') == question_type:
existing_answer = item existing_answer = item
self._enter_text(text_field, existing_answer['answer']) logger.debug(f"Found existing answer in the data: {existing_answer['answer']}")
return True break
if existing_answer:
self._enter_text(text_field, existing_answer['answer'])
logger.debug("Entered existing answer into the textbox.")
time.sleep(1)
text_field.send_keys(Keys.ARROW_DOWN)
text_field.send_keys(Keys.ENTER)
logger.debug("Selected first option from the dropdown.")
return True
if is_numeric:
answer = self.gpt_answerer.answer_question_numeric(question_text)
logger.debug(f"Generated numeric answer: {answer}")
else:
answer = self.gpt_answerer.answer_question_textual_wide_range(question_text)
logger.debug(f"Generated textual answer: {answer}")
self._save_questions_to_json({'type': question_type, 'question': question_text, 'answer': answer}) self._save_questions_to_json({'type': question_type, 'question': question_text, 'answer': answer})
self._enter_text(text_field, answer) self._enter_text(text_field, answer)
logger.debug("Entered new answer into the textbox and saved it to JSON.")
time.sleep(1)
text_field.send_keys(Keys.ARROW_DOWN)
text_field.send_keys(Keys.ENTER)
logger.debug("Selected first option from the dropdown.")
return True return True
logger.debug("No text fields found in the section.")
return False return False
def _find_and_handle_date_question(self, section: WebElement) -> bool: def _find_and_handle_date_question(self, section: WebElement) -> bool:
@ -294,49 +683,80 @@ class LinkedInEasyApplier:
existing_answer = None existing_answer = None
for item in self.all_data: for item in self.all_data:
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'date': if self._sanitize_text(question_text) in item['question'] and item['type'] == 'date':
existing_answer = item existing_answer = item
self._enter_text(date_field, existing_answer['answer'])
return True break
if existing_answer:
self._enter_text(date_field, existing_answer['answer'])
logger.debug("Entered existing date answer")
return True
self._save_questions_to_json({'type': 'date', 'question': question_text, 'answer': answer_text}) self._save_questions_to_json({'type': 'date', 'question': question_text, 'answer': answer_text})
self._enter_text(date_field, answer_text) self._enter_text(date_field, answer_text)
logger.debug("Entered new date answer")
return True return True
return False return False
def _find_and_handle_dropdown_question(self, section: WebElement) -> bool: def _find_and_handle_dropdown_question(self, section: WebElement) -> bool:
try: try:
question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element') question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element')
question_text = question.find_element(By.TAG_NAME, 'label').text.lower() question_text = question.find_element(By.TAG_NAME, 'label').text.lower()
dropdown = question.find_element(By.TAG_NAME, 'select') logger.debug(f"Processing dropdown or combobox question: {question_text}")
if dropdown:
dropdowns = question.find_elements(By.TAG_NAME, 'select')
if dropdowns:
dropdown = dropdowns[0]
select = Select(dropdown) select = Select(dropdown)
options = [option.text for option in select.options] options = [option.text for option in select.options]
logger.debug(f"Dropdown options found: {options}")
current_selection = select.first_selected_option.text
logger.debug(f"Current selection: {current_selection}")
existing_answer = None existing_answer = None
for item in self.all_data: for item in self.all_data:
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown': if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown':
existing_answer = item existing_answer = item['answer']
self._select_dropdown_option(dropdown, existing_answer['answer']) break
return True
if existing_answer:
logger.debug(f"Found existing answer for question '{question_text}': {existing_answer}")
if current_selection != existing_answer:
logger.debug(f"Updating selection to: {existing_answer}")
self._select_dropdown_option(dropdown, existing_answer)
return True
logger.debug(f"No existing answer found, querying model for: {question_text}")
answer = self.gpt_answerer.answer_question_from_options(question_text, options) answer = self.gpt_answerer.answer_question_from_options(question_text, options)
self._save_questions_to_json({'type': 'dropdown', 'question': question_text, 'answer': answer}) self._save_questions_to_json({'type': 'dropdown', 'question': question_text, 'answer': answer})
self._select_dropdown_option(dropdown, answer) self._select_dropdown_option(dropdown, answer)
logger.debug(f"Selected new dropdown answer: {answer}")
return True return True
except Exception:
return False
except Exception as e:
logger.warning(f"Failed to handle dropdown or combobox question: {e}")
return False return False
def _is_numeric_field(self, field: WebElement) -> bool: def _is_numeric_field(self, field: WebElement) -> bool:
field_type = field.get_attribute('type').lower() field_type = field.get_attribute('type').lower()
if 'numeric' in field_type: field_id = field.get_attribute("id").lower()
return True is_numeric = 'numeric' in field_id or field_type == 'number' or ('text' == field_type and 'numeric' in field_id)
class_attribute = field.get_attribute("id") logger.debug("Field type: %s, Field ID: %s, Is numeric: %s", field_type, field_id, is_numeric)
return class_attribute and 'numeric' in class_attribute return is_numeric
def _enter_text(self, element: WebElement, text: str) -> None: def _enter_text(self, element: WebElement, text: str) -> None:
logger.debug("Entering text: %s", text)
element.clear() element.clear()
element.send_keys(text) element.send_keys(text)
def _select_radio(self, radios: List[WebElement], answer: str) -> None: def _select_radio(self, radios: List[WebElement], answer: str) -> None:
logger.debug("Selecting radio option: %s", answer)
for radio in radios: for radio in radios:
if answer in radio.text.lower(): if answer in radio.text.lower():
radio.find_element(By.TAG_NAME, 'label').click() radio.find_element(By.TAG_NAME, 'label').click()
@ -344,12 +764,14 @@ class LinkedInEasyApplier:
radios[-1].find_element(By.TAG_NAME, 'label').click() radios[-1].find_element(By.TAG_NAME, 'label').click()
def _select_dropdown_option(self, element: WebElement, text: str) -> None: def _select_dropdown_option(self, element: WebElement, text: str) -> None:
logger.debug("Selecting dropdown option: %s", text)
select = Select(element) select = Select(element)
select.select_by_visible_text(text) select.select_by_visible_text(text)
def _save_questions_to_json(self, question_data: dict) -> None: def _save_questions_to_json(self, question_data: dict) -> None:
output_file = 'answers.json' output_file = 'answers.json'
question_data['question'] = self._sanitize_text(question_data['question']) question_data['question'] = self._sanitize_text(question_data['question'])
logger.debug("Saving question data to JSON: %s", question_data)
try: try:
try: try:
with open(output_file, 'r') as f: with open(output_file, 'r') as f:
@ -358,22 +780,22 @@ class LinkedInEasyApplier:
if not isinstance(data, list): if not isinstance(data, list):
raise ValueError("JSON file format is incorrect. Expected a list of questions.") raise ValueError("JSON file format is incorrect. Expected a list of questions.")
except json.JSONDecodeError: except json.JSONDecodeError:
logger.error("JSON decoding failed")
data = [] data = []
except FileNotFoundError: except FileNotFoundError:
logger.warning("JSON file not found, creating new file")
data = [] data = []
data.append(question_data) data.append(question_data)
with open(output_file, 'w') as f: with open(output_file, 'w') as f:
json.dump(data, f, indent=4) json.dump(data, f, indent=4)
logger.debug("Question data saved successfully to JSON")
except Exception: except Exception:
tb_str = traceback.format_exc() tb_str = traceback.format_exc()
logger.error("Error saving questions data to JSON file: %s", tb_str)
raise Exception(f"Error saving questions data to JSON file: \nTraceback:\n{tb_str}") raise Exception(f"Error saving questions data to JSON file: \nTraceback:\n{tb_str}")
def _sanitize_text(self, text: str) -> str: def _sanitize_text(self, text: str) -> str:
sanitized_text = text.lower() sanitized_text = text.lower().strip().replace('"', '').replace('\\', '')
sanitized_text = sanitized_text.strip() sanitized_text = re.sub(r'[\x00-\x1F\x7F]', '', sanitized_text).replace('\n', ' ').replace('\r', '').rstrip(',')
sanitized_text = sanitized_text.replace('"', '') logger.debug("Sanitized text: %s", sanitized_text)
sanitized_text = sanitized_text.replace('\\', '')
sanitized_text = re.sub(r'[\x00-\x1F\x7F]', '', sanitized_text)
sanitized_text = sanitized_text.replace('\n', ' ').replace('\r', '')
sanitized_text = sanitized_text.rstrip(',')
return sanitized_text return sanitized_text

View file

@ -1,37 +1,50 @@
import json
import os import os
import random import random
import time import time
import traceback
from itertools import product from itertools import product
from pathlib import Path from pathlib import Path
from selenium.common.exceptions import NoSuchElementException from selenium.common.exceptions import NoSuchElementException
from selenium.webdriver.common.by import By from selenium.webdriver.common.by import By
import src.utils as utils import src.utils as utils
from src.job import Job from src.job import Job
from src.linkedIn_easy_applier import LinkedInEasyApplier from src.linkedIn_easy_applier import LinkedInEasyApplier
import json from src.utils import logger
class EnvironmentKeys: class EnvironmentKeys:
def __init__(self): def __init__(self):
logger.debug("Initializing EnvironmentKeys")
self.skip_apply = self._read_env_key_bool("SKIP_APPLY") self.skip_apply = self._read_env_key_bool("SKIP_APPLY")
self.disable_description_filter = self._read_env_key_bool("DISABLE_DESCRIPTION_FILTER") self.disable_description_filter = self._read_env_key_bool("DISABLE_DESCRIPTION_FILTER")
logger.debug("EnvironmentKeys initialized: skip_apply=%s, disable_description_filter=%s",
self.skip_apply, self.disable_description_filter)
@staticmethod @staticmethod
def _read_env_key(key: str) -> str: def _read_env_key(key: str) -> str:
return os.getenv(key, "") value = os.getenv(key, "")
logger.debug("Read environment key %s: %s", key, value)
return value
@staticmethod @staticmethod
def _read_env_key_bool(key: str) -> bool: def _read_env_key_bool(key: str) -> bool:
return os.getenv(key) == "True" value = os.getenv(key) == "True"
logger.debug("Read environment key %s as bool: %s", key, value)
return value
class LinkedInJobManager: class LinkedInJobManager:
def __init__(self, driver): def __init__(self, driver):
logger.debug("Initializing LinkedInJobManager")
self.driver = driver self.driver = driver
self.set_old_answers = set() self.set_old_answers = set()
self.easy_applier_component = None self.easy_applier_component = None
logger.debug("LinkedInJobManager initialized successfully")
def set_parameters(self, parameters): def set_parameters(self, parameters):
logger.debug("Setting parameters for LinkedInJobManager")
self.company_blacklist = parameters.get('companyBlacklist', []) or [] self.company_blacklist = parameters.get('companyBlacklist', []) or []
self.title_blacklist = parameters.get('titleBlacklist', []) or [] self.title_blacklist = parameters.get('titleBlacklist', []) or []
self.positions = parameters.get('positions', []) self.positions = parameters.get('positions', [])
@ -40,22 +53,23 @@ class LinkedInJobManager:
self.base_search_url = self.get_base_search_url(parameters) self.base_search_url = self.get_base_search_url(parameters)
self.seen_jobs = [] self.seen_jobs = []
resume_path = parameters.get('uploads', {}).get('resume', None) resume_path = parameters.get('uploads', {}).get('resume', None)
if resume_path is not None and Path(resume_path).exists(): self.resume_path = Path(resume_path) if resume_path and Path(resume_path).exists() else None
self.resume_path = Path(resume_path)
else:
self.resume_path = None
self.output_file_directory = Path(parameters['outputFileDirectory']) self.output_file_directory = Path(parameters['outputFileDirectory'])
self.env_config = EnvironmentKeys() self.env_config = EnvironmentKeys()
#self.old_question() logger.debug("Parameters set successfully")
def set_gpt_answerer(self, gpt_answerer): def set_gpt_answerer(self, gpt_answerer):
logger.debug("Setting GPT answerer")
self.gpt_answerer = gpt_answerer self.gpt_answerer = gpt_answerer
def set_resume_generator_manager(self, resume_generator_manager): def set_resume_generator_manager(self, resume_generator_manager):
logger.debug("Setting resume generator manager")
self.resume_generator_manager = resume_generator_manager self.resume_generator_manager = resume_generator_manager
def start_applying(self): def start_applying(self):
self.easy_applier_component = LinkedInEasyApplier(self.driver, self.resume_path, self.set_old_answers, self.gpt_answerer, self.resume_generator_manager) logger.debug("Starting job application process")
self.easy_applier_component = LinkedInEasyApplier(self.driver, self.resume_path, self.set_old_answers,
self.gpt_answerer, self.resume_generator_manager)
searches = list(product(self.positions, self.locations)) searches = list(product(self.positions, self.locations))
random.shuffle(searches) random.shuffle(searches)
page_sleep = 0 page_sleep = 0
@ -75,51 +89,113 @@ class LinkedInJobManager:
self.next_job_page(position, location_url, job_page_number) self.next_job_page(position, location_url, job_page_number)
time.sleep(random.uniform(1.5, 3.5)) time.sleep(random.uniform(1.5, 3.5))
utils.printyellow("Starting the application process for this page...") utils.printyellow("Starting the application process for this page...")
self.apply_jobs()
try:
jobs = self.get_jobs_from_page()
if not jobs:
utils.printyellow("No more jobs found on this page. Exiting loop.")
break
except Exception as e:
logger.error(f"Failed to retrieve jobs: {e}")
break
try:
self.apply_jobs()
except Exception as e:
logger.error("Error during job application: %s", e)
utils.printred(f"Error during job application: {e}")
continue
utils.printyellow("Applying to jobs on this page has been completed!") utils.printyellow("Applying to jobs on this page has been completed!")
time_left = minimum_page_time - time.time() time_left = minimum_page_time - time.time()
if time_left > 0: if time_left > 0:
utils.printyellow(f"Sleeping for {time_left} seconds.") utils.printyellow(f"Sleeping for {time_left} seconds.")
logger.debug("Sleeping for %d seconds", time_left)
time.sleep(time_left) time.sleep(time_left)
minimum_page_time = time.time() + minimum_time minimum_page_time = time.time() + minimum_time
if page_sleep % 5 == 0: if page_sleep % 5 == 0:
sleep_time = random.randint(5, 34) sleep_time = random.randint(5, 34)
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.")
logger.debug("Sleeping for %d seconds", sleep_time)
time.sleep(sleep_time) time.sleep(sleep_time)
page_sleep += 1 page_sleep += 1
except Exception: except Exception as e:
traceback.format_exc() logger.error("Unexpected error during job search: %s", e)
pass utils.printred(f"Unexpected error: {e}")
continue
time_left = minimum_page_time - time.time() time_left = minimum_page_time - time.time()
if time_left > 0: if time_left > 0:
utils.printyellow(f"Sleeping for {time_left} seconds.") utils.printyellow(f"Sleeping for {time_left} seconds.")
logger.debug("Sleeping for %d seconds", time_left)
time.sleep(time_left) time.sleep(time_left)
minimum_page_time = time.time() + minimum_time minimum_page_time = time.time() + minimum_time
if page_sleep % 5 == 0: if page_sleep % 5 == 0:
sleep_time = random.randint(50, 90) sleep_time = random.randint(50, 90)
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.")
logger.debug("Sleeping for %d seconds", sleep_time)
time.sleep(sleep_time) time.sleep(sleep_time)
page_sleep += 1 page_sleep += 1
def get_jobs_from_page(self):
try:
no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand')
if 'No matching jobs found' in no_jobs_element.text or 'unfortunately, things aren' in self.driver.page_source.lower():
utils.printyellow("No matching jobs found on this page.")
logger.debug("No matching jobs found on this page, skipping.")
return []
except NoSuchElementException:
pass
try:
job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list")
utils.scroll_slow(self.driver, job_results)
utils.scroll_slow(self.driver, job_results, step=300, reverse=True)
job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[
0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item')
if not job_list_elements:
utils.printyellow("No job class elements found on page.")
logger.debug("No job class elements found on page, skipping.")
return []
return job_list_elements
except NoSuchElementException:
logger.debug("No job results found on the page.")
return []
except Exception as e:
logger.error(f"Error while fetching job elements: {e}")
return []
def apply_jobs(self): def apply_jobs(self):
try: try:
no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand') no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand')
if 'No matching jobs found' in no_jobs_element.text or 'unfortunately, things aren' in self.driver.page_source.lower(): if 'No matching jobs found' in no_jobs_element.text or 'unfortunately, things aren' in self.driver.page_source.lower():
raise Exception("No more jobs on this page") utils.printyellow("No matching jobs found on this page, moving to next.")
logger.debug("No matching jobs found on this page, skipping")
return
except NoSuchElementException: except NoSuchElementException:
pass pass
job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list") job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list")
utils.scroll_slow(self.driver, job_results) utils.scroll_slow(self.driver, job_results)
utils.scroll_slow(self.driver, job_results, step=300, reverse=True) utils.scroll_slow(self.driver, job_results, step=300, reverse=True)
job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item') job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[
0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item')
if not job_list_elements: if not job_list_elements:
raise Exception("No job class elements found on page") utils.printyellow("No job class elements found on page, moving to next page.")
job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements] logger.debug("No job class elements found on page, skipping")
return
job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements]
for job in job_list: for job in job_list:
if self.is_blacklisted(job.title, job.company, job.link): if self.is_blacklisted(job.title, job.company, job.link):
utils.printyellow(f"Blacklisted {job.title} at {job.company}, skipping...") utils.printyellow(f"Blacklisted {job.title} at {job.company}, skipping...")
logger.debug("Job blacklisted: %s at %s", job.title, job.company)
self.write_to_file(job, "skipped") self.write_to_file(job, "skipped")
continue continue
if self.is_already_applied_to_job(job.title, job.company, job.link): if self.is_already_applied_to_job(job.title, job.company, job.link):
@ -132,12 +208,15 @@ class LinkedInJobManager:
if job.apply_method not in {"Continue", "Applied", "Apply"}: if job.apply_method not in {"Continue", "Applied", "Apply"}:
self.easy_applier_component.job_apply(job) self.easy_applier_component.job_apply(job)
self.write_to_file(job, "success") self.write_to_file(job, "success")
logger.debug("Applied to job: %s at %s", job.title, job.company)
except Exception as e: except Exception as e:
utils.printred(traceback.format_exc()) logger.error("Failed to apply for %s at %s: %s", job.title, job.company, e)
utils.printred(f"Failed to apply for {job.title} at {job.company}: {e}")
self.write_to_file(job, "failed") self.write_to_file(job, "failed")
continue continue
def write_to_file(self, job, file_name): def write_to_file(self, job, file_name):
logger.debug("Writing job application result to file: %s", file_name)
pdf_path = Path(job.pdf_path).resolve() pdf_path = Path(job.pdf_path).resolve()
pdf_path = pdf_path.as_uri() pdf_path = pdf_path.as_uri()
data = { data = {
@ -152,22 +231,27 @@ class LinkedInJobManager:
if not file_path.exists(): if not file_path.exists():
with open(file_path, 'w', encoding='utf-8') as f: with open(file_path, 'w', encoding='utf-8') as f:
json.dump([data], f, indent=4) json.dump([data], f, indent=4)
logger.debug("Job data written to new file: %s", file_path)
else: else:
with open(file_path, 'r+', encoding='utf-8') as f: with open(file_path, 'r+', encoding='utf-8') as f:
try: try:
existing_data = json.load(f) existing_data = json.load(f)
except json.JSONDecodeError: except json.JSONDecodeError:
logger.error("JSON decode error in file: %s", file_path)
existing_data = [] existing_data = []
existing_data.append(data) existing_data.append(data)
f.seek(0) f.seek(0)
json.dump(existing_data, f, indent=4) json.dump(existing_data, f, indent=4)
f.truncate() f.truncate()
logger.debug("Job data appended to existing file: %s", file_path)
def get_base_search_url(self, parameters): def get_base_search_url(self, parameters):
logger.debug("Constructing base search URL")
url_parts = [] url_parts = []
if parameters['remote']: if parameters['remote']:
url_parts.append("f_CF=f_WRA") url_parts.append("f_CF=f_WRA")
experience_levels = [str(i+1) for i, (level, v) in enumerate(parameters.get('experienceLevel', {}).items()) if v] experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experienceLevel', {}).items()) if
v]
if experience_levels: if experience_levels:
url_parts.append(f"f_E={','.join(experience_levels)}") url_parts.append(f"f_E={','.join(experience_levels)}")
url_parts.append(f"distance={parameters['distance']}") url_parts.append(f"distance={parameters['distance']}")
@ -183,35 +267,51 @@ class LinkedInJobManager:
date_param = next((v for k, v in date_mapping.items() if parameters.get('date', {}).get(k)), "") date_param = next((v for k, v in date_mapping.items() if parameters.get('date', {}).get(k)), "")
url_parts.append("f_LF=f_AL") # Easy Apply url_parts.append("f_LF=f_AL") # Easy Apply
base_url = "&".join(url_parts) base_url = "&".join(url_parts)
return f"?{base_url}{date_param}" full_url = f"?{base_url}{date_param}"
logger.debug("Base search URL constructed: %s", full_url)
return full_url
def next_job_page(self, position, location, job_page): def next_job_page(self, position, location, job_page):
self.driver.get(f"https://www.linkedin.com/jobs/search/{self.base_search_url}&keywords={position}{location}&start={job_page * 25}") logger.debug("Navigating to next job page: %s in %s, page %d", position, location, job_page)
self.driver.get(
f"https://www.linkedin.com/jobs/search/{self.base_search_url}&keywords={position}{location}&start={job_page * 25}")
def extract_job_information_from_tile(self, job_tile): def extract_job_information_from_tile(self, job_tile):
logger.debug("Extracting job information from tile")
job_title, company, job_location, apply_method, link = "", "", "", "", "" job_title, company, job_location, apply_method, link = "", "", "", "", ""
try: try:
job_title = job_tile.find_element(By.CLASS_NAME, 'job-card-list__title').text job_title = job_tile.find_element(By.CLASS_NAME, 'job-card-list__title').text
link = job_tile.find_element(By.CLASS_NAME, 'job-card-list__title').get_attribute('href').split('?')[0] link = job_tile.find_element(By.CLASS_NAME, 'job-card-list__title').get_attribute('href').split('?')[0]
company = job_tile.find_element(By.CLASS_NAME, 'job-card-container__primary-description').text company = job_tile.find_element(By.CLASS_NAME, 'job-card-container__primary-description').text
except: logger.debug("Job information extracted: %s at %s", job_title, company)
pass except NoSuchElementException:
utils.printyellow("Some job information (title, link, or company) is missing.")
logger.warning("Some job information (title, link, or company) is missing.")
try: try:
job_location = job_tile.find_element(By.CLASS_NAME, 'job-card-container__metadata-item').text job_location = job_tile.find_element(By.CLASS_NAME, 'job-card-container__metadata-item').text
except: except NoSuchElementException:
pass utils.printyellow("Job location is missing.")
logger.warning("Job location is missing.")
try: try:
apply_method = job_tile.find_element(By.CLASS_NAME, 'job-card-container__apply-method').text apply_method = job_tile.find_element(By.CLASS_NAME, 'job-card-container__apply-method').text
except: except NoSuchElementException:
apply_method = "Applied" apply_method = "Applied"
utils.printyellow("Apply method not found, assuming 'Applied'.")
logger.warning("Apply method not found, assuming 'Applied'.")
return job_title, company, job_location, link, apply_method return job_title, company, job_location, link, apply_method
def is_blacklisted(self, job_title, company, link): def is_blacklisted(self, job_title, company, link):
logger.debug("Checking if job is blacklisted: %s at %s", job_title, company)
job_title_words = job_title.lower().split(' ') job_title_words = job_title.lower().split(' ')
title_blacklisted = any(word in job_title_words for word in self.title_blacklist) title_blacklisted = any(word in job_title_words for word in self.title_blacklist)
company_blacklisted = company.strip().lower() in (word.strip().lower() for word in self.company_blacklist) company_blacklisted = company.strip().lower() in (word.strip().lower() for word in self.company_blacklist)
link_seen = link in self.seen_jobs link_seen = link in self.seen_jobs
is_blacklisted = title_blacklisted or company_blacklisted or link_seen
logger.debug("Job blacklisted status: %s", is_blacklisted)
return is_blacklisted
return title_blacklisted or company_blacklisted or link_seen return title_blacklisted or company_blacklisted or link_seen
def is_already_applied_to_job(self, job_title, company, link): def is_already_applied_to_job(self, job_title, company, link):
@ -237,4 +337,5 @@ class LinkedInJobManager:
return True return True
except json.JSONDecodeError: except json.JSONDecodeError:
continue continue
return False return False

View file

@ -181,7 +181,7 @@ Answer the following question based on the provided language skills.
- Answer questions directly. - Answer questions directly.
- If it seems likely that you have the experience, even if not explicitly defined, answer as if you have the experience. - If it seems likely that you have the experience, even if not explicitly defined, answer as if you have the experience.
- If unsure, respond with "I have no experience with that, but I learn fast" or "Not yet, but willing to learn." - If unsure, respond with "I have no experience with that, but I learn fast" or "Not yet, but willing to learn."
- Keep the answer under 140 characters. - Keep the answer under 140 characters. Do not add any additional languages what is not in my experience
## Example ## Example
My resume: Fluent in Italian and English. My resume: Fluent in Italian and English.
@ -238,7 +238,6 @@ This comprehensive overview will serve as a guideline for the recruitment proces
# Job Description Summary""" # Job Description Summary"""
coverletter_template = """ coverletter_template = """
Compose a brief and impactful cover letter based on the provided job description and resume. The letter should be no longer than three paragraphs and should be written in a professional, yet conversational tone. Avoid using any placeholders, and ensure that the letter flows naturally and is tailored to the job. Compose a brief and impactful cover letter based on the provided job description and resume. The letter should be no longer than three paragraphs and should be written in a professional, yet conversational tone. Avoid using any placeholders, and ensure that the letter flows naturally and is tailored to the job.
@ -371,7 +370,6 @@ Options: [1-2, 3-5, 6-10, 10+]
## """ ## """
try_to_fix_template = """\ try_to_fix_template = """\
The objective is to fix the text of a form input on a web page. The objective is to fix the text of a form input on a web page.

View file

@ -1,103 +1,171 @@
import logging
import os import os
import random import random
import time import time
from selenium import webdriver from selenium import webdriver
log_file = "app_log.log"
logging.basicConfig(
level=logging.DEBUG,
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s',
handlers=[
logging.FileHandler(log_file, mode='a', encoding='utf-8'),
logging.StreamHandler()
],
force=True # This will reset the root logger's handlers and apply the new configuration
)
logger = logging.getLogger(__name__)
file_handler = logging.FileHandler(log_file, mode='a', encoding='utf-8')
formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s')
file_handler.setFormatter(formatter)
logger.addHandler(file_handler)
logger.setLevel(logging.DEBUG)
chromeProfilePath = os.path.join(os.getcwd(), "chrome_profile", "linkedin_profile") chromeProfilePath = os.path.join(os.getcwd(), "chrome_profile", "linkedin_profile")
def ensure_chrome_profile(): def ensure_chrome_profile():
logger.debug("Ensuring Chrome profile exists at path: %s", chromeProfilePath)
profile_dir = os.path.dirname(chromeProfilePath) profile_dir = os.path.dirname(chromeProfilePath)
if not os.path.exists(profile_dir): if not os.path.exists(profile_dir):
os.makedirs(profile_dir) os.makedirs(profile_dir)
logger.debug("Created directory for Chrome profile: %s", profile_dir)
if not os.path.exists(chromeProfilePath): if not os.path.exists(chromeProfilePath):
os.makedirs(chromeProfilePath) os.makedirs(chromeProfilePath)
logger.debug("Created Chrome profile directory: %s", chromeProfilePath)
return chromeProfilePath return chromeProfilePath
def is_scrollable(element): def is_scrollable(element):
scroll_height = element.get_attribute("scrollHeight") scroll_height = element.get_attribute("scrollHeight")
client_height = element.get_attribute("clientHeight") client_height = element.get_attribute("clientHeight")
return int(scroll_height) > int(client_height) scrollable = int(scroll_height) > int(client_height)
logger.debug("Element scrollable check: scrollHeight=%s, clientHeight=%s, scrollable=%s", scroll_height,
client_height, scrollable)
return scrollable
def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse=False):
logger.debug("Starting slow scroll: start=%d, end=%d, step=%d, reverse=%s", start, end, step, reverse)
def scroll_slow(driver, scrollable_element, start=0, end=3600, step=100, reverse=False):
if reverse: if reverse:
start, end = end, start start, end = end, start
step = -step step = -step
if step == 0: if step == 0:
logger.error("Step value cannot be zero.")
raise ValueError("Step cannot be zero.") raise ValueError("Step cannot be zero.")
max_scroll_height = int(scrollable_element.get_attribute("scrollHeight"))
current_scroll_position = int(scrollable_element.get_attribute("scrollTop"))
logger.debug("Max scroll height of the element: %d", max_scroll_height)
logger.debug("Current scroll position: %d", current_scroll_position)
if reverse:
if current_scroll_position < start:
start = current_scroll_position
logger.debug("Adjusted start position for upward scroll: %d", start)
else:
if end > max_scroll_height:
logger.warning("End value exceeds the scroll height. Adjusting end to %d", max_scroll_height)
end = max_scroll_height
script_scroll_to = "arguments[0].scrollTop = arguments[1];" script_scroll_to = "arguments[0].scrollTop = arguments[1];"
try: try:
if scrollable_element.is_displayed(): if scrollable_element.is_displayed():
if not is_scrollable(scrollable_element): if not is_scrollable(scrollable_element):
logger.warning("The element is not scrollable.")
print("The element is not scrollable.") print("The element is not scrollable.")
return return
if (step > 0 and start >= end) or (step < 0 and start <= end): if (step > 0 and start >= end) or (step < 0 and start <= end):
logger.warning("No scrolling will occur due to incorrect start/end values.")
print("No scrolling will occur due to incorrect start/end values.") print("No scrolling will occur due to incorrect start/end values.")
return return
for position in range(start, end, step):
position = start
while (step > 0 and position < end) or (step < 0 and position > end):
try: try:
driver.execute_script(script_scroll_to, scrollable_element, position) driver.execute_script(script_scroll_to, scrollable_element, position)
logger.debug("Scrolled to position: %d", position)
except Exception as e: except Exception as e:
logger.error("Error during scrolling: %s", e)
print(f"Error during scrolling: {e}") print(f"Error during scrolling: {e}")
time.sleep(random.uniform(1.0, 2.6))
position += step
step = max(10, abs(step) - 10) * (-1 if reverse else 1)
time.sleep(random.uniform(0.6, 1.5))
driver.execute_script(script_scroll_to, scrollable_element, end) driver.execute_script(script_scroll_to, scrollable_element, end)
time.sleep(1) logger.debug("Scrolled to final position: %d", end)
time.sleep(0.5)
else: else:
logger.warning("The element is not visible.")
print("The element is not visible.") print("The element is not visible.")
except Exception as e: except Exception as e:
logger.error("Exception occurred during scrolling: %s", e)
print(f"Exception occurred: {e}") print(f"Exception occurred: {e}")
def chromeBrowserOptions():
def chrome_browser_options():
logger.debug("Setting Chrome browser options")
ensure_chrome_profile() ensure_chrome_profile()
options = webdriver.ChromeOptions() options = webdriver.ChromeOptions()
options.add_argument("--start-maximized") # Avvia il browser a schermo intero options.add_argument("--start-maximized")
options.add_argument("--no-sandbox") # Disabilita la sandboxing per migliorare le prestazioni options.add_argument("--no-sandbox")
options.add_argument("--disable-dev-shm-usage") # Utilizza una directory temporanea per la memoria condivisa options.add_argument("--disable-dev-shm-usage")
options.add_argument("--ignore-certificate-errors") # Ignora gli errori dei certificati SSL options.add_argument("--ignore-certificate-errors")
options.add_argument("--disable-extensions") # Disabilita le estensioni del browser options.add_argument("--disable-extensions")
options.add_argument("--disable-gpu") # Disabilita l'accelerazione GPU options.add_argument("--disable-gpu")
options.add_argument("window-size=1200x800") # Imposta la dimensione della finestra del browser options.add_argument("window-size=1200x800")
options.add_argument("--disable-background-timer-throttling") # Disabilita il throttling dei timer in background options.add_argument("--disable-background-timer-throttling")
options.add_argument("--disable-backgrounding-occluded-windows") # Disabilita la sospensione delle finestre occluse options.add_argument("--disable-backgrounding-occluded-windows")
options.add_argument("--disable-translate") # Disabilita il traduttore automatico options.add_argument("--disable-translate")
options.add_argument("--disable-popup-blocking") # Disabilita il blocco dei popup options.add_argument("--disable-popup-blocking")
options.add_argument("--no-first-run") # Disabilita la configurazione iniziale del browser options.add_argument("--no-first-run")
options.add_argument("--no-default-browser-check") # Disabilita il controllo del browser predefinito options.add_argument("--no-default-browser-check")
options.add_argument("--disable-logging") # Disabilita il logging options.add_argument("--disable-logging")
options.add_argument("--disable-autofill") # Disabilita l'autocompletamento dei moduli options.add_argument("--disable-autofill")
options.add_argument("--disable-plugins") # Disabilita i plugin del browser options.add_argument("--disable-plugins")
options.add_argument("--disable-animations") # Disabilita le animazioni options.add_argument("--disable-animations")
options.add_argument("--disable-cache") # Disabilita la cache options.add_argument("--disable-cache")
options.add_experimental_option("excludeSwitches", ["enable-automation", "enable-logging"]) # Esclude switch della modalità automatica e logging options.add_experimental_option("excludeSwitches", ["enable-automation", "enable-logging"])
# Preferenze per contenuti
prefs = { prefs = {
"profile.default_content_setting_values.images": 2, # Disabilita il caricamento delle immagini "profile.default_content_setting_values.images": 2,
"profile.managed_default_content_settings.stylesheets": 2, # Disabilita il caricamento dei fogli di stile "profile.managed_default_content_settings.stylesheets": 2,
} }
options.add_experimental_option("prefs", prefs) options.add_experimental_option("prefs", prefs)
if len(chromeProfilePath) > 0: if len(chromeProfilePath) > 0:
initialPath = os.path.dirname(chromeProfilePath) initial_path = os.path.dirname(chromeProfilePath)
profileDir = os.path.basename(chromeProfilePath) profile_dir = os.path.basename(chromeProfilePath)
options.add_argument('--user-data-dir=' + initialPath) options.add_argument('--user-data-dir=' + initial_path)
options.add_argument("--profile-directory=" + profileDir) options.add_argument("--profile-directory=" + profile_dir)
logger.debug("Using Chrome profile directory: %s", chromeProfilePath)
else: else:
options.add_argument("--incognito") options.add_argument("--incognito")
logger.debug("Using Chrome in incognito mode")
return options return options
def printred(text): def printred(text):
# Codice colore ANSI per il rosso red = "\033[91m"
RED = "\033[91m" reset = "\033[0m"
RESET = "\033[0m" logger.debug("Printing text in red: %s", text)
# Stampa il testo in rosso print(f"{red}{text}{reset}")
print(f"{RED}{text}{RESET}")
def printyellow(text): def printyellow(text):
# Codice colore ANSI per il giallo yellow = "\033[93m"
YELLOW = "\033[93m" reset = "\033[0m"
RESET = "\033[0m" logger.debug("Printing text in yellow: %s", text)
# Stampa il testo in giallo print(f"{yellow}{text}{reset}")
print(f"{YELLOW}{text}{RESET}")