diff --git a/data_folder/config.yaml b/data_folder/config.yaml index a037034..1cbe9ed 100644 --- a/data_folder/config.yaml +++ b/data_folder/config.yaml @@ -1,6 +1,6 @@ remote: [true/false] -experience_level: +experienceLevel: internship: [true/false] entry: [true/false] associate: [true/false] @@ -31,7 +31,7 @@ locations: - Country1 - Country2 -apply_once_at_company: [ true/false] +applyOnceAtCompany: [true/false] distance: 100 @@ -39,8 +39,7 @@ company_blacklist: - Company1 - Company2 - -title_blacklist: +titleBlacklist: - word1 - word2 diff --git a/data_folder_example/config.yaml b/data_folder_example/config.yaml index 316ab8f..b9ccefa 100644 --- a/data_folder_example/config.yaml +++ b/data_folder_example/config.yaml @@ -1,6 +1,6 @@ remote: true -experience_level: +experienceLevel: internship: true entry: true associate: true @@ -29,15 +29,15 @@ positions: locations: - USA -apply_once_at_company: [true/false] +applyOnceAtCompany: [true/false] distance: 100 -company_blacklist: +companyBlacklist: - Noir - Crossover -title_blacklist: +titleBlacklist: llm_model_type: openai llm_model: 'gpt-4o' diff --git a/main.py b/main.py index 68a0527..afa9044 100644 --- a/main.py +++ b/main.py @@ -7,9 +7,9 @@ import click from selenium import webdriver from selenium.webdriver.chrome.service import Service as ChromeService from webdriver_manager.chrome import ChromeDriverManager -from selenium.common.exceptions import WebDriverException -from lib_resume_builder_AIHawk import Resume, StyleManager, FacadeManager, ResumeGenerator -from src.utils import chrome_browser_options +from selenium.common.exceptions import WebDriverException, TimeoutException +from lib_resume_builder_AIHawk import Resume,StyleManager,FacadeManager,ResumeGenerator +from src.utils import chromeBrowserOptions from src.gpt import GPTAnswerer from src.linkedIn_authenticator import LinkedInAuthenticator from src.linkedIn_bot_facade import LinkedInBotFacade @@ -19,16 +19,14 @@ from src.job_application_profile import JobApplicationProfile # Suppress stderr sys.stderr = open(os.devnull, 'w') - class ConfigError(Exception): pass - class ConfigValidator: @staticmethod def validate_email(email: str) -> bool: return re.match(r'^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$', email) is not None - + @staticmethod def validate_yaml_file(yaml_path: Path) -> dict: try: @@ -38,37 +36,37 @@ class ConfigValidator: raise ConfigError(f"Error reading file {yaml_path}: {exc}") except FileNotFoundError: raise ConfigError(f"File not found: {yaml_path}") - + + def validate_config(config_yaml_path: Path) -> dict: parameters = ConfigValidator.validate_yaml_file(config_yaml_path) required_keys = { 'remote': bool, - 'experience_level': dict, + 'experienceLevel': dict, 'jobTypes': dict, 'date': dict, 'positions': list, 'locations': list, 'distance': int, - 'company_blacklist': list, - 'title_blacklist': list + 'companyBlacklist': list, + 'titleBlacklist': list } for key, expected_type in required_keys.items(): if key not in parameters: - if key in ['company_blacklist', 'title_blacklist']: + if key in ['companyBlacklist', 'titleBlacklist']: parameters[key] = [] else: raise ConfigError(f"Missing or invalid key '{key}' in config file {config_yaml_path}") elif not isinstance(parameters[key], expected_type): - if key in ['company_blacklist', 'title_blacklist'] and parameters[key] is None: + if key in ['companyBlacklist', 'titleBlacklist'] and parameters[key] is None: parameters[key] = [] else: - raise ConfigError( - f"Invalid type for key '{key}' in config file {config_yaml_path}. Expected {expected_type}.") + raise ConfigError(f"Invalid type for key '{key}' in config file {config_yaml_path}. Expected {expected_type}.") experience_levels = ['internship', 'entry', 'associate', 'mid-senior level', 'director', 'executive'] for level in experience_levels: - if not isinstance(parameters['experience_level'].get(level), bool): + if not isinstance(parameters['experienceLevel'].get(level), bool): raise ConfigError(f"Experience level '{level}' must be a boolean in config file {config_yaml_path}") job_types = ['full-time', 'contract', 'part-time', 'temporary', 'internship', 'other', 'volunteer'] @@ -88,10 +86,9 @@ class ConfigValidator: approved_distances = {0, 5, 10, 25, 50, 100} if parameters['distance'] not in approved_distances: - raise ConfigError( - f"Invalid distance value in config file {config_yaml_path}. Must be one of: {approved_distances}") + raise ConfigError(f"Invalid distance value in config file {config_yaml_path}. Must be one of: {approved_distances}") - for blacklist in ['company_blacklist', 'title_blacklist']: + for blacklist in ['companyBlacklist', 'titleBlacklist']: if not isinstance(parameters.get(blacklist), list): raise ConfigError(f"'{blacklist}' must be a list in config file {config_yaml_path}") if parameters[blacklist] is None: @@ -99,6 +96,8 @@ class ConfigValidator: return parameters + + @staticmethod def validate_secrets(secrets_yaml_path: Path) -> tuple: secrets = ConfigValidator.validate_yaml_file(secrets_yaml_path) @@ -114,13 +113,10 @@ class ConfigValidator: raise ConfigError(f"Password cannot be empty in secrets file {secrets_yaml_path}.") return secrets['email'], str(secrets['password']), secrets['llm_api_key'] - class FileManager: @staticmethod def find_file(name_containing: str, with_extension: str, at_path: Path) -> Path: - return next((file for file in at_path.iterdir() if - name_containing.lower() in file.name.lower() and file.suffix.lower() == with_extension.lower()), - None) + return next((file for file in at_path.iterdir() if name_containing.lower() in file.name.lower() and file.suffix.lower() == with_extension.lower()), None) @staticmethod def validate_data_folder(app_data_folder: Path) -> tuple: @@ -129,15 +125,13 @@ class FileManager: required_files = ['secrets.yaml', 'config.yaml', 'plain_text_resume.yaml'] missing_files = [file for file in required_files if not (app_data_folder / file).exists()] - + if missing_files: raise FileNotFoundError(f"Missing files in the data folder: {', '.join(missing_files)}") output_folder = app_data_folder / 'output' output_folder.mkdir(exist_ok=True) - return ( - app_data_folder / 'secrets.yaml', app_data_folder / 'config.yaml', app_data_folder / 'plain_text_resume.yaml', - output_folder) + return (app_data_folder / 'secrets.yaml', app_data_folder / 'config.yaml', app_data_folder / 'plain_text_resume.yaml', output_folder) @staticmethod def file_paths_to_dict(resume_file: Path | None, plain_text_resume_file: Path) -> dict: @@ -153,16 +147,14 @@ class FileManager: return result - def init_browser() -> webdriver.Chrome: try: - options = chrome_browser_options() + options = chromeBrowserOptions() service = ChromeService(ChromeDriverManager().install()) return webdriver.Chrome(service=service, options=options) except Exception as e: raise RuntimeError(f"Failed to initialize browser: {str(e)}") - def create_and_run_bot(email, password, parameters, llm_api_key): try: style_manager = StyleManager() @@ -170,14 +162,13 @@ def create_and_run_bot(email, password, parameters, llm_api_key): with open(parameters['uploads']['plainTextResume'], "r", encoding='utf-8') as file: plain_text_resume = file.read() resume_object = Resume(plain_text_resume) - resume_generator_manager = FacadeManager(llm_api_key, style_manager, resume_generator, resume_object, - Path("data_folder/output")) + resume_generator_manager = FacadeManager(llm_api_key, style_manager, resume_generator, resume_object, Path("data_folder/output")) os.system('cls' if os.name == 'nt' else 'clear') resume_generator_manager.choose_style() os.system('cls' if os.name == 'nt' else 'clear') - + job_application_profile_object = JobApplicationProfile(plain_text_resume) - + browser = init_browser() login_component = LinkedInAuthenticator(browser) apply_component = LinkedInJobManager(browser) @@ -196,40 +187,34 @@ def create_and_run_bot(email, password, parameters, llm_api_key): @click.command() -@click.option('--resume', type=click.Path(exists=True, file_okay=True, dir_okay=False, path_type=Path), - help="Path to the resume PDF file") +@click.option('--resume', type=click.Path(exists=True, file_okay=True, dir_okay=False, path_type=Path), help="Path to the resume PDF file") def main(resume: Path = None): try: data_folder = Path("data_folder") secrets_file, config_file, plain_text_resume_file, output_folder = FileManager.validate_data_folder(data_folder) - + parameters = ConfigValidator.validate_config(config_file) email, password, llm_api_key = ConfigValidator.validate_secrets(secrets_file) - + parameters['uploads'] = FileManager.file_paths_to_dict(resume, plain_text_resume_file) parameters['outputFileDirectory'] = output_folder - + create_and_run_bot(email, password, parameters, llm_api_key) except ConfigError as ce: print(f"Configuration error: {str(ce)}") - print( - "Refer to the configuration guide for troubleshooting: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") + print("Refer to the configuration guide for troubleshooting: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") except FileNotFoundError as fnf: print(f"File not found: {str(fnf)}") print("Ensure all required files are present in the data folder.") - print( - "Refer to the file setup guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") + print("Refer to the file setup guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") except RuntimeError as re: print(f"Runtime error: {str(re)}") - print( - "Refer to the configuration and troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") + print("Refer to the configuration and troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") except Exception as e: print(f"An unexpected error occurred: {str(e)}") - print( - "Refer to the general troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") - + print("Refer to the general troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration") if __name__ == "__main__": main() diff --git a/requirements.txt b/requirements.txt index 7e3d816..03290b7 100644 --- a/requirements.txt +++ b/requirements.txt @@ -13,9 +13,4 @@ webdriver-manager==4.0.2 click git+https://github.com/feder-cr/lib_resume_builder_AIHawk.git linkedin-api -pdfminer.six==20221105 -inputimeout==1.0.4 -langchain-ollama==0.1.3 -langchain-anthropic==0.1.3 -jsonschema==4.23.0 -jsonschema-specifications==2023.12.1 \ No newline at end of file +pdfminer.six==20221105 \ No newline at end of file diff --git a/resume_yaml_generator.py b/resume_yaml_generator.py index 053245f..336a23d 100644 --- a/resume_yaml_generator.py +++ b/resume_yaml_generator.py @@ -7,22 +7,19 @@ import re from jsonschema import validate, ValidationError from pdfminer.high_level import extract_text - def load_yaml(file_path: str) -> Dict[str, Any]: with open(file_path, 'r') as file: return yaml.safe_load(file) - def load_resume_text(file_path: str) -> str: with open(file_path, 'r') as file: return file.read() - def get_api_key() -> str: secrets_path = os.path.join('data_folder', 'secrets.yaml') if not os.path.exists(secrets_path): raise FileNotFoundError(f"Secrets file not found at {secrets_path}") - + secrets = load_yaml(secrets_path) if not 'llm_api_key' in secrets: @@ -31,10 +28,9 @@ def get_api_key() -> str: api_key = secrets.get('llm_api_key') if not api_key: raise ValueError("LLM API key not found in secrets.yaml") - + return api_key - def generate_yaml_from_resume(resume_text: str, schema: Dict[str, Any], api_key: str) -> str: client = OpenAI(api_key=api_key) @@ -87,15 +83,14 @@ def generate_yaml_from_resume(resume_text: str, schema: Dict[str, Any], api_key: response = client.chat.completions.create( model="gpt-4o-mini", messages=[ - {"role": "system", - "content": "You are a helpful assistant that generates structured YAML content from resume files, paying close attention to format requirements and schema structure."}, + {"role": "system", "content": "You are a helpful assistant that generates structured YAML content from resume files, paying close attention to format requirements and schema structure."}, {"role": "user", "content": prompt} ], temperature=0.5, ) yaml_content = response.choices[0].message.content.strip() - + # Extract YAML content from between the tags match = re.search(r'(.*?)', yaml_content, re.DOTALL) if match: @@ -103,12 +98,10 @@ def generate_yaml_from_resume(resume_text: str, schema: Dict[str, Any], api_key: else: raise ValueError("YAML content not found in the expected format") - def save_yaml(data: str, output_file: str): with open(output_file, 'w') as file: file.write(data) - def validate_yaml(yaml_content: str, schema: Dict[str, Any]) -> Dict[str, Any]: try: yaml_dict = yaml.safe_load(yaml_content) @@ -117,7 +110,6 @@ def validate_yaml(yaml_content: str, schema: Dict[str, Any]) -> Dict[str, Any]: except ValidationError as e: return {"valid": False, "errors": str(e)} - def generate_report(validation_result: Dict[str, Any], output_file: str): report = f"Validation Report for {output_file}\n" report += "=" * 40 + "\n" @@ -126,17 +118,14 @@ def generate_report(validation_result: Dict[str, Any], output_file: str): else: report += "YAML is not valid. Errors:\n" report += validation_result["errors"] + "\n" - + print(report) - def pdf_to_text(pdf_path: str) -> str: return extract_text(pdf_path) - def main(): - parser = argparse.ArgumentParser( - description="Generate a resume YAML file from a PDF or text resume using OpenAI API") + parser = argparse.ArgumentParser(description="Generate a resume YAML file from a PDF or text resume using OpenAI API") parser.add_argument("--input", required=True, help="Path to the input resume file (PDF or TXT)") parser.add_argument("--output", default="data_folder/plain_text_resume.yaml", help="Path to the output YAML file") args = parser.parse_args() @@ -167,6 +156,5 @@ def main(): except Exception as e: print(f"An error occurred: {e}") - if __name__ == "__main__": main() diff --git a/src/gpt.py b/src/gpt.py index d5f78ad..4107797 100644 --- a/src/gpt.py +++ b/src/gpt.py @@ -6,7 +6,8 @@ import time from abc import ABC, abstractmethod from datetime import datetime from pathlib import Path -from typing import Dict, List, Union +from typing import Dict, List +from typing import Union import httpx from Levenshtein import distance @@ -37,7 +38,7 @@ class OpenAIModel(AIModel): def invoke(self, prompt: str) -> str: print("invoke in openai") response = self.model.invoke(prompt) - return response.content + return response class ClaudeModel(AIModel): @@ -48,7 +49,7 @@ class ClaudeModel(AIModel): def invoke(self, prompt: str) -> str: response = self.model.invoke(prompt) - return response.content + return response class OllamaModel(AIModel): @@ -58,14 +59,14 @@ class OllamaModel(AIModel): def invoke(self, prompt: str) -> str: response = self.model.invoke(prompt) - return response.content + return response class AIAdapter: def __init__(self, config: dict, api_key: str): self.model = self._create_model(config, api_key) - def _create_model(self, config: dict, api_key: str) -> Union[OpenAIModel, OllamaModel, ClaudeModel]: + def _create_model(self, config: dict, api_key: str) -> AIModel: llm_model_type = config['llm_model_type'] llm_model = config['llm_model'] llm_api_url = config['llm_api_url'] @@ -78,7 +79,7 @@ class AIAdapter: elif llm_model_type == "ollama": return OllamaModel(api_key, llm_model, llm_api_url) else: - raise ValueError(f"Unsupported model type: {llm_model_type}") + raise ValueError(f"Unsupported model type: {model_type}") def invoke(self, prompt: str) -> str: return self.model.invoke(prompt) @@ -108,34 +109,25 @@ class LLMLogger: logger.debug("Prompts are of type StringPromptValue") prompts = prompts.text logger.debug("Prompts converted to text: %s", prompts) - elif isinstance(prompts, dict): - logger.debug("Prompts are of type dict") + elif isinstance(prompts, Dict): + logger.debug("Prompts are of type Dict") try: - if "messages" in prompts: - logger.debug("Prompts contain 'messages' key") - prompts = { - f"prompt_{i + 1}": prompt["content"] - for i, prompt in enumerate(prompts["messages"]) - } - logger.debug("Prompts converted to dictionary: %s", prompts) - else: - logger.debug("Prompts dictionary does not contain 'messages' key") + prompts = { + f"prompt_{i + 1}": prompt.content + for i, prompt in enumerate(prompts.messages) + } + logger.debug("Prompts converted to dictionary: %s", prompts) except Exception as e: logger.error("Error converting prompts to dictionary: %s", str(e)) raise else: logger.debug("Prompts are of unknown type, attempting default conversion") try: - if hasattr(prompts, "messages"): - logger.debug("Prompts have 'messages' attribute") - prompts = { - f"prompt_{i + 1}": prompt.content - for i, prompt in enumerate(prompts.messages) - } - logger.debug("Prompts converted to dictionary using default method: %s", prompts) - else: - logger.error("Prompts do not have 'messages' attribute, and default conversion failed") - raise ValueError("Prompts structure is not supported.") + prompts = { + f"prompt_{i + 1}": prompt.content + for i, prompt in enumerate(prompts.messages) + } + logger.debug("Prompts converted to dictionary using default method: %s", prompts) except Exception as e: logger.error("Error converting prompts using default method: %s", str(e)) raise @@ -299,7 +291,7 @@ class GPTAnswerer: def __init__(self, config, llm_api_key): self.ai_adapter = AIAdapter(config, llm_api_key) - self.llm_cheap = LoggerChatModel(self.ai_adapter.model) + self.llm_cheap = LoggerChatModel(self.ai_adapter) @property def job_description(self): diff --git a/src/linkedIn_easy_applier.py b/src/linkedIn_easy_applier.py index 9363ac5..cf64245 100644 --- a/src/linkedIn_easy_applier.py +++ b/src/linkedIn_easy_applier.py @@ -5,8 +5,7 @@ import random import re import time import traceback -from pathlib import Path -from typing import List, Optional, Any, Tuple, Set +from typing import List, Optional, Any, Tuple from httpx import HTTPStatusError from reportlab.lib.pagesizes import A4 @@ -24,13 +23,11 @@ from src.utils import logger class LinkedInEasyApplier: - def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: Set[Tuple[str, str, str]], + def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: List[Tuple[str, str, str]], gpt_answerer: Any, resume_generator_manager): logger.debug("Initializing LinkedInEasyApplier") if resume_dir is None or not os.path.exists(resume_dir): resume_dir = None - else: - resume_dir = Path(resume_dir) self.driver = driver self.resume_path = resume_dir self.set_old_answers = set_old_answers @@ -541,19 +538,17 @@ class LinkedInEasyApplier: lines = split_text_by_width(cover_letter_text, "Helvetica", 12, max_width) - line_height = 14 - max_lines_per_page = int(available_height // line_height) - for line in lines: text_height = text_object.getY() + if text_height > bottom_margin: + text_object.textLine(line) + else: - if text_height - line_height < bottom_margin: c.drawText(text_object) c.showPage() text_object = c.beginText(50, page_height - 50) text_object.setFont("Helvetica", 12) - - text_object.textLine(line) + text_object.textLine(line) c.drawText(text_object) c.save() diff --git a/src/linkedIn_job_manager.py b/src/linkedIn_job_manager.py index 8be9f02..9308708 100644 --- a/src/linkedIn_job_manager.py +++ b/src/linkedIn_job_manager.py @@ -47,10 +47,10 @@ class LinkedInJobManager: def set_parameters(self, parameters): logger.debug("Setting parameters for LinkedInJobManager") self.company_blacklist = parameters.get('company_blacklist', []) or [] - self.title_blacklist = parameters.get('title_blacklist', []) or [] + self.title_blacklist = parameters.get('titleBlacklist', []) or [] self.positions = parameters.get('positions', []) self.locations = parameters.get('locations', []) - self.apply_once_at_company = parameters.get('apply_once_at_company', False) + self.apply_once_at_company = parameters.get('applyOnceAtCompany', False) self.base_search_url = self.get_base_search_url(parameters) self.seen_jobs = [] @@ -272,7 +272,7 @@ class LinkedInJobManager: logger.debug(f"Applicants text found: {applicants_text}") # Extract numeric digits from the text (e.g., "70 applicants" -> "70") - applicants_count = ''.join([char for char in str(applicants_text) if char.isdigit()]) + applicants_count = ''.join(filter(str.isdigit, applicants_text)) logger.debug(f"Extracted applicants count: {applicants_count}") if applicants_count: @@ -370,7 +370,7 @@ class LinkedInJobManager: url_parts = [] if parameters['remote']: url_parts.append("f_CF=f_WRA") - experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experience_level', {}).items()) if + experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experienceLevel', {}).items()) if v] if experience_levels: url_parts.append(f"f_E={','.join(experience_levels)}") @@ -429,6 +429,7 @@ class LinkedInJobManager: link_seen = link in self.seen_jobs is_blacklisted = title_blacklisted or company_blacklisted or link_seen logger.debug("Job blacklisted status: %s", is_blacklisted) + return is_blacklisted return title_blacklisted or company_blacklisted or link_seen diff --git a/src/utils.py b/src/utils.py index e8b8429..f4e4d4a 100644 --- a/src/utils.py +++ b/src/utils.py @@ -179,8 +179,3 @@ def printyellow(text): reset = "\033[0m" logger.debug("Printing text in yellow: %s", text) print(f"{yellow}{text}{reset}") - - -def stringWidth(text, font, font_size): - bbox = font.getbbox(text) - return bbox[2] - bbox[0]