Revert "fixed some issues"

This reverts commit 6540bbbb40.
This commit is contained in:
queukat 2024-09-09 23:39:12 +03:00
parent 6540bbbb40
commit b554357be9
9 changed files with 77 additions and 127 deletions

View file

@ -1,6 +1,6 @@
remote: [true/false]
experience_level:
experienceLevel:
internship: [true/false]
entry: [true/false]
associate: [true/false]
@ -31,7 +31,7 @@ locations:
- Country1
- Country2
apply_once_at_company: [ true/false]
applyOnceAtCompany: [true/false]
distance: 100
@ -39,8 +39,7 @@ company_blacklist:
- Company1
- Company2
title_blacklist:
titleBlacklist:
- word1
- word2

View file

@ -1,6 +1,6 @@
remote: true
experience_level:
experienceLevel:
internship: true
entry: true
associate: true
@ -29,15 +29,15 @@ positions:
locations:
- USA
apply_once_at_company: [true/false]
applyOnceAtCompany: [true/false]
distance: 100
company_blacklist:
companyBlacklist:
- Noir
- Crossover
title_blacklist:
titleBlacklist:
llm_model_type: openai
llm_model: 'gpt-4o'

63
main.py
View file

@ -7,9 +7,9 @@ import click
from selenium import webdriver
from selenium.webdriver.chrome.service import Service as ChromeService
from webdriver_manager.chrome import ChromeDriverManager
from selenium.common.exceptions import WebDriverException
from lib_resume_builder_AIHawk import Resume, StyleManager, FacadeManager, ResumeGenerator
from src.utils import chrome_browser_options
from selenium.common.exceptions import WebDriverException, TimeoutException
from lib_resume_builder_AIHawk import Resume,StyleManager,FacadeManager,ResumeGenerator
from src.utils import chromeBrowserOptions
from src.gpt import GPTAnswerer
from src.linkedIn_authenticator import LinkedInAuthenticator
from src.linkedIn_bot_facade import LinkedInBotFacade
@ -19,11 +19,9 @@ from src.job_application_profile import JobApplicationProfile
# Suppress stderr
sys.stderr = open(os.devnull, 'w')
class ConfigError(Exception):
pass
class ConfigValidator:
@staticmethod
def validate_email(email: str) -> bool:
@ -39,36 +37,36 @@ class ConfigValidator:
except FileNotFoundError:
raise ConfigError(f"File not found: {yaml_path}")
def validate_config(config_yaml_path: Path) -> dict:
parameters = ConfigValidator.validate_yaml_file(config_yaml_path)
required_keys = {
'remote': bool,
'experience_level': dict,
'experienceLevel': dict,
'jobTypes': dict,
'date': dict,
'positions': list,
'locations': list,
'distance': int,
'company_blacklist': list,
'title_blacklist': list
'companyBlacklist': list,
'titleBlacklist': list
}
for key, expected_type in required_keys.items():
if key not in parameters:
if key in ['company_blacklist', 'title_blacklist']:
if key in ['companyBlacklist', 'titleBlacklist']:
parameters[key] = []
else:
raise ConfigError(f"Missing or invalid key '{key}' in config file {config_yaml_path}")
elif not isinstance(parameters[key], expected_type):
if key in ['company_blacklist', 'title_blacklist'] and parameters[key] is None:
if key in ['companyBlacklist', 'titleBlacklist'] and parameters[key] is None:
parameters[key] = []
else:
raise ConfigError(
f"Invalid type for key '{key}' in config file {config_yaml_path}. Expected {expected_type}.")
raise ConfigError(f"Invalid type for key '{key}' in config file {config_yaml_path}. Expected {expected_type}.")
experience_levels = ['internship', 'entry', 'associate', 'mid-senior level', 'director', 'executive']
for level in experience_levels:
if not isinstance(parameters['experience_level'].get(level), bool):
if not isinstance(parameters['experienceLevel'].get(level), bool):
raise ConfigError(f"Experience level '{level}' must be a boolean in config file {config_yaml_path}")
job_types = ['full-time', 'contract', 'part-time', 'temporary', 'internship', 'other', 'volunteer']
@ -88,10 +86,9 @@ class ConfigValidator:
approved_distances = {0, 5, 10, 25, 50, 100}
if parameters['distance'] not in approved_distances:
raise ConfigError(
f"Invalid distance value in config file {config_yaml_path}. Must be one of: {approved_distances}")
raise ConfigError(f"Invalid distance value in config file {config_yaml_path}. Must be one of: {approved_distances}")
for blacklist in ['company_blacklist', 'title_blacklist']:
for blacklist in ['companyBlacklist', 'titleBlacklist']:
if not isinstance(parameters.get(blacklist), list):
raise ConfigError(f"'{blacklist}' must be a list in config file {config_yaml_path}")
if parameters[blacklist] is None:
@ -99,6 +96,8 @@ class ConfigValidator:
return parameters
@staticmethod
def validate_secrets(secrets_yaml_path: Path) -> tuple:
secrets = ConfigValidator.validate_yaml_file(secrets_yaml_path)
@ -114,13 +113,10 @@ class ConfigValidator:
raise ConfigError(f"Password cannot be empty in secrets file {secrets_yaml_path}.")
return secrets['email'], str(secrets['password']), secrets['llm_api_key']
class FileManager:
@staticmethod
def find_file(name_containing: str, with_extension: str, at_path: Path) -> Path:
return next((file for file in at_path.iterdir() if
name_containing.lower() in file.name.lower() and file.suffix.lower() == with_extension.lower()),
None)
return next((file for file in at_path.iterdir() if name_containing.lower() in file.name.lower() and file.suffix.lower() == with_extension.lower()), None)
@staticmethod
def validate_data_folder(app_data_folder: Path) -> tuple:
@ -135,9 +131,7 @@ class FileManager:
output_folder = app_data_folder / 'output'
output_folder.mkdir(exist_ok=True)
return (
app_data_folder / 'secrets.yaml', app_data_folder / 'config.yaml', app_data_folder / 'plain_text_resume.yaml',
output_folder)
return (app_data_folder / 'secrets.yaml', app_data_folder / 'config.yaml', app_data_folder / 'plain_text_resume.yaml', output_folder)
@staticmethod
def file_paths_to_dict(resume_file: Path | None, plain_text_resume_file: Path) -> dict:
@ -153,16 +147,14 @@ class FileManager:
return result
def init_browser() -> webdriver.Chrome:
try:
options = chrome_browser_options()
options = chromeBrowserOptions()
service = ChromeService(ChromeDriverManager().install())
return webdriver.Chrome(service=service, options=options)
except Exception as e:
raise RuntimeError(f"Failed to initialize browser: {str(e)}")
def create_and_run_bot(email, password, parameters, llm_api_key):
try:
style_manager = StyleManager()
@ -170,8 +162,7 @@ def create_and_run_bot(email, password, parameters, llm_api_key):
with open(parameters['uploads']['plainTextResume'], "r", encoding='utf-8') as file:
plain_text_resume = file.read()
resume_object = Resume(plain_text_resume)
resume_generator_manager = FacadeManager(llm_api_key, style_manager, resume_generator, resume_object,
Path("data_folder/output"))
resume_generator_manager = FacadeManager(llm_api_key, style_manager, resume_generator, resume_object, Path("data_folder/output"))
os.system('cls' if os.name == 'nt' else 'clear')
resume_generator_manager.choose_style()
os.system('cls' if os.name == 'nt' else 'clear')
@ -196,8 +187,7 @@ def create_and_run_bot(email, password, parameters, llm_api_key):
@click.command()
@click.option('--resume', type=click.Path(exists=True, file_okay=True, dir_okay=False, path_type=Path),
help="Path to the resume PDF file")
@click.option('--resume', type=click.Path(exists=True, file_okay=True, dir_okay=False, path_type=Path), help="Path to the resume PDF file")
def main(resume: Path = None):
try:
data_folder = Path("data_folder")
@ -212,24 +202,19 @@ def main(resume: Path = None):
create_and_run_bot(email, password, parameters, llm_api_key)
except ConfigError as ce:
print(f"Configuration error: {str(ce)}")
print(
"Refer to the configuration guide for troubleshooting: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
print("Refer to the configuration guide for troubleshooting: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
except FileNotFoundError as fnf:
print(f"File not found: {str(fnf)}")
print("Ensure all required files are present in the data folder.")
print(
"Refer to the file setup guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
print("Refer to the file setup guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
except RuntimeError as re:
print(f"Runtime error: {str(re)}")
print(
"Refer to the configuration and troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
print("Refer to the configuration and troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
except Exception as e:
print(f"An unexpected error occurred: {str(e)}")
print(
"Refer to the general troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
print("Refer to the general troubleshooting guide: https://github.com/feder-cr/LinkedIn_AIHawk_automatic_job_application/blob/main/readme.md#configuration")
if __name__ == "__main__":
main()

View file

@ -14,8 +14,3 @@ click
git+https://github.com/feder-cr/lib_resume_builder_AIHawk.git
linkedin-api
pdfminer.six==20221105
inputimeout==1.0.4
langchain-ollama==0.1.3
langchain-anthropic==0.1.3
jsonschema==4.23.0
jsonschema-specifications==2023.12.1

View file

@ -7,17 +7,14 @@ import re
from jsonschema import validate, ValidationError
from pdfminer.high_level import extract_text
def load_yaml(file_path: str) -> Dict[str, Any]:
with open(file_path, 'r') as file:
return yaml.safe_load(file)
def load_resume_text(file_path: str) -> str:
with open(file_path, 'r') as file:
return file.read()
def get_api_key() -> str:
secrets_path = os.path.join('data_folder', 'secrets.yaml')
if not os.path.exists(secrets_path):
@ -34,7 +31,6 @@ def get_api_key() -> str:
return api_key
def generate_yaml_from_resume(resume_text: str, schema: Dict[str, Any], api_key: str) -> str:
client = OpenAI(api_key=api_key)
@ -87,8 +83,7 @@ def generate_yaml_from_resume(resume_text: str, schema: Dict[str, Any], api_key:
response = client.chat.completions.create(
model="gpt-4o-mini",
messages=[
{"role": "system",
"content": "You are a helpful assistant that generates structured YAML content from resume files, paying close attention to format requirements and schema structure."},
{"role": "system", "content": "You are a helpful assistant that generates structured YAML content from resume files, paying close attention to format requirements and schema structure."},
{"role": "user", "content": prompt}
],
temperature=0.5,
@ -103,12 +98,10 @@ def generate_yaml_from_resume(resume_text: str, schema: Dict[str, Any], api_key:
else:
raise ValueError("YAML content not found in the expected format")
def save_yaml(data: str, output_file: str):
with open(output_file, 'w') as file:
file.write(data)
def validate_yaml(yaml_content: str, schema: Dict[str, Any]) -> Dict[str, Any]:
try:
yaml_dict = yaml.safe_load(yaml_content)
@ -117,7 +110,6 @@ def validate_yaml(yaml_content: str, schema: Dict[str, Any]) -> Dict[str, Any]:
except ValidationError as e:
return {"valid": False, "errors": str(e)}
def generate_report(validation_result: Dict[str, Any], output_file: str):
report = f"Validation Report for {output_file}\n"
report += "=" * 40 + "\n"
@ -129,14 +121,11 @@ def generate_report(validation_result: Dict[str, Any], output_file: str):
print(report)
def pdf_to_text(pdf_path: str) -> str:
return extract_text(pdf_path)
def main():
parser = argparse.ArgumentParser(
description="Generate a resume YAML file from a PDF or text resume using OpenAI API")
parser = argparse.ArgumentParser(description="Generate a resume YAML file from a PDF or text resume using OpenAI API")
parser.add_argument("--input", required=True, help="Path to the input resume file (PDF or TXT)")
parser.add_argument("--output", default="data_folder/plain_text_resume.yaml", help="Path to the output YAML file")
args = parser.parse_args()
@ -167,6 +156,5 @@ def main():
except Exception as e:
print(f"An error occurred: {e}")
if __name__ == "__main__":
main()

View file

@ -6,7 +6,8 @@ import time
from abc import ABC, abstractmethod
from datetime import datetime
from pathlib import Path
from typing import Dict, List, Union
from typing import Dict, List
from typing import Union
import httpx
from Levenshtein import distance
@ -37,7 +38,7 @@ class OpenAIModel(AIModel):
def invoke(self, prompt: str) -> str:
print("invoke in openai")
response = self.model.invoke(prompt)
return response.content
return response
class ClaudeModel(AIModel):
@ -48,7 +49,7 @@ class ClaudeModel(AIModel):
def invoke(self, prompt: str) -> str:
response = self.model.invoke(prompt)
return response.content
return response
class OllamaModel(AIModel):
@ -58,14 +59,14 @@ class OllamaModel(AIModel):
def invoke(self, prompt: str) -> str:
response = self.model.invoke(prompt)
return response.content
return response
class AIAdapter:
def __init__(self, config: dict, api_key: str):
self.model = self._create_model(config, api_key)
def _create_model(self, config: dict, api_key: str) -> Union[OpenAIModel, OllamaModel, ClaudeModel]:
def _create_model(self, config: dict, api_key: str) -> AIModel:
llm_model_type = config['llm_model_type']
llm_model = config['llm_model']
llm_api_url = config['llm_api_url']
@ -78,7 +79,7 @@ class AIAdapter:
elif llm_model_type == "ollama":
return OllamaModel(api_key, llm_model, llm_api_url)
else:
raise ValueError(f"Unsupported model type: {llm_model_type}")
raise ValueError(f"Unsupported model type: {model_type}")
def invoke(self, prompt: str) -> str:
return self.model.invoke(prompt)
@ -108,34 +109,25 @@ class LLMLogger:
logger.debug("Prompts are of type StringPromptValue")
prompts = prompts.text
logger.debug("Prompts converted to text: %s", prompts)
elif isinstance(prompts, dict):
logger.debug("Prompts are of type dict")
elif isinstance(prompts, Dict):
logger.debug("Prompts are of type Dict")
try:
if "messages" in prompts:
logger.debug("Prompts contain 'messages' key")
prompts = {
f"prompt_{i + 1}": prompt["content"]
for i, prompt in enumerate(prompts["messages"])
}
logger.debug("Prompts converted to dictionary: %s", prompts)
else:
logger.debug("Prompts dictionary does not contain 'messages' key")
prompts = {
f"prompt_{i + 1}": prompt.content
for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts converted to dictionary: %s", prompts)
except Exception as e:
logger.error("Error converting prompts to dictionary: %s", str(e))
raise
else:
logger.debug("Prompts are of unknown type, attempting default conversion")
try:
if hasattr(prompts, "messages"):
logger.debug("Prompts have 'messages' attribute")
prompts = {
f"prompt_{i + 1}": prompt.content
for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts converted to dictionary using default method: %s", prompts)
else:
logger.error("Prompts do not have 'messages' attribute, and default conversion failed")
raise ValueError("Prompts structure is not supported.")
prompts = {
f"prompt_{i + 1}": prompt.content
for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts converted to dictionary using default method: %s", prompts)
except Exception as e:
logger.error("Error converting prompts using default method: %s", str(e))
raise
@ -299,7 +291,7 @@ class GPTAnswerer:
def __init__(self, config, llm_api_key):
self.ai_adapter = AIAdapter(config, llm_api_key)
self.llm_cheap = LoggerChatModel(self.ai_adapter.model)
self.llm_cheap = LoggerChatModel(self.ai_adapter)
@property
def job_description(self):

View file

@ -5,8 +5,7 @@ import random
import re
import time
import traceback
from pathlib import Path
from typing import List, Optional, Any, Tuple, Set
from typing import List, Optional, Any, Tuple
from httpx import HTTPStatusError
from reportlab.lib.pagesizes import A4
@ -24,13 +23,11 @@ from src.utils import logger
class LinkedInEasyApplier:
def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: Set[Tuple[str, str, str]],
def __init__(self, driver: Any, resume_dir: Optional[str], set_old_answers: List[Tuple[str, str, str]],
gpt_answerer: Any, resume_generator_manager):
logger.debug("Initializing LinkedInEasyApplier")
if resume_dir is None or not os.path.exists(resume_dir):
resume_dir = None
else:
resume_dir = Path(resume_dir)
self.driver = driver
self.resume_path = resume_dir
self.set_old_answers = set_old_answers
@ -541,19 +538,17 @@ class LinkedInEasyApplier:
lines = split_text_by_width(cover_letter_text, "Helvetica", 12, max_width)
line_height = 14
max_lines_per_page = int(available_height // line_height)
for line in lines:
text_height = text_object.getY()
if text_height > bottom_margin:
text_object.textLine(line)
else:
if text_height - line_height < bottom_margin:
c.drawText(text_object)
c.showPage()
text_object = c.beginText(50, page_height - 50)
text_object.setFont("Helvetica", 12)
text_object.textLine(line)
text_object.textLine(line)
c.drawText(text_object)
c.save()

View file

@ -47,10 +47,10 @@ class LinkedInJobManager:
def set_parameters(self, parameters):
logger.debug("Setting parameters for LinkedInJobManager")
self.company_blacklist = parameters.get('company_blacklist', []) or []
self.title_blacklist = parameters.get('title_blacklist', []) or []
self.title_blacklist = parameters.get('titleBlacklist', []) or []
self.positions = parameters.get('positions', [])
self.locations = parameters.get('locations', [])
self.apply_once_at_company = parameters.get('apply_once_at_company', False)
self.apply_once_at_company = parameters.get('applyOnceAtCompany', False)
self.base_search_url = self.get_base_search_url(parameters)
self.seen_jobs = []
@ -272,7 +272,7 @@ class LinkedInJobManager:
logger.debug(f"Applicants text found: {applicants_text}")
# Extract numeric digits from the text (e.g., "70 applicants" -> "70")
applicants_count = ''.join([char for char in str(applicants_text) if char.isdigit()])
applicants_count = ''.join(filter(str.isdigit, applicants_text))
logger.debug(f"Extracted applicants count: {applicants_count}")
if applicants_count:
@ -370,7 +370,7 @@ class LinkedInJobManager:
url_parts = []
if parameters['remote']:
url_parts.append("f_CF=f_WRA")
experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experience_level', {}).items()) if
experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experienceLevel', {}).items()) if
v]
if experience_levels:
url_parts.append(f"f_E={','.join(experience_levels)}")
@ -429,6 +429,7 @@ class LinkedInJobManager:
link_seen = link in self.seen_jobs
is_blacklisted = title_blacklisted or company_blacklisted or link_seen
logger.debug("Job blacklisted status: %s", is_blacklisted)
return is_blacklisted
return title_blacklisted or company_blacklisted or link_seen

View file

@ -179,8 +179,3 @@ def printyellow(text):
reset = "\033[0m"
logger.debug("Printing text in yellow: %s", text)
print(f"{yellow}{text}{reset}")
def stringWidth(text, font, font_size):
bbox = font.getbbox(text)
return bbox[2] - bbox[0]