add logs and some bugs fixes

This commit is contained in:
queukat 2024-09-06 01:53:11 +03:00
parent 0c4ae18064
commit 525e794f8b
5 changed files with 565 additions and 224 deletions

View file

@ -2,20 +2,21 @@ import json
import os
import re
import textwrap
import time
from datetime import datetime
from typing import Dict, List
from functools import wraps
from pathlib import Path
from typing import Dict, List
import httpx
from Levenshtein import distance
from dotenv import load_dotenv
from httpx import HTTPStatusError
from langchain_core.messages.ai import AIMessage
from langchain_core.output_parsers import StrOutputParser
from langchain_core.prompt_values import StringPromptValue
from langchain_core.prompts import ChatPromptTemplate
from langchain_openai import ChatOpenAI
from Levenshtein import distance
import time
from functools import wraps
from openai import RateLimitError, OpenAIError, APIError
import src.strings as strings
from src.utils import logger
@ -42,156 +43,209 @@ def global_rate_limiter(min_interval):
return decorator
def parse_wait_time_from_error_message(error_message: str) -> int:
logger.debug("Parsing wait time from error message: %s", error_message)
match = re.search(r"Please try again in (\d+)([smhd])", error_message)
if match:
value, unit = int(match.group(1)), match.group(2)
logger.debug("Extracted wait time: %d %s", value, unit)
if unit == 's':
return value
elif unit == 'm':
return value * 60
elif unit == 'h':
return value * 3600
elif unit == 'd':
return value * 86400
logger.debug("Default wait time applied: 30 seconds")
return 30
class LLMLogger:
def __init__(self, llm: ChatOpenAI):
logger.debug("Initializing LLMLogger with LLM: %s", llm)
self.llm = llm
logger.debug("LLMLogger initialized with LLM: %s", llm)
logger.debug("LLMLogger successfully initialized with LLM: %s", llm)
@staticmethod
def log_request(prompts, parsed_reply: Dict[str, Dict]):
logger.debug("Logging request with prompts: %s", prompts)
calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json")
logger.debug("Starting log_request method")
logger.debug("Prompts received: %s", prompts)
logger.debug("Parsed reply received: %s", parsed_reply)
# Определяем путь к файлу для записи логов
try:
calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json")
logger.debug("Logging path determined: %s", calls_log)
except Exception as e:
logger.error("Error determining the log path: %s", str(e))
raise
# Преобразование prompts в текст или словарь
if isinstance(prompts, StringPromptValue):
logger.debug("Prompts are of type StringPromptValue")
prompts = prompts.text
logger.debug("Prompts converted to text: %s", prompts)
elif isinstance(prompts, Dict):
# Convert prompts to a dictionary if they are not in the expected format
prompts = {
f"prompt_{i+1}": prompt.content
for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts are of type Dict")
try:
prompts = {
f"prompt_{i+1}": prompt.content
for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts converted to dictionary: %s", prompts)
except Exception as e:
logger.error("Error converting prompts to dictionary: %s", str(e))
raise
else:
prompts = {
f"prompt_{i+1}": prompt.content
for i, prompt in enumerate(prompts.messages)
logger.debug("Prompts are of unknown type, attempting default conversion")
try:
prompts = {
f"prompt_{i+1}": prompt.content
for i, prompt in enumerate(prompts.messages)
}
logger.debug("Prompts converted to dictionary using default method: %s", prompts)
except Exception as e:
logger.error("Error converting prompts using default method: %s", str(e))
raise
# Получение текущего времени
try:
current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
logger.debug("Current time obtained: %s", current_time)
except Exception as e:
logger.error("Error obtaining current time: %s", str(e))
raise
# Извлечение информации о токенах
try:
token_usage = parsed_reply["usage_metadata"]
output_tokens = token_usage["output_tokens"]
input_tokens = token_usage["input_tokens"]
total_tokens = token_usage["total_tokens"]
logger.debug("Token usage - Input: %d, Output: %d, Total: %d", input_tokens, output_tokens, total_tokens)
except KeyError as e:
logger.error("KeyError in parsed_reply structure: %s", str(e))
raise
# Извлечение имени модели
try:
model_name = parsed_reply["response_metadata"]["model_name"]
logger.debug("Model name: %s", model_name)
except KeyError as e:
logger.error("KeyError in response_metadata: %s", str(e))
raise
# Вычисление стоимости использования API
try:
prompt_price_per_token = 0.00000015
completion_price_per_token = 0.0000006
total_cost = (input_tokens * prompt_price_per_token) + (output_tokens * completion_price_per_token)
logger.debug("Total cost calculated: %f", total_cost)
except Exception as e:
logger.error("Error calculating total cost: %s", str(e))
raise
# Формирование записи лога
try:
log_entry = {
"model": model_name,
"time": current_time,
"prompts": prompts,
"replies": parsed_reply["content"], # Контент ответа
"total_tokens": total_tokens,
"input_tokens": input_tokens,
"output_tokens": output_tokens,
"total_cost": total_cost,
}
logger.debug("Log entry created: %s", log_entry)
except KeyError as e:
logger.error("Error creating log entry: missing key %s in parsed_reply", str(e))
raise
current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
logger.debug("Current time: %s", current_time)
# Extract token usage details from the response
token_usage = parsed_reply["usage_metadata"]
output_tokens = token_usage["output_tokens"]
input_tokens = token_usage["input_tokens"]
total_tokens = token_usage["total_tokens"]
logger.debug("Token usage - Input: %d, Output: %d, Total: %d", input_tokens, output_tokens, total_tokens)
model_name = parsed_reply["response_metadata"]["model_name"]
prompt_price_per_token = 0.00000015
completion_price_per_token = 0.0000006
# Calculate the total cost of the API call
total_cost = (input_tokens * prompt_price_per_token) + (
output_tokens * completion_price_per_token
)
logger.debug("Total cost calculated: %f", total_cost)
log_entry = {
"model": model_name,
"time": current_time,
"prompts": prompts,
"replies": parsed_reply["content"], # Response content
"total_tokens": total_tokens,
"input_tokens": input_tokens,
"output_tokens": output_tokens,
"total_cost": total_cost,
}
logger.debug("Log entry created: %s", log_entry)
with open(calls_log, "a", encoding="utf-8") as f:
json_string = json.dumps(log_entry, ensure_ascii=False, indent=4)
f.write(json_string + "\n")
logger.debug("Log entry written to file: %s", calls_log)
# Запись в файл
try:
with open(calls_log, "a", encoding="utf-8") as f:
json_string = json.dumps(log_entry, ensure_ascii=False, indent=4)
f.write(json_string + "\n")
logger.debug("Log entry written to file: %s", calls_log)
except Exception as e:
logger.error("Error writing log entry to file: %s", str(e))
raise
class LoggerChatModel:
def __init__(self, llm: ChatOpenAI):
logger.debug("Initializing LoggerChatModel with LLM: %s", llm)
self.llm = llm
logger.debug("LoggerChatModel initialized with LLM: %s", llm)
logger.debug("LoggerChatModel successfully initialized with LLM: %s", llm)
def __call__(self, messages: List[Dict[str, str]]) -> str:
logger.debug("Calling LoggerChatModel with messages: %s", messages)
while True:
logger.debug("Entering __call__ method with messages: %s", messages)
while True: # Бесконечный цикл до успешного выполнения
try:
# Попытка вызвать модель
reply = self.llm(messages)
logger.debug("Model reply received: %s", reply)
logger.debug("Attempting to call the LLM with messages")
reply = self.llm(messages) # Вызов LLM
logger.debug("LLM response received: %s", reply)
parsed_reply = self.parse_llmresult(reply)
logger.debug("Parsed LLM reply: %s", parsed_reply)
# Логируем запрос и ответ
LLMLogger.log_request(prompts=messages, parsed_reply=parsed_reply)
return reply
except RateLimitError as err:
# Handle RateLimitError
wait_time = self.parse_wait_time_from_error_message(str(err))
logger.warning("Rate limit exceeded. Waiting for %d seconds before retrying...", wait_time)
time.sleep(wait_time)
logger.debug("Request successfully logged")
return reply # Возвращаем корректный ответ, завершаем цикл
except httpx.HTTPStatusError as e:
logger.error("HTTPStatusError encountered: %s", str(e))
if e.response.status_code == 429:
retry_after = e.response.headers.get('retry-after')
retry_after_ms = e.response.headers.get('retry-after-ms')
if retry_after:
wait_time = int(retry_after)
logger.warning("Rate limit exceeded. Waiting for %d seconds before retrying (extracted from 'retry-after' header)...", wait_time)
time.sleep(wait_time)
elif retry_after_ms:
wait_time = int(retry_after_ms) / 1000.0
logger.warning("Rate limit exceeded. Waiting for %f seconds before retrying (extracted from 'retry-after-ms' header)...", wait_time)
time.sleep(wait_time)
else:
wait_time = 30 # Время ожидания по умолчанию
logger.warning("'retry-after' header not found. Waiting for %d seconds before retrying (default)...", wait_time)
time.sleep(wait_time)
else:
logger.error("HTTP error occurred with status code: %d, waiting 30 seconds before retrying", e.response.status_code)
time.sleep(30)
except Exception as e:
logger.error("Unexpected error occurred: %s", str(e))
raise
logger.info("Waiting for 30 seconds before retrying due to an unexpected error.")
time.sleep(30)
continue # Продолжаем цикл
def parse_llmresult(self, llmresult: AIMessage) -> Dict[str, Dict]:
logger.debug("Parsing LLM result: %s", llmresult)
content = llmresult.content
response_metadata = llmresult.response_metadata
id_ = llmresult.id
usage_metadata = llmresult.usage_metadata
parsed_result = {
"content": content,
"response_metadata": {
"model_name": response_metadata.get("model_name", ""),
"system_fingerprint": response_metadata.get("system_fingerprint", ""),
"finish_reason": response_metadata.get("finish_reason", ""),
"logprobs": response_metadata.get("logprobs", None),
},
"id": id_,
"usage_metadata": {
"input_tokens": usage_metadata.get("input_tokens", 0),
"output_tokens": usage_metadata.get("output_tokens", 0),
"total_tokens": usage_metadata.get("total_tokens", 0),
},
}
logger.debug("Parsed LLM result: %s", parsed_result)
return parsed_result
def parse_wait_time_from_error_message(self, error_message: str) -> int:
logger.debug("Parsing wait time from error message: %s", error_message)
match = re.search(r"Please try again in (\d+)([smhd])", error_message)
if match:
value, unit = match.groups()
value = int(value)
logger.debug("Extracted wait time: %d %s", value, unit)
if unit == "s":
return value
elif unit == "m":
return value * 60
elif unit == "h":
return value * 3600
elif unit == "d":
return value * 86400
logger.debug("Default wait time applied: 30 seconds")
return 30
# Извлечение данных из ответа
try:
content = llmresult.content
response_metadata = llmresult.response_metadata
id_ = llmresult.id
usage_metadata = llmresult.usage_metadata
parsed_result = {
"content": content,
"response_metadata": {
"model_name": response_metadata.get("model_name", ""),
"system_fingerprint": response_metadata.get("system_fingerprint", ""),
"finish_reason": response_metadata.get("finish_reason", ""),
"logprobs": response_metadata.get("logprobs", None),
},
"id": id_,
"usage_metadata": {
"input_tokens": usage_metadata.get("input_tokens", 0),
"output_tokens": usage_metadata.get("output_tokens", 0),
"total_tokens": usage_metadata.get("total_tokens", 0),
},
}
logger.debug("Parsed LLM result successfully: %s", parsed_result)
return parsed_result
except KeyError as e:
logger.error("KeyError while parsing LLM result: missing key %s", str(e))
raise # Повторно выбрасываем исключение, чтобы оно обрабатывалось выше
except Exception as e:
logger.error("Unexpected error while parsing LLM result: %s", str(e))
raise
class GPTAnswerer:
@ -239,7 +293,7 @@ class GPTAnswerer:
logger.debug("Setting job application profile: %s", job_application_profile)
self.job_application_profile = job_application_profile
@global_rate_limiter(25)
#@global_rate_limiter(25)
def summarize_job_description(self, text: str) -> str:
logger.debug("Summarizing job description: %s", text)
strings.summarize_prompt_template = self._preprocess_template_string(
@ -256,7 +310,7 @@ class GPTAnswerer:
prompt = ChatPromptTemplate.from_template(template)
return prompt | self.llm_cheap | StrOutputParser()
@global_rate_limiter(25)
#@global_rate_limiter(25)
def answer_question_textual_wide_range(self, question: str) -> str:
logger.debug("Answering textual question: %s", question)
chains = {
@ -384,7 +438,7 @@ class GPTAnswerer:
logger.debug("Question answered: %s", output)
return output
@global_rate_limiter(25)
#@global_rate_limiter(25)
def answer_question_numeric(self, question: str, default_experience: int = 3) -> int:
logger.debug("Answering numeric question: %s", question)
func_template = self._preprocess_template_string(strings.numeric_question_template)
@ -410,7 +464,7 @@ class GPTAnswerer:
logger.error("No numbers found in the string")
raise ValueError("No numbers found in the string")
@global_rate_limiter(25)
#@global_rate_limiter(25)
def answer_question_from_options(self, question: str, options: list[str]) -> str:
logger.debug("Answering question from options: %s", question)
func_template = self._preprocess_template_string(strings.options_template)
@ -422,11 +476,11 @@ class GPTAnswerer:
logger.debug("Best option determined: %s", best_option)
return best_option
@global_rate_limiter(25)
#@global_rate_limiter(25)
def resume_or_cover(self, phrase: str) -> str:
logger.debug("Determining if phrase refers to resume or cover letter: %s", phrase)
prompt_template = """
Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. Do not provide any additional information or explanations.
Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. If the phrase contains only the word 'upload', consider it as 'cover'. Do not provide any additional information or explanations.
phrase: {phrase}
"""

View file

@ -25,7 +25,14 @@ class LinkedInAuthenticator:
logger.info("Starting Chrome browser to log in to LinkedIn.")
self.driver.get('https://www.linkedin.com/feed')
self.wait_for_page_load()
if not self.is_logged_in():
time.sleep(3)
if self.is_logged_in():
logger.info("User is already logged in. Skipping login process.")
return
else:
logger.info("User is not logged in. Proceeding with login.")
self.handle_login()
def handle_login(self):
@ -82,12 +89,12 @@ class LinkedInAuthenticator:
print("Security check not completed. Please try again later.")
def is_logged_in(self):
target_url = 'https://www.linkedin.com/feed'
# Navigate to the target URL if not already there
if self.driver.current_url != target_url:
logger.debug("Navigating to target URL: %s", target_url)
self.driver.get(target_url)
# target_url = 'https://www.linkedin.com/feed'
#
# # Navigate to the target URL if not already there
# if self.driver.current_url != target_url:
# logger.debug("Navigating to target URL: %s", target_url)
# self.driver.get(target_url)
try:
# Increase the wait time for the page elements to load
@ -98,38 +105,29 @@ class LinkedInAuthenticator:
# Check for the presence of the "Start a post" button
buttons = self.driver.find_elements(By.CLASS_NAME, 'share-box-feed-entry__trigger')
if any(button.text.strip() == 'Start a post' for button in buttons):
logger.info("User is already logged in.")
logger.debug("Found %d 'Start a post' buttons", len(buttons))
try:
# Wait for the profile picture and name to load
profile_img = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.XPATH, "//img[contains(@alt, 'Photo of')]"))
)
profile_name = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.XPATH, "//div[@class='t-16 t-black t-bold']"))
)
# Выведем текст всех найденных кнопок в лог для диагностики
for i, button in enumerate(buttons):
logger.debug("Button %d text: %s", i + 1, button.text.strip())
if profile_img and profile_name:
logger.info("Profile picture found for user: %s", profile_name.text)
return True
except NoSuchElementException:
logger.warning("Profile picture or name not found.")
print("Profile picture or name not found.")
return False
except TimeoutException:
logger.warning("Profile picture or name took too long to load.")
print("Profile picture or name took too long to load.")
return False
if any(button.text.strip().lower() == 'start a post' for button in buttons):
logger.info("Found 'Start a post' button indicating user is logged in.")
return True
# Альтернативная проверка авторизации по наличию изображения профиля
profile_img_elements = self.driver.find_elements(By.XPATH, "//img[contains(@alt, 'Photo of')]")
if profile_img_elements:
logger.info("Profile image found. Assuming user is logged in.")
return True
logger.info("Did not find 'Start a post' button or profile image. User might not be logged in.")
return False
except TimeoutException:
logger.error("Page elements took too long to load or were not found.")
print("Page elements took too long to load or were not found.")
return False
return False
def wait_for_page_load(self, timeout=10):
try:
logger.debug("Waiting for page to load with timeout: %s seconds", timeout)

View file

@ -8,6 +8,9 @@ import time
import traceback
from datetime import date
from typing import List, Optional, Any, Tuple
from httpx import HTTPStatusError
from openai import RateLimitError
from reportlab.lib.pagesizes import letter
from reportlab.pdfgen import canvas
from selenium.common.exceptions import NoSuchElementException, TimeoutException
@ -57,54 +60,131 @@ class LinkedInEasyApplier:
def job_apply(self, job: Any):
logger.debug("Starting job application for job: %s", job)
self.driver.get(job.link)
time.sleep(random.uniform(3, 5))
# Открываем страницу с вакансией
try:
self.driver.get(job.link)
logger.debug("Navigated to job link: %s", job.link)
except Exception as e:
logger.error("Failed to navigate to job link: %s, error: %s", job.link, str(e))
raise
# Добавляем небольшую паузу для загрузки страницы
time.sleep(random.uniform(3, 5))
try:
# Поиск кнопки 'Easy Apply'
logger.debug("Searching for 'Easy Apply' button on job page")
easy_apply_button = self._find_easy_apply_button()
job.set_job_description(self._get_job_description())
job.set_recruiter_link(self._get_job_recruiter())
# Получаем описание вакансии
logger.debug("Retrieving job description")
job_description = self._get_job_description()
job.set_job_description(job_description)
logger.debug("Job description set: %s", job_description[:100]) # Логируем только первые 100 символов
# Получаем ссылку на рекрутера (если есть)
logger.debug("Retrieving recruiter link")
recruiter_link = self._get_job_recruiter()
job.set_recruiter_link(recruiter_link)
logger.debug("Recruiter link set: %s", recruiter_link)
# Действие: нажимаем на кнопку 'Easy Apply'
logger.debug("Attempting to click 'Easy Apply' button")
actions = ActionChains(self.driver)
actions.move_to_element(easy_apply_button).click().perform()
logger.debug("'Easy Apply' button clicked successfully")
# Передача информации о работе для дальнейшей обработки
logger.debug("Passing job information to GPT Answerer")
self.gpt_answerer.set_job(job)
# Заполнение формы подачи заявки
logger.debug("Filling out application form")
self._fill_application_form(job)
logger.debug("Job application process completed for job: %s", job)
except Exception:
logger.debug("Job application process completed successfully for job: %s", job)
except Exception as e:
# Захват и логирование полного traceback в случае ошибки
tb_str = traceback.format_exc()
logger.error("Failed to apply to job: %s", tb_str)
logger.error("Failed to apply to job: %s. Error traceback: %s", job, tb_str)
# Отмена заявки в случае ошибки
logger.debug("Discarding application due to failure")
self._discard_application()
raise Exception(f"Failed to apply to job! Original exception: \nTraceback:\n{tb_str}")
# Поднятие исключения с оригинальной ошибкой
raise Exception(f"Failed to apply to job! Original exception:\nTraceback:\n{tb_str}")
def _find_easy_apply_button(self) -> WebElement:
logger.debug("Searching for 'Easy Apply' button")
attempt = 0
# Список методов поиска кнопки
search_methods = [
{
'description': "find all 'Easy Apply' buttons using find_elements",
'find_elements': True, # Используем find_elements для поиска всех кнопок
'xpath': '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]'
},
{
'description': "'aria-label' containing 'Easy Apply to'",
'xpath': '//button[contains(@aria-label, "Easy Apply to")]'
},
{
'description': "button text search",
'xpath': '//button[contains(text(), "Easy Apply") or contains(text(), "Apply now")]'
}
]
while attempt < 2:
self._scroll_page()
try:
buttons = WebDriverWait(self.driver, 10).until(
EC.presence_of_all_elements_located(
(By.XPATH, '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]')
)
)
for index, _ in enumerate(buttons):
try:
button = WebDriverWait(self.driver, 10).until(
EC.element_to_be_clickable(
(By.XPATH, f'(//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")])[{index + 1}]')
)
)
logger.debug("Found and clicking 'Easy Apply' button")
return button
except Exception as e:
logger.warning("Failed to click 'Easy Apply' button on attempt %d: %s", attempt + 1, e)
except TimeoutException:
logger.warning("Timeout while searching for 'Easy Apply' button")
for method in search_methods:
try:
logger.debug(f"Attempting search using {method['description']}")
# Если метод использует find_elements
if method.get('find_elements'):
# Поиск всех кнопок "Easy Apply"
buttons = self.driver.find_elements(By.XPATH, method['xpath'])
if buttons:
for index, button in enumerate(buttons):
try:
# Проверка видимости и кликабельности каждой кнопки
WebDriverWait(self.driver, 10).until(EC.visibility_of(button))
WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button))
logger.debug(f"Found 'Easy Apply' button {index + 1}, attempting to click")
return button
except Exception as e:
logger.warning(f"Button {index + 1} found but not clickable: {e}")
else:
raise TimeoutException("No 'Easy Apply' buttons found")
else:
# Стандартный метод с WebDriverWait для одного элемента
button = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.XPATH, method['xpath']))
)
WebDriverWait(self.driver, 10).until(EC.visibility_of(button))
WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button))
logger.debug("Found 'Easy Apply' button, attempting to click")
return button
except TimeoutException:
logger.warning(f"Timeout during search using {method['description']}")
except Exception as e:
logger.warning(f"Failed to click 'Easy Apply' button using {method['description']} on attempt {attempt + 1}: {e}")
# Обновление страницы после первой неудачной попытки
if attempt == 0:
logger.debug("Refreshing page to retry finding 'Easy Apply' button")
self.driver.refresh()
time.sleep(random.randint(3, 5))
attempt += 1
logger.error("No clickable 'Easy Apply' button found after 2 attempts")
# Если не удалось найти кнопку, выводим HTML для отладки
page_source = self.driver.page_source
logger.error("No clickable 'Easy Apply' button found after 2 attempts. Page source:\n%s", page_source)
raise Exception("No clickable 'Easy Apply' button found")
def _get_job_description(self) -> str:
@ -136,10 +216,18 @@ class LinkedInEasyApplier:
hiring_team_section = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.XPATH, '//h2[text()="Meet the hiring team"]'))
)
recruiter_element = hiring_team_section.find_element(By.XPATH, './/following::a[contains(@href, "linkedin.com/in/")]')
recruiter_link = recruiter_element.get_attribute('href')
logger.debug("Job recruiter link retrieved successfully")
return recruiter_link
logger.debug("Hiring team section found")
recruiter_elements = hiring_team_section.find_elements(By.XPATH, './/following::a[contains(@href, "linkedin.com/in/")]')
if recruiter_elements:
recruiter_element = recruiter_elements[0]
recruiter_link = recruiter_element.get_attribute('href')
logger.debug("Job recruiter link retrieved successfully: %s", recruiter_link)
return recruiter_link
else:
logger.debug("No recruiter link found in the hiring team section")
return ""
except Exception as e:
logger.warning("Failed to retrieve recruiter information: %s", e)
return ""
@ -202,11 +290,19 @@ class LinkedInEasyApplier:
def fill_up(self, job) -> None:
logger.debug("Filling up form sections for job: %s", job)
easy_apply_content = self.driver.find_element(By.CLASS_NAME, 'jobs-easy-apply-content')
pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4')
for element in pb4_elements:
self._process_form_element(element, job)
# Используем WebDriverWait для ожидания элемента с классом 'jobs-easy-apply-content'
try:
easy_apply_content = WebDriverWait(self.driver, 10).until(
EC.presence_of_element_located((By.CLASS_NAME, 'jobs-easy-apply-content'))
)
# После нахождения 'jobs-easy-apply-content' ищем элементы с классом 'pb4'
pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4')
for element in pb4_elements:
self._process_form_element(element, job)
except Exception as e:
logger.error(f"Failed to find form elements: {e}")
def _process_form_element(self, element: WebElement, job) -> None:
logger.debug("Processing form element")
if self._is_upload_field(element):
@ -221,40 +317,114 @@ class LinkedInEasyApplier:
def _handle_upload_fields(self, element: WebElement, job) -> None:
logger.debug("Handling upload fields")
try:
show_more_button = self.driver.find_element(By.XPATH, "//button[contains(@aria-label, 'Show more resumes')]")
show_more_button.click()
logger.debug("Clicked 'Show more resumes' button")
except NoSuchElementException:
logger.debug("'Show more resumes' button not found, continuing...")
file_upload_elements = self.driver.find_elements(By.XPATH, "//input[@type='file']")
for element in file_upload_elements:
parent = element.find_element(By.XPATH, "..")
self.driver.execute_script("arguments[0].classList.remove('hidden')", element)
output = self.gpt_answerer.resume_or_cover(parent.text.lower())
if 'resume' in output:
logger.debug("Uploading resume")
if self.resume_path is not None and self.resume_path.resolve().is_file():
element.send_keys(str(self.resume_path.resolve()))
logger.debug(f"Resume uploaded from path: {self.resume_path.resolve()}")
else:
logger.debug("Resume path not found or invalid, generating new resume")
self._create_and_upload_resume(element, job)
elif 'cover' in output:
logger.debug("Uploading cover letter")
self._create_and_upload_cover_letter(element)
logger.debug("Finished handling upload fields")
def _create_and_upload_resume(self, element, job):
logger.debug("Creating and uploading resume")
folder_path = 'generated_cv'
os.makedirs(folder_path, exist_ok=True)
try:
timestamp = int(time.time())
file_path_pdf = os.path.join(folder_path, f"CV_{timestamp}.pdf")
logger.debug("Starting the process of creating and uploading resume.")
folder_path = 'generated_cv'
with open(file_path_pdf, "xb") as f: # gjcvjn
f.write(base64.b64decode(self.resume_generator_manager.pdf_base64(job_description_text=job.description)))
try:
if not os.path.exists(folder_path):
logger.debug(f"Creating directory at path: {folder_path}")
os.makedirs(folder_path, exist_ok=True)
except Exception as e:
logger.error(f"Failed to create directory: {folder_path}. Error: {e}")
raise
element.send_keys(os.path.abspath(file_path_pdf))
job.pdf_path = os.path.abspath(file_path_pdf)
time.sleep(2)
logger.debug("Resume created and uploaded successfully: %s", file_path_pdf)
except Exception:
tb_str = traceback.format_exc()
logger.error("Resume upload failed: %s", tb_str)
raise Exception(f"Upload failed: \nTraceback:\n{tb_str}")
while True:
try:
timestamp = int(time.time())
file_path_pdf = os.path.join(folder_path, f"CV_{timestamp}.pdf")
logger.debug(f"Generated file path for resume: {file_path_pdf}")
logger.debug(f"Generating resume for job: {job.title} at {job.company}")
resume_pdf_base64 = self.resume_generator_manager.pdf_base64(job_description_text=job.description)
with open(file_path_pdf, "xb") as f:
f.write(base64.b64decode(resume_pdf_base64))
logger.debug(f"Resume successfully generated and saved to: {file_path_pdf}")
break
except HTTPStatusError as e:
if e.response.status_code == 429:
retry_after = e.response.headers.get('retry-after')
retry_after_ms = e.response.headers.get('retry-after-ms')
if retry_after:
wait_time = int(retry_after)
logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...")
elif retry_after_ms:
wait_time = int(retry_after_ms) / 1000.0
logger.warning(f"Rate limit exceeded, waiting {wait_time} milliseconds before retrying...")
else:
wait_time = 20
logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...")
time.sleep(wait_time)
else:
logger.error(f"HTTP error: {e}")
raise
except Exception as e:
logger.error(f"Failed to generate resume: {e}")
tb_str = traceback.format_exc()
logger.error(f"Traceback: {tb_str}")
if "RateLimitError" in str(e):
logger.warning("Rate limit error encountered, retrying...")
time.sleep(20)
else:
raise
file_size = os.path.getsize(file_path_pdf)
max_file_size = 2 * 1024 * 1024 # 2 MB
logger.debug(f"Resume file size: {file_size} bytes")
if file_size > max_file_size:
logger.error(f"Resume file size exceeds 2 MB: {file_size} bytes")
raise ValueError("Resume file size exceeds the maximum limit of 2 MB.")
allowed_extensions = {'.pdf', '.doc', '.docx'}
file_extension = os.path.splitext(file_path_pdf)[1].lower()
logger.debug(f"Resume file extension: {file_extension}")
if file_extension not in allowed_extensions:
logger.error(f"Invalid resume file format: {file_extension}")
raise ValueError("Resume file format is not allowed. Only PDF, DOC, and DOCX formats are supported.")
try:
logger.debug(f"Uploading resume from path: {file_path_pdf}")
element.send_keys(os.path.abspath(file_path_pdf))
job.pdf_path = os.path.abspath(file_path_pdf)
time.sleep(2)
logger.debug(f"Resume created and uploaded successfully: {file_path_pdf}")
except Exception as e:
tb_str = traceback.format_exc()
logger.error(f"Resume upload failed: {tb_str}")
raise Exception(f"Upload failed: \nTraceback:\n{tb_str}")
def _create_and_upload_cover_letter(self, element: WebElement) -> None:
logger.debug("Creating and uploading cover letter")
@ -329,30 +499,56 @@ class LinkedInEasyApplier:
return False
def _find_and_handle_textbox_question(self, section: WebElement) -> bool:
logger.debug("Searching for text fields in the section.")
text_fields = section.find_elements(By.TAG_NAME, 'input') + section.find_elements(By.TAG_NAME, 'textarea')
if text_fields:
text_field = text_fields[0]
question_text = section.find_element(By.TAG_NAME, 'label').text.lower()
logger.debug(f"Found text field with label: {question_text}")
is_numeric = self._is_numeric_field(text_field)
logger.debug(f"Is the field numeric? {'Yes' if is_numeric else 'No'}")
if is_numeric:
question_type = 'numeric'
answer = self.gpt_answerer.answer_question_numeric(question_text)
logger.debug(f"Generated numeric answer: {answer}")
else:
question_type = 'textbox'
answer = self.gpt_answerer.answer_question_textual_wide_range(question_text)
logger.debug(f"Generated textual answer: {answer}")
existing_answer = None
for item in self.all_data:
if item['question'] == self._sanitize_text(question_text) and item['type'] == question_type:
existing_answer = item
logger.debug(f"Found existing answer in the data: {existing_answer['answer']}")
break
if existing_answer:
self._enter_text(text_field, existing_answer['answer'])
logger.debug("Entered existing textbox answer")
logger.debug("Entered existing textbox answer.")
# Нажать "Вниз" и "Enter" для выбора первого элемента в выпадающем списке
time.sleep(1) # Ожидание появления выпадающего списка
text_field.send_keys(Keys.ARROW_DOWN)
text_field.send_keys(Keys.ENTER)
logger.debug("Selected first option from the dropdown.")
return True
self._save_questions_to_json({'type': question_type, 'question': question_text, 'answer': answer})
self._enter_text(text_field, answer)
logger.debug("Entered new textbox answer")
logger.debug("Entered new textbox answer and saved it to JSON.")
# Нажать "Вниз" и "Enter" для выбора первого элемента в выпадающем списке
time.sleep(1) # Ожидание появления выпадающего списка
text_field.send_keys(Keys.ARROW_DOWN)
text_field.send_keys(Keys.ENTER)
logger.debug("Selected first option from the dropdown.")
return True
logger.debug("No text fields found in the section.")
return False
def _find_and_handle_date_question(self, section: WebElement) -> bool:
@ -384,16 +580,20 @@ class LinkedInEasyApplier:
try:
question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element')
question_text = question.find_element(By.TAG_NAME, 'label').text.lower()
dropdown = question.find_element(By.TAG_NAME, 'select')
if dropdown:
logger.debug(f"Processing dropdown or combobox question: {question_text}")
try:
dropdown = question.find_element(By.TAG_NAME, 'select')
select = Select(dropdown)
options = [option.text for option in select.options]
logger.debug(f"Dropdown options found: {options}")
existing_answer = None
for item in self.all_data:
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown':
existing_answer = item
break
if existing_answer:
self._select_dropdown_option(dropdown, existing_answer['answer'])
logger.debug("Selected existing dropdown answer")
@ -404,14 +604,37 @@ class LinkedInEasyApplier:
self._select_dropdown_option(dropdown, answer)
logger.debug("Selected new dropdown answer")
return True
except NoSuchElementException:
combobox = question.find_element(By.TAG_NAME, 'input')
logger.debug(f"Found combobox with ID: {combobox.get_attribute('id')}")
existing_answer = None
for item in self.all_data:
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'combobox':
existing_answer = item
break
if existing_answer:
self._enter_text(combobox, existing_answer['answer'])
logger.debug("Entered existing combobox answer")
return True
answer = self.gpt_answerer.answer_question_textual_wide_range(question_text)
self._save_questions_to_json({'type': 'combobox', 'question': question_text, 'answer': answer})
self._enter_text(combobox, answer)
logger.debug("Entered new combobox answer")
return True
except Exception as e:
logger.warning("Failed to handle dropdown question: %s", e)
logger.warning("Failed to handle dropdown or combobox question: %s", e)
return False
def _is_numeric_field(self, field: WebElement) -> bool:
field_type = field.get_attribute('type').lower()
is_numeric = 'numeric' in field_type or ('id' in field.get_attribute("id") and 'numeric' in field.get_attribute("id"))
logger.debug("Field is numeric: %s", is_numeric)
field_id = field.get_attribute("id").lower()
is_numeric = 'numeric' in field_id or field_type == 'number' or ('text' == field_type and 'numeric' in field_id)
logger.debug("Field type: %s, Field ID: %s, Is numeric: %s", field_type, field_id, is_numeric)
return is_numeric
def _enter_text(self, element: WebElement, text: str) -> None:

View file

@ -85,12 +85,24 @@ class LinkedInJobManager:
self.next_job_page(position, location_url, job_page_number)
time.sleep(random.uniform(1.5, 3.5))
utils.printyellow("Starting the application process for this page...")
# Проверка на наличие вакансий на странице
try:
jobs = self.get_jobs_from_page()
if not jobs:
utils.printyellow("No more jobs found on this page. Exiting loop.")
break
except Exception as e:
logger.error(f"Failed to retrieve jobs: {e}")
break # Выходим из цикла, если не удалось получить вакансии
try:
self.apply_jobs()
except Exception as e:
logger.error("Error during job application: %s", e)
utils.printred(f"Error during job application: {e}")
continue
utils.printyellow("Applying to jobs on this page has been completed!")
time_left = minimum_page_time - time.time()
@ -122,6 +134,47 @@ class LinkedInJobManager:
time.sleep(sleep_time)
page_sleep += 1
def get_jobs_from_page(self):
"""
Функция для получения списка вакансий на текущей странице.
Если вакансии не найдены, возвращает пустой список.
"""
try:
# Проверка на отсутствие вакансий
no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand')
if 'No matching jobs found' in no_jobs_element.text or 'unfortunately, things aren' in self.driver.page_source.lower():
utils.printyellow("No matching jobs found on this page.")
logger.debug("No matching jobs found on this page, skipping.")
return [] # Возвращаем пустой список, если нет вакансий
except NoSuchElementException:
pass # Если элемент не найден, продолжаем поиск вакансий
# Поиск контейнера результатов с вакансиями
try:
job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list")
utils.scroll_slow(self.driver, job_results)
utils.scroll_slow(self.driver, job_results, step=300, reverse=True)
# Поиск элементов списка вакансий
job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item')
if not job_list_elements:
utils.printyellow("No job class elements found on page.")
logger.debug("No job class elements found on page, skipping.")
return []
# Возвращаем список найденных вакансий
return job_list_elements
except NoSuchElementException:
logger.debug("No job results found on the page.")
return [] # Если не найден контейнер с результатами, возвращаем пустой список
except Exception as e:
logger.error(f"Error while fetching job elements: {e}")
return []
def apply_jobs(self):
try:
no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand')

View file

@ -10,6 +10,11 @@ import logging
logging.basicConfig(level=logging.DEBUG, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
# Отключаем логирование для selenium и urllib3
logging.getLogger("selenium.webdriver.remote.remote_connection").setLevel(logging.WARNING)
logging.getLogger("urllib3").setLevel(logging.WARNING)
logging.getLogger("httpcore").setLevel(logging.WARNING)
chromeProfilePath = os.path.join(os.getcwd(), "chrome_profile", "linkedin_profile")
@ -31,7 +36,7 @@ def is_scrollable(element):
logger.debug("Element scrollable check: scrollHeight=%s, clientHeight=%s, scrollable=%s", scroll_height, client_height, scrollable)
return scrollable
def scroll_slow(driver, scrollable_element, start=0, end=3600, step=100, reverse=False):
def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse=False):
logger.debug("Starting slow scroll: start=%d, end=%d, step=%d, reverse=%s", start, end, step, reverse)
if reverse:
start, end = end, start
@ -39,6 +44,14 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=100, reverse
if step == 0:
logger.error("Step value cannot be zero.")
raise ValueError("Step cannot be zero.")
max_scroll_height = int(scrollable_element.get_attribute("scrollHeight"))
logger.debug("Max scroll height of the element: %d", max_scroll_height)
if end > max_scroll_height:
logger.warning("End value exceeds the scroll height. Adjusting end to %d", max_scroll_height)
end = max_scroll_height
script_scroll_to = "arguments[0].scrollTop = arguments[1];"
try:
if scrollable_element.is_displayed():