add logs and some bugs fixes
This commit is contained in:
parent
0c4ae18064
commit
525e794f8b
5 changed files with 565 additions and 224 deletions
316
src/gpt.py
316
src/gpt.py
|
|
@ -2,20 +2,21 @@ import json
|
|||
import os
|
||||
import re
|
||||
import textwrap
|
||||
import time
|
||||
from datetime import datetime
|
||||
from typing import Dict, List
|
||||
from functools import wraps
|
||||
from pathlib import Path
|
||||
from typing import Dict, List
|
||||
|
||||
import httpx
|
||||
from Levenshtein import distance
|
||||
from dotenv import load_dotenv
|
||||
from httpx import HTTPStatusError
|
||||
from langchain_core.messages.ai import AIMessage
|
||||
from langchain_core.output_parsers import StrOutputParser
|
||||
from langchain_core.prompt_values import StringPromptValue
|
||||
from langchain_core.prompts import ChatPromptTemplate
|
||||
from langchain_openai import ChatOpenAI
|
||||
from Levenshtein import distance
|
||||
import time
|
||||
from functools import wraps
|
||||
from openai import RateLimitError, OpenAIError, APIError
|
||||
|
||||
|
||||
import src.strings as strings
|
||||
from src.utils import logger
|
||||
|
|
@ -42,156 +43,209 @@ def global_rate_limiter(min_interval):
|
|||
|
||||
return decorator
|
||||
|
||||
def parse_wait_time_from_error_message(error_message: str) -> int:
|
||||
logger.debug("Parsing wait time from error message: %s", error_message)
|
||||
match = re.search(r"Please try again in (\d+)([smhd])", error_message)
|
||||
if match:
|
||||
value, unit = int(match.group(1)), match.group(2)
|
||||
logger.debug("Extracted wait time: %d %s", value, unit)
|
||||
if unit == 's':
|
||||
return value
|
||||
elif unit == 'm':
|
||||
return value * 60
|
||||
elif unit == 'h':
|
||||
return value * 3600
|
||||
elif unit == 'd':
|
||||
return value * 86400
|
||||
logger.debug("Default wait time applied: 30 seconds")
|
||||
return 30
|
||||
|
||||
|
||||
class LLMLogger:
|
||||
|
||||
def __init__(self, llm: ChatOpenAI):
|
||||
logger.debug("Initializing LLMLogger with LLM: %s", llm)
|
||||
self.llm = llm
|
||||
logger.debug("LLMLogger initialized with LLM: %s", llm)
|
||||
logger.debug("LLMLogger successfully initialized with LLM: %s", llm)
|
||||
|
||||
@staticmethod
|
||||
def log_request(prompts, parsed_reply: Dict[str, Dict]):
|
||||
logger.debug("Logging request with prompts: %s", prompts)
|
||||
calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json")
|
||||
logger.debug("Starting log_request method")
|
||||
logger.debug("Prompts received: %s", prompts)
|
||||
logger.debug("Parsed reply received: %s", parsed_reply)
|
||||
|
||||
# Определяем путь к файлу для записи логов
|
||||
try:
|
||||
calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json")
|
||||
logger.debug("Logging path determined: %s", calls_log)
|
||||
except Exception as e:
|
||||
logger.error("Error determining the log path: %s", str(e))
|
||||
raise
|
||||
|
||||
# Преобразование prompts в текст или словарь
|
||||
if isinstance(prompts, StringPromptValue):
|
||||
logger.debug("Prompts are of type StringPromptValue")
|
||||
prompts = prompts.text
|
||||
logger.debug("Prompts converted to text: %s", prompts)
|
||||
elif isinstance(prompts, Dict):
|
||||
# Convert prompts to a dictionary if they are not in the expected format
|
||||
prompts = {
|
||||
f"prompt_{i+1}": prompt.content
|
||||
for i, prompt in enumerate(prompts.messages)
|
||||
}
|
||||
logger.debug("Prompts are of type Dict")
|
||||
try:
|
||||
prompts = {
|
||||
f"prompt_{i+1}": prompt.content
|
||||
for i, prompt in enumerate(prompts.messages)
|
||||
}
|
||||
logger.debug("Prompts converted to dictionary: %s", prompts)
|
||||
except Exception as e:
|
||||
logger.error("Error converting prompts to dictionary: %s", str(e))
|
||||
raise
|
||||
else:
|
||||
prompts = {
|
||||
f"prompt_{i+1}": prompt.content
|
||||
for i, prompt in enumerate(prompts.messages)
|
||||
logger.debug("Prompts are of unknown type, attempting default conversion")
|
||||
try:
|
||||
prompts = {
|
||||
f"prompt_{i+1}": prompt.content
|
||||
for i, prompt in enumerate(prompts.messages)
|
||||
}
|
||||
logger.debug("Prompts converted to dictionary using default method: %s", prompts)
|
||||
except Exception as e:
|
||||
logger.error("Error converting prompts using default method: %s", str(e))
|
||||
raise
|
||||
|
||||
# Получение текущего времени
|
||||
try:
|
||||
current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
logger.debug("Current time obtained: %s", current_time)
|
||||
except Exception as e:
|
||||
logger.error("Error obtaining current time: %s", str(e))
|
||||
raise
|
||||
|
||||
# Извлечение информации о токенах
|
||||
try:
|
||||
token_usage = parsed_reply["usage_metadata"]
|
||||
output_tokens = token_usage["output_tokens"]
|
||||
input_tokens = token_usage["input_tokens"]
|
||||
total_tokens = token_usage["total_tokens"]
|
||||
logger.debug("Token usage - Input: %d, Output: %d, Total: %d", input_tokens, output_tokens, total_tokens)
|
||||
except KeyError as e:
|
||||
logger.error("KeyError in parsed_reply structure: %s", str(e))
|
||||
raise
|
||||
|
||||
# Извлечение имени модели
|
||||
try:
|
||||
model_name = parsed_reply["response_metadata"]["model_name"]
|
||||
logger.debug("Model name: %s", model_name)
|
||||
except KeyError as e:
|
||||
logger.error("KeyError in response_metadata: %s", str(e))
|
||||
raise
|
||||
|
||||
# Вычисление стоимости использования API
|
||||
try:
|
||||
prompt_price_per_token = 0.00000015
|
||||
completion_price_per_token = 0.0000006
|
||||
total_cost = (input_tokens * prompt_price_per_token) + (output_tokens * completion_price_per_token)
|
||||
logger.debug("Total cost calculated: %f", total_cost)
|
||||
except Exception as e:
|
||||
logger.error("Error calculating total cost: %s", str(e))
|
||||
raise
|
||||
|
||||
# Формирование записи лога
|
||||
try:
|
||||
log_entry = {
|
||||
"model": model_name,
|
||||
"time": current_time,
|
||||
"prompts": prompts,
|
||||
"replies": parsed_reply["content"], # Контент ответа
|
||||
"total_tokens": total_tokens,
|
||||
"input_tokens": input_tokens,
|
||||
"output_tokens": output_tokens,
|
||||
"total_cost": total_cost,
|
||||
}
|
||||
logger.debug("Log entry created: %s", log_entry)
|
||||
except KeyError as e:
|
||||
logger.error("Error creating log entry: missing key %s in parsed_reply", str(e))
|
||||
raise
|
||||
|
||||
current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
logger.debug("Current time: %s", current_time)
|
||||
|
||||
# Extract token usage details from the response
|
||||
token_usage = parsed_reply["usage_metadata"]
|
||||
output_tokens = token_usage["output_tokens"]
|
||||
input_tokens = token_usage["input_tokens"]
|
||||
total_tokens = token_usage["total_tokens"]
|
||||
|
||||
logger.debug("Token usage - Input: %d, Output: %d, Total: %d", input_tokens, output_tokens, total_tokens)
|
||||
|
||||
model_name = parsed_reply["response_metadata"]["model_name"]
|
||||
prompt_price_per_token = 0.00000015
|
||||
completion_price_per_token = 0.0000006
|
||||
|
||||
# Calculate the total cost of the API call
|
||||
total_cost = (input_tokens * prompt_price_per_token) + (
|
||||
output_tokens * completion_price_per_token
|
||||
)
|
||||
|
||||
logger.debug("Total cost calculated: %f", total_cost)
|
||||
|
||||
log_entry = {
|
||||
"model": model_name,
|
||||
"time": current_time,
|
||||
"prompts": prompts,
|
||||
"replies": parsed_reply["content"], # Response content
|
||||
"total_tokens": total_tokens,
|
||||
"input_tokens": input_tokens,
|
||||
"output_tokens": output_tokens,
|
||||
"total_cost": total_cost,
|
||||
}
|
||||
|
||||
logger.debug("Log entry created: %s", log_entry)
|
||||
|
||||
with open(calls_log, "a", encoding="utf-8") as f:
|
||||
json_string = json.dumps(log_entry, ensure_ascii=False, indent=4)
|
||||
f.write(json_string + "\n")
|
||||
logger.debug("Log entry written to file: %s", calls_log)
|
||||
# Запись в файл
|
||||
try:
|
||||
with open(calls_log, "a", encoding="utf-8") as f:
|
||||
json_string = json.dumps(log_entry, ensure_ascii=False, indent=4)
|
||||
f.write(json_string + "\n")
|
||||
logger.debug("Log entry written to file: %s", calls_log)
|
||||
except Exception as e:
|
||||
logger.error("Error writing log entry to file: %s", str(e))
|
||||
raise
|
||||
|
||||
|
||||
class LoggerChatModel:
|
||||
|
||||
def __init__(self, llm: ChatOpenAI):
|
||||
logger.debug("Initializing LoggerChatModel with LLM: %s", llm)
|
||||
self.llm = llm
|
||||
logger.debug("LoggerChatModel initialized with LLM: %s", llm)
|
||||
logger.debug("LoggerChatModel successfully initialized with LLM: %s", llm)
|
||||
|
||||
def __call__(self, messages: List[Dict[str, str]]) -> str:
|
||||
logger.debug("Calling LoggerChatModel with messages: %s", messages)
|
||||
while True:
|
||||
logger.debug("Entering __call__ method with messages: %s", messages)
|
||||
while True: # Бесконечный цикл до успешного выполнения
|
||||
try:
|
||||
# Попытка вызвать модель
|
||||
reply = self.llm(messages)
|
||||
logger.debug("Model reply received: %s", reply)
|
||||
logger.debug("Attempting to call the LLM with messages")
|
||||
reply = self.llm(messages) # Вызов LLM
|
||||
logger.debug("LLM response received: %s", reply)
|
||||
|
||||
parsed_reply = self.parse_llmresult(reply)
|
||||
logger.debug("Parsed LLM reply: %s", parsed_reply)
|
||||
|
||||
# Логируем запрос и ответ
|
||||
LLMLogger.log_request(prompts=messages, parsed_reply=parsed_reply)
|
||||
return reply
|
||||
except RateLimitError as err:
|
||||
# Handle RateLimitError
|
||||
wait_time = self.parse_wait_time_from_error_message(str(err))
|
||||
logger.warning("Rate limit exceeded. Waiting for %d seconds before retrying...", wait_time)
|
||||
time.sleep(wait_time)
|
||||
logger.debug("Request successfully logged")
|
||||
|
||||
return reply # Возвращаем корректный ответ, завершаем цикл
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error("HTTPStatusError encountered: %s", str(e))
|
||||
if e.response.status_code == 429:
|
||||
retry_after = e.response.headers.get('retry-after')
|
||||
retry_after_ms = e.response.headers.get('retry-after-ms')
|
||||
|
||||
if retry_after:
|
||||
wait_time = int(retry_after)
|
||||
logger.warning("Rate limit exceeded. Waiting for %d seconds before retrying (extracted from 'retry-after' header)...", wait_time)
|
||||
time.sleep(wait_time)
|
||||
elif retry_after_ms:
|
||||
wait_time = int(retry_after_ms) / 1000.0
|
||||
logger.warning("Rate limit exceeded. Waiting for %f seconds before retrying (extracted from 'retry-after-ms' header)...", wait_time)
|
||||
time.sleep(wait_time)
|
||||
else:
|
||||
wait_time = 30 # Время ожидания по умолчанию
|
||||
logger.warning("'retry-after' header not found. Waiting for %d seconds before retrying (default)...", wait_time)
|
||||
time.sleep(wait_time)
|
||||
else:
|
||||
logger.error("HTTP error occurred with status code: %d, waiting 30 seconds before retrying", e.response.status_code)
|
||||
time.sleep(30)
|
||||
|
||||
except Exception as e:
|
||||
logger.error("Unexpected error occurred: %s", str(e))
|
||||
raise
|
||||
logger.info("Waiting for 30 seconds before retrying due to an unexpected error.")
|
||||
time.sleep(30)
|
||||
continue # Продолжаем цикл
|
||||
|
||||
def parse_llmresult(self, llmresult: AIMessage) -> Dict[str, Dict]:
|
||||
logger.debug("Parsing LLM result: %s", llmresult)
|
||||
content = llmresult.content
|
||||
response_metadata = llmresult.response_metadata
|
||||
id_ = llmresult.id
|
||||
usage_metadata = llmresult.usage_metadata
|
||||
parsed_result = {
|
||||
"content": content,
|
||||
"response_metadata": {
|
||||
"model_name": response_metadata.get("model_name", ""),
|
||||
"system_fingerprint": response_metadata.get("system_fingerprint", ""),
|
||||
"finish_reason": response_metadata.get("finish_reason", ""),
|
||||
"logprobs": response_metadata.get("logprobs", None),
|
||||
},
|
||||
"id": id_,
|
||||
"usage_metadata": {
|
||||
"input_tokens": usage_metadata.get("input_tokens", 0),
|
||||
"output_tokens": usage_metadata.get("output_tokens", 0),
|
||||
"total_tokens": usage_metadata.get("total_tokens", 0),
|
||||
},
|
||||
}
|
||||
logger.debug("Parsed LLM result: %s", parsed_result)
|
||||
return parsed_result
|
||||
|
||||
def parse_wait_time_from_error_message(self, error_message: str) -> int:
|
||||
logger.debug("Parsing wait time from error message: %s", error_message)
|
||||
match = re.search(r"Please try again in (\d+)([smhd])", error_message)
|
||||
if match:
|
||||
value, unit = match.groups()
|
||||
value = int(value)
|
||||
logger.debug("Extracted wait time: %d %s", value, unit)
|
||||
if unit == "s":
|
||||
return value
|
||||
elif unit == "m":
|
||||
return value * 60
|
||||
elif unit == "h":
|
||||
return value * 3600
|
||||
elif unit == "d":
|
||||
return value * 86400
|
||||
logger.debug("Default wait time applied: 30 seconds")
|
||||
return 30
|
||||
# Извлечение данных из ответа
|
||||
try:
|
||||
content = llmresult.content
|
||||
response_metadata = llmresult.response_metadata
|
||||
id_ = llmresult.id
|
||||
usage_metadata = llmresult.usage_metadata
|
||||
|
||||
parsed_result = {
|
||||
"content": content,
|
||||
"response_metadata": {
|
||||
"model_name": response_metadata.get("model_name", ""),
|
||||
"system_fingerprint": response_metadata.get("system_fingerprint", ""),
|
||||
"finish_reason": response_metadata.get("finish_reason", ""),
|
||||
"logprobs": response_metadata.get("logprobs", None),
|
||||
},
|
||||
"id": id_,
|
||||
"usage_metadata": {
|
||||
"input_tokens": usage_metadata.get("input_tokens", 0),
|
||||
"output_tokens": usage_metadata.get("output_tokens", 0),
|
||||
"total_tokens": usage_metadata.get("total_tokens", 0),
|
||||
},
|
||||
}
|
||||
|
||||
logger.debug("Parsed LLM result successfully: %s", parsed_result)
|
||||
return parsed_result
|
||||
|
||||
except KeyError as e:
|
||||
logger.error("KeyError while parsing LLM result: missing key %s", str(e))
|
||||
raise # Повторно выбрасываем исключение, чтобы оно обрабатывалось выше
|
||||
|
||||
except Exception as e:
|
||||
logger.error("Unexpected error while parsing LLM result: %s", str(e))
|
||||
raise
|
||||
|
||||
|
||||
|
||||
class GPTAnswerer:
|
||||
|
|
@ -239,7 +293,7 @@ class GPTAnswerer:
|
|||
logger.debug("Setting job application profile: %s", job_application_profile)
|
||||
self.job_application_profile = job_application_profile
|
||||
|
||||
@global_rate_limiter(25)
|
||||
#@global_rate_limiter(25)
|
||||
def summarize_job_description(self, text: str) -> str:
|
||||
logger.debug("Summarizing job description: %s", text)
|
||||
strings.summarize_prompt_template = self._preprocess_template_string(
|
||||
|
|
@ -256,7 +310,7 @@ class GPTAnswerer:
|
|||
prompt = ChatPromptTemplate.from_template(template)
|
||||
return prompt | self.llm_cheap | StrOutputParser()
|
||||
|
||||
@global_rate_limiter(25)
|
||||
#@global_rate_limiter(25)
|
||||
def answer_question_textual_wide_range(self, question: str) -> str:
|
||||
logger.debug("Answering textual question: %s", question)
|
||||
chains = {
|
||||
|
|
@ -384,7 +438,7 @@ class GPTAnswerer:
|
|||
logger.debug("Question answered: %s", output)
|
||||
return output
|
||||
|
||||
@global_rate_limiter(25)
|
||||
#@global_rate_limiter(25)
|
||||
def answer_question_numeric(self, question: str, default_experience: int = 3) -> int:
|
||||
logger.debug("Answering numeric question: %s", question)
|
||||
func_template = self._preprocess_template_string(strings.numeric_question_template)
|
||||
|
|
@ -410,7 +464,7 @@ class GPTAnswerer:
|
|||
logger.error("No numbers found in the string")
|
||||
raise ValueError("No numbers found in the string")
|
||||
|
||||
@global_rate_limiter(25)
|
||||
#@global_rate_limiter(25)
|
||||
def answer_question_from_options(self, question: str, options: list[str]) -> str:
|
||||
logger.debug("Answering question from options: %s", question)
|
||||
func_template = self._preprocess_template_string(strings.options_template)
|
||||
|
|
@ -422,11 +476,11 @@ class GPTAnswerer:
|
|||
logger.debug("Best option determined: %s", best_option)
|
||||
return best_option
|
||||
|
||||
@global_rate_limiter(25)
|
||||
#@global_rate_limiter(25)
|
||||
def resume_or_cover(self, phrase: str) -> str:
|
||||
logger.debug("Determining if phrase refers to resume or cover letter: %s", phrase)
|
||||
prompt_template = """
|
||||
Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. Do not provide any additional information or explanations.
|
||||
Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. If the phrase contains only the word 'upload', consider it as 'cover'. Do not provide any additional information or explanations.
|
||||
|
||||
phrase: {phrase}
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -25,7 +25,14 @@ class LinkedInAuthenticator:
|
|||
logger.info("Starting Chrome browser to log in to LinkedIn.")
|
||||
self.driver.get('https://www.linkedin.com/feed')
|
||||
self.wait_for_page_load()
|
||||
if not self.is_logged_in():
|
||||
|
||||
time.sleep(3)
|
||||
|
||||
if self.is_logged_in():
|
||||
logger.info("User is already logged in. Skipping login process.")
|
||||
return
|
||||
else:
|
||||
logger.info("User is not logged in. Proceeding with login.")
|
||||
self.handle_login()
|
||||
|
||||
def handle_login(self):
|
||||
|
|
@ -82,12 +89,12 @@ class LinkedInAuthenticator:
|
|||
print("Security check not completed. Please try again later.")
|
||||
|
||||
def is_logged_in(self):
|
||||
target_url = 'https://www.linkedin.com/feed'
|
||||
|
||||
# Navigate to the target URL if not already there
|
||||
if self.driver.current_url != target_url:
|
||||
logger.debug("Navigating to target URL: %s", target_url)
|
||||
self.driver.get(target_url)
|
||||
# target_url = 'https://www.linkedin.com/feed'
|
||||
#
|
||||
# # Navigate to the target URL if not already there
|
||||
# if self.driver.current_url != target_url:
|
||||
# logger.debug("Navigating to target URL: %s", target_url)
|
||||
# self.driver.get(target_url)
|
||||
|
||||
try:
|
||||
# Increase the wait time for the page elements to load
|
||||
|
|
@ -98,38 +105,29 @@ class LinkedInAuthenticator:
|
|||
|
||||
# Check for the presence of the "Start a post" button
|
||||
buttons = self.driver.find_elements(By.CLASS_NAME, 'share-box-feed-entry__trigger')
|
||||
if any(button.text.strip() == 'Start a post' for button in buttons):
|
||||
logger.info("User is already logged in.")
|
||||
logger.debug("Found %d 'Start a post' buttons", len(buttons))
|
||||
|
||||
try:
|
||||
# Wait for the profile picture and name to load
|
||||
profile_img = WebDriverWait(self.driver, 10).until(
|
||||
EC.presence_of_element_located((By.XPATH, "//img[contains(@alt, 'Photo of')]"))
|
||||
)
|
||||
profile_name = WebDriverWait(self.driver, 10).until(
|
||||
EC.presence_of_element_located((By.XPATH, "//div[@class='t-16 t-black t-bold']"))
|
||||
)
|
||||
# Выведем текст всех найденных кнопок в лог для диагностики
|
||||
for i, button in enumerate(buttons):
|
||||
logger.debug("Button %d text: %s", i + 1, button.text.strip())
|
||||
|
||||
if profile_img and profile_name:
|
||||
logger.info("Profile picture found for user: %s", profile_name.text)
|
||||
return True
|
||||
except NoSuchElementException:
|
||||
logger.warning("Profile picture or name not found.")
|
||||
print("Profile picture or name not found.")
|
||||
return False
|
||||
except TimeoutException:
|
||||
logger.warning("Profile picture or name took too long to load.")
|
||||
print("Profile picture or name took too long to load.")
|
||||
return False
|
||||
if any(button.text.strip().lower() == 'start a post' for button in buttons):
|
||||
logger.info("Found 'Start a post' button indicating user is logged in.")
|
||||
return True
|
||||
|
||||
# Альтернативная проверка авторизации по наличию изображения профиля
|
||||
profile_img_elements = self.driver.find_elements(By.XPATH, "//img[contains(@alt, 'Photo of')]")
|
||||
if profile_img_elements:
|
||||
logger.info("Profile image found. Assuming user is logged in.")
|
||||
return True
|
||||
|
||||
logger.info("Did not find 'Start a post' button or profile image. User might not be logged in.")
|
||||
return False
|
||||
|
||||
except TimeoutException:
|
||||
logger.error("Page elements took too long to load or were not found.")
|
||||
print("Page elements took too long to load or were not found.")
|
||||
return False
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def wait_for_page_load(self, timeout=10):
|
||||
try:
|
||||
logger.debug("Waiting for page to load with timeout: %s seconds", timeout)
|
||||
|
|
|
|||
|
|
@ -8,6 +8,9 @@ import time
|
|||
import traceback
|
||||
from datetime import date
|
||||
from typing import List, Optional, Any, Tuple
|
||||
|
||||
from httpx import HTTPStatusError
|
||||
from openai import RateLimitError
|
||||
from reportlab.lib.pagesizes import letter
|
||||
from reportlab.pdfgen import canvas
|
||||
from selenium.common.exceptions import NoSuchElementException, TimeoutException
|
||||
|
|
@ -57,54 +60,131 @@ class LinkedInEasyApplier:
|
|||
|
||||
def job_apply(self, job: Any):
|
||||
logger.debug("Starting job application for job: %s", job)
|
||||
self.driver.get(job.link)
|
||||
time.sleep(random.uniform(3, 5))
|
||||
|
||||
# Открываем страницу с вакансией
|
||||
try:
|
||||
self.driver.get(job.link)
|
||||
logger.debug("Navigated to job link: %s", job.link)
|
||||
except Exception as e:
|
||||
logger.error("Failed to navigate to job link: %s, error: %s", job.link, str(e))
|
||||
raise
|
||||
|
||||
# Добавляем небольшую паузу для загрузки страницы
|
||||
time.sleep(random.uniform(3, 5))
|
||||
|
||||
try:
|
||||
# Поиск кнопки 'Easy Apply'
|
||||
logger.debug("Searching for 'Easy Apply' button on job page")
|
||||
easy_apply_button = self._find_easy_apply_button()
|
||||
job.set_job_description(self._get_job_description())
|
||||
job.set_recruiter_link(self._get_job_recruiter())
|
||||
|
||||
# Получаем описание вакансии
|
||||
logger.debug("Retrieving job description")
|
||||
job_description = self._get_job_description()
|
||||
job.set_job_description(job_description)
|
||||
logger.debug("Job description set: %s", job_description[:100]) # Логируем только первые 100 символов
|
||||
|
||||
# Получаем ссылку на рекрутера (если есть)
|
||||
logger.debug("Retrieving recruiter link")
|
||||
recruiter_link = self._get_job_recruiter()
|
||||
job.set_recruiter_link(recruiter_link)
|
||||
logger.debug("Recruiter link set: %s", recruiter_link)
|
||||
|
||||
# Действие: нажимаем на кнопку 'Easy Apply'
|
||||
logger.debug("Attempting to click 'Easy Apply' button")
|
||||
actions = ActionChains(self.driver)
|
||||
actions.move_to_element(easy_apply_button).click().perform()
|
||||
logger.debug("'Easy Apply' button clicked successfully")
|
||||
|
||||
# Передача информации о работе для дальнейшей обработки
|
||||
logger.debug("Passing job information to GPT Answerer")
|
||||
self.gpt_answerer.set_job(job)
|
||||
|
||||
# Заполнение формы подачи заявки
|
||||
logger.debug("Filling out application form")
|
||||
self._fill_application_form(job)
|
||||
logger.debug("Job application process completed for job: %s", job)
|
||||
except Exception:
|
||||
logger.debug("Job application process completed successfully for job: %s", job)
|
||||
|
||||
except Exception as e:
|
||||
# Захват и логирование полного traceback в случае ошибки
|
||||
tb_str = traceback.format_exc()
|
||||
logger.error("Failed to apply to job: %s", tb_str)
|
||||
logger.error("Failed to apply to job: %s. Error traceback: %s", job, tb_str)
|
||||
|
||||
# Отмена заявки в случае ошибки
|
||||
logger.debug("Discarding application due to failure")
|
||||
self._discard_application()
|
||||
raise Exception(f"Failed to apply to job! Original exception: \nTraceback:\n{tb_str}")
|
||||
|
||||
# Поднятие исключения с оригинальной ошибкой
|
||||
raise Exception(f"Failed to apply to job! Original exception:\nTraceback:\n{tb_str}")
|
||||
|
||||
def _find_easy_apply_button(self) -> WebElement:
|
||||
logger.debug("Searching for 'Easy Apply' button")
|
||||
attempt = 0
|
||||
|
||||
# Список методов поиска кнопки
|
||||
search_methods = [
|
||||
{
|
||||
'description': "find all 'Easy Apply' buttons using find_elements",
|
||||
'find_elements': True, # Используем find_elements для поиска всех кнопок
|
||||
'xpath': '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]'
|
||||
},
|
||||
{
|
||||
'description': "'aria-label' containing 'Easy Apply to'",
|
||||
'xpath': '//button[contains(@aria-label, "Easy Apply to")]'
|
||||
},
|
||||
{
|
||||
'description': "button text search",
|
||||
'xpath': '//button[contains(text(), "Easy Apply") or contains(text(), "Apply now")]'
|
||||
}
|
||||
]
|
||||
|
||||
while attempt < 2:
|
||||
self._scroll_page()
|
||||
try:
|
||||
buttons = WebDriverWait(self.driver, 10).until(
|
||||
EC.presence_of_all_elements_located(
|
||||
(By.XPATH, '//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")]')
|
||||
)
|
||||
)
|
||||
for index, _ in enumerate(buttons):
|
||||
try:
|
||||
button = WebDriverWait(self.driver, 10).until(
|
||||
EC.element_to_be_clickable(
|
||||
(By.XPATH, f'(//button[contains(@class, "jobs-apply-button") and contains(., "Easy Apply")])[{index + 1}]')
|
||||
)
|
||||
)
|
||||
logger.debug("Found and clicking 'Easy Apply' button")
|
||||
return button
|
||||
except Exception as e:
|
||||
logger.warning("Failed to click 'Easy Apply' button on attempt %d: %s", attempt + 1, e)
|
||||
except TimeoutException:
|
||||
logger.warning("Timeout while searching for 'Easy Apply' button")
|
||||
|
||||
for method in search_methods:
|
||||
try:
|
||||
logger.debug(f"Attempting search using {method['description']}")
|
||||
|
||||
# Если метод использует find_elements
|
||||
if method.get('find_elements'):
|
||||
# Поиск всех кнопок "Easy Apply"
|
||||
buttons = self.driver.find_elements(By.XPATH, method['xpath'])
|
||||
if buttons:
|
||||
for index, button in enumerate(buttons):
|
||||
try:
|
||||
# Проверка видимости и кликабельности каждой кнопки
|
||||
WebDriverWait(self.driver, 10).until(EC.visibility_of(button))
|
||||
WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button))
|
||||
logger.debug(f"Found 'Easy Apply' button {index + 1}, attempting to click")
|
||||
return button
|
||||
except Exception as e:
|
||||
logger.warning(f"Button {index + 1} found but not clickable: {e}")
|
||||
else:
|
||||
raise TimeoutException("No 'Easy Apply' buttons found")
|
||||
else:
|
||||
# Стандартный метод с WebDriverWait для одного элемента
|
||||
button = WebDriverWait(self.driver, 10).until(
|
||||
EC.presence_of_element_located((By.XPATH, method['xpath']))
|
||||
)
|
||||
WebDriverWait(self.driver, 10).until(EC.visibility_of(button))
|
||||
WebDriverWait(self.driver, 10).until(EC.element_to_be_clickable(button))
|
||||
logger.debug("Found 'Easy Apply' button, attempting to click")
|
||||
return button
|
||||
|
||||
except TimeoutException:
|
||||
logger.warning(f"Timeout during search using {method['description']}")
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to click 'Easy Apply' button using {method['description']} on attempt {attempt + 1}: {e}")
|
||||
|
||||
# Обновление страницы после первой неудачной попытки
|
||||
if attempt == 0:
|
||||
logger.debug("Refreshing page to retry finding 'Easy Apply' button")
|
||||
self.driver.refresh()
|
||||
time.sleep(random.randint(3, 5))
|
||||
attempt += 1
|
||||
logger.error("No clickable 'Easy Apply' button found after 2 attempts")
|
||||
|
||||
# Если не удалось найти кнопку, выводим HTML для отладки
|
||||
page_source = self.driver.page_source
|
||||
logger.error("No clickable 'Easy Apply' button found after 2 attempts. Page source:\n%s", page_source)
|
||||
raise Exception("No clickable 'Easy Apply' button found")
|
||||
|
||||
def _get_job_description(self) -> str:
|
||||
|
|
@ -136,10 +216,18 @@ class LinkedInEasyApplier:
|
|||
hiring_team_section = WebDriverWait(self.driver, 10).until(
|
||||
EC.presence_of_element_located((By.XPATH, '//h2[text()="Meet the hiring team"]'))
|
||||
)
|
||||
recruiter_element = hiring_team_section.find_element(By.XPATH, './/following::a[contains(@href, "linkedin.com/in/")]')
|
||||
recruiter_link = recruiter_element.get_attribute('href')
|
||||
logger.debug("Job recruiter link retrieved successfully")
|
||||
return recruiter_link
|
||||
logger.debug("Hiring team section found")
|
||||
|
||||
recruiter_elements = hiring_team_section.find_elements(By.XPATH, './/following::a[contains(@href, "linkedin.com/in/")]')
|
||||
|
||||
if recruiter_elements:
|
||||
recruiter_element = recruiter_elements[0]
|
||||
recruiter_link = recruiter_element.get_attribute('href')
|
||||
logger.debug("Job recruiter link retrieved successfully: %s", recruiter_link)
|
||||
return recruiter_link
|
||||
else:
|
||||
logger.debug("No recruiter link found in the hiring team section")
|
||||
return ""
|
||||
except Exception as e:
|
||||
logger.warning("Failed to retrieve recruiter information: %s", e)
|
||||
return ""
|
||||
|
|
@ -202,11 +290,19 @@ class LinkedInEasyApplier:
|
|||
|
||||
def fill_up(self, job) -> None:
|
||||
logger.debug("Filling up form sections for job: %s", job)
|
||||
easy_apply_content = self.driver.find_element(By.CLASS_NAME, 'jobs-easy-apply-content')
|
||||
pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4')
|
||||
for element in pb4_elements:
|
||||
self._process_form_element(element, job)
|
||||
|
||||
|
||||
# Используем WebDriverWait для ожидания элемента с классом 'jobs-easy-apply-content'
|
||||
try:
|
||||
easy_apply_content = WebDriverWait(self.driver, 10).until(
|
||||
EC.presence_of_element_located((By.CLASS_NAME, 'jobs-easy-apply-content'))
|
||||
)
|
||||
|
||||
# После нахождения 'jobs-easy-apply-content' ищем элементы с классом 'pb4'
|
||||
pb4_elements = easy_apply_content.find_elements(By.CLASS_NAME, 'pb4')
|
||||
for element in pb4_elements:
|
||||
self._process_form_element(element, job)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to find form elements: {e}")
|
||||
def _process_form_element(self, element: WebElement, job) -> None:
|
||||
logger.debug("Processing form element")
|
||||
if self._is_upload_field(element):
|
||||
|
|
@ -221,40 +317,114 @@ class LinkedInEasyApplier:
|
|||
|
||||
def _handle_upload_fields(self, element: WebElement, job) -> None:
|
||||
logger.debug("Handling upload fields")
|
||||
|
||||
try:
|
||||
show_more_button = self.driver.find_element(By.XPATH, "//button[contains(@aria-label, 'Show more resumes')]")
|
||||
show_more_button.click()
|
||||
logger.debug("Clicked 'Show more resumes' button")
|
||||
except NoSuchElementException:
|
||||
logger.debug("'Show more resumes' button not found, continuing...")
|
||||
|
||||
file_upload_elements = self.driver.find_elements(By.XPATH, "//input[@type='file']")
|
||||
for element in file_upload_elements:
|
||||
parent = element.find_element(By.XPATH, "..")
|
||||
self.driver.execute_script("arguments[0].classList.remove('hidden')", element)
|
||||
|
||||
output = self.gpt_answerer.resume_or_cover(parent.text.lower())
|
||||
if 'resume' in output:
|
||||
logger.debug("Uploading resume")
|
||||
if self.resume_path is not None and self.resume_path.resolve().is_file():
|
||||
element.send_keys(str(self.resume_path.resolve()))
|
||||
logger.debug(f"Resume uploaded from path: {self.resume_path.resolve()}")
|
||||
else:
|
||||
logger.debug("Resume path not found or invalid, generating new resume")
|
||||
self._create_and_upload_resume(element, job)
|
||||
elif 'cover' in output:
|
||||
logger.debug("Uploading cover letter")
|
||||
self._create_and_upload_cover_letter(element)
|
||||
|
||||
logger.debug("Finished handling upload fields")
|
||||
|
||||
def _create_and_upload_resume(self, element, job):
|
||||
logger.debug("Creating and uploading resume")
|
||||
folder_path = 'generated_cv'
|
||||
os.makedirs(folder_path, exist_ok=True)
|
||||
try:
|
||||
timestamp = int(time.time())
|
||||
file_path_pdf = os.path.join(folder_path, f"CV_{timestamp}.pdf")
|
||||
logger.debug("Starting the process of creating and uploading resume.")
|
||||
folder_path = 'generated_cv'
|
||||
|
||||
with open(file_path_pdf, "xb") as f: # gjcvjn
|
||||
f.write(base64.b64decode(self.resume_generator_manager.pdf_base64(job_description_text=job.description)))
|
||||
try:
|
||||
if not os.path.exists(folder_path):
|
||||
logger.debug(f"Creating directory at path: {folder_path}")
|
||||
os.makedirs(folder_path, exist_ok=True)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create directory: {folder_path}. Error: {e}")
|
||||
raise
|
||||
|
||||
element.send_keys(os.path.abspath(file_path_pdf))
|
||||
job.pdf_path = os.path.abspath(file_path_pdf)
|
||||
time.sleep(2)
|
||||
logger.debug("Resume created and uploaded successfully: %s", file_path_pdf)
|
||||
except Exception:
|
||||
tb_str = traceback.format_exc()
|
||||
logger.error("Resume upload failed: %s", tb_str)
|
||||
raise Exception(f"Upload failed: \nTraceback:\n{tb_str}")
|
||||
while True:
|
||||
try:
|
||||
timestamp = int(time.time())
|
||||
file_path_pdf = os.path.join(folder_path, f"CV_{timestamp}.pdf")
|
||||
logger.debug(f"Generated file path for resume: {file_path_pdf}")
|
||||
|
||||
logger.debug(f"Generating resume for job: {job.title} at {job.company}")
|
||||
resume_pdf_base64 = self.resume_generator_manager.pdf_base64(job_description_text=job.description)
|
||||
with open(file_path_pdf, "xb") as f:
|
||||
f.write(base64.b64decode(resume_pdf_base64))
|
||||
logger.debug(f"Resume successfully generated and saved to: {file_path_pdf}")
|
||||
|
||||
break
|
||||
except HTTPStatusError as e:
|
||||
if e.response.status_code == 429:
|
||||
|
||||
retry_after = e.response.headers.get('retry-after')
|
||||
retry_after_ms = e.response.headers.get('retry-after-ms')
|
||||
|
||||
if retry_after:
|
||||
wait_time = int(retry_after)
|
||||
logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...")
|
||||
elif retry_after_ms:
|
||||
wait_time = int(retry_after_ms) / 1000.0
|
||||
logger.warning(f"Rate limit exceeded, waiting {wait_time} milliseconds before retrying...")
|
||||
else:
|
||||
wait_time = 20
|
||||
logger.warning(f"Rate limit exceeded, waiting {wait_time} seconds before retrying...")
|
||||
|
||||
time.sleep(wait_time)
|
||||
else:
|
||||
logger.error(f"HTTP error: {e}")
|
||||
raise
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to generate resume: {e}")
|
||||
tb_str = traceback.format_exc()
|
||||
logger.error(f"Traceback: {tb_str}")
|
||||
if "RateLimitError" in str(e):
|
||||
logger.warning("Rate limit error encountered, retrying...")
|
||||
time.sleep(20)
|
||||
else:
|
||||
raise
|
||||
|
||||
file_size = os.path.getsize(file_path_pdf)
|
||||
max_file_size = 2 * 1024 * 1024 # 2 MB
|
||||
logger.debug(f"Resume file size: {file_size} bytes")
|
||||
if file_size > max_file_size:
|
||||
logger.error(f"Resume file size exceeds 2 MB: {file_size} bytes")
|
||||
raise ValueError("Resume file size exceeds the maximum limit of 2 MB.")
|
||||
|
||||
allowed_extensions = {'.pdf', '.doc', '.docx'}
|
||||
file_extension = os.path.splitext(file_path_pdf)[1].lower()
|
||||
logger.debug(f"Resume file extension: {file_extension}")
|
||||
if file_extension not in allowed_extensions:
|
||||
logger.error(f"Invalid resume file format: {file_extension}")
|
||||
raise ValueError("Resume file format is not allowed. Only PDF, DOC, and DOCX formats are supported.")
|
||||
|
||||
try:
|
||||
logger.debug(f"Uploading resume from path: {file_path_pdf}")
|
||||
element.send_keys(os.path.abspath(file_path_pdf))
|
||||
job.pdf_path = os.path.abspath(file_path_pdf)
|
||||
time.sleep(2)
|
||||
logger.debug(f"Resume created and uploaded successfully: {file_path_pdf}")
|
||||
except Exception as e:
|
||||
tb_str = traceback.format_exc()
|
||||
logger.error(f"Resume upload failed: {tb_str}")
|
||||
raise Exception(f"Upload failed: \nTraceback:\n{tb_str}")
|
||||
|
||||
def _create_and_upload_cover_letter(self, element: WebElement) -> None:
|
||||
logger.debug("Creating and uploading cover letter")
|
||||
|
|
@ -329,30 +499,56 @@ class LinkedInEasyApplier:
|
|||
return False
|
||||
|
||||
def _find_and_handle_textbox_question(self, section: WebElement) -> bool:
|
||||
logger.debug("Searching for text fields in the section.")
|
||||
text_fields = section.find_elements(By.TAG_NAME, 'input') + section.find_elements(By.TAG_NAME, 'textarea')
|
||||
|
||||
if text_fields:
|
||||
text_field = text_fields[0]
|
||||
question_text = section.find_element(By.TAG_NAME, 'label').text.lower()
|
||||
logger.debug(f"Found text field with label: {question_text}")
|
||||
|
||||
is_numeric = self._is_numeric_field(text_field)
|
||||
logger.debug(f"Is the field numeric? {'Yes' if is_numeric else 'No'}")
|
||||
|
||||
if is_numeric:
|
||||
question_type = 'numeric'
|
||||
answer = self.gpt_answerer.answer_question_numeric(question_text)
|
||||
logger.debug(f"Generated numeric answer: {answer}")
|
||||
else:
|
||||
question_type = 'textbox'
|
||||
answer = self.gpt_answerer.answer_question_textual_wide_range(question_text)
|
||||
logger.debug(f"Generated textual answer: {answer}")
|
||||
|
||||
existing_answer = None
|
||||
for item in self.all_data:
|
||||
if item['question'] == self._sanitize_text(question_text) and item['type'] == question_type:
|
||||
existing_answer = item
|
||||
logger.debug(f"Found existing answer in the data: {existing_answer['answer']}")
|
||||
break
|
||||
|
||||
if existing_answer:
|
||||
self._enter_text(text_field, existing_answer['answer'])
|
||||
logger.debug("Entered existing textbox answer")
|
||||
logger.debug("Entered existing textbox answer.")
|
||||
|
||||
# Нажать "Вниз" и "Enter" для выбора первого элемента в выпадающем списке
|
||||
time.sleep(1) # Ожидание появления выпадающего списка
|
||||
text_field.send_keys(Keys.ARROW_DOWN)
|
||||
text_field.send_keys(Keys.ENTER)
|
||||
logger.debug("Selected first option from the dropdown.")
|
||||
return True
|
||||
|
||||
self._save_questions_to_json({'type': question_type, 'question': question_text, 'answer': answer})
|
||||
self._enter_text(text_field, answer)
|
||||
logger.debug("Entered new textbox answer")
|
||||
logger.debug("Entered new textbox answer and saved it to JSON.")
|
||||
|
||||
# Нажать "Вниз" и "Enter" для выбора первого элемента в выпадающем списке
|
||||
time.sleep(1) # Ожидание появления выпадающего списка
|
||||
text_field.send_keys(Keys.ARROW_DOWN)
|
||||
text_field.send_keys(Keys.ENTER)
|
||||
logger.debug("Selected first option from the dropdown.")
|
||||
return True
|
||||
|
||||
logger.debug("No text fields found in the section.")
|
||||
return False
|
||||
|
||||
def _find_and_handle_date_question(self, section: WebElement) -> bool:
|
||||
|
|
@ -384,16 +580,20 @@ class LinkedInEasyApplier:
|
|||
try:
|
||||
question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element')
|
||||
question_text = question.find_element(By.TAG_NAME, 'label').text.lower()
|
||||
dropdown = question.find_element(By.TAG_NAME, 'select')
|
||||
if dropdown:
|
||||
logger.debug(f"Processing dropdown or combobox question: {question_text}")
|
||||
|
||||
try:
|
||||
dropdown = question.find_element(By.TAG_NAME, 'select')
|
||||
select = Select(dropdown)
|
||||
options = [option.text for option in select.options]
|
||||
logger.debug(f"Dropdown options found: {options}")
|
||||
|
||||
existing_answer = None
|
||||
for item in self.all_data:
|
||||
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'dropdown':
|
||||
existing_answer = item
|
||||
break
|
||||
|
||||
if existing_answer:
|
||||
self._select_dropdown_option(dropdown, existing_answer['answer'])
|
||||
logger.debug("Selected existing dropdown answer")
|
||||
|
|
@ -404,14 +604,37 @@ class LinkedInEasyApplier:
|
|||
self._select_dropdown_option(dropdown, answer)
|
||||
logger.debug("Selected new dropdown answer")
|
||||
return True
|
||||
|
||||
except NoSuchElementException:
|
||||
combobox = question.find_element(By.TAG_NAME, 'input')
|
||||
logger.debug(f"Found combobox with ID: {combobox.get_attribute('id')}")
|
||||
|
||||
existing_answer = None
|
||||
for item in self.all_data:
|
||||
if self._sanitize_text(question_text) in item['question'] and item['type'] == 'combobox':
|
||||
existing_answer = item
|
||||
break
|
||||
|
||||
if existing_answer:
|
||||
self._enter_text(combobox, existing_answer['answer'])
|
||||
logger.debug("Entered existing combobox answer")
|
||||
return True
|
||||
|
||||
answer = self.gpt_answerer.answer_question_textual_wide_range(question_text)
|
||||
self._save_questions_to_json({'type': 'combobox', 'question': question_text, 'answer': answer})
|
||||
self._enter_text(combobox, answer)
|
||||
logger.debug("Entered new combobox answer")
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
logger.warning("Failed to handle dropdown question: %s", e)
|
||||
logger.warning("Failed to handle dropdown or combobox question: %s", e)
|
||||
return False
|
||||
|
||||
def _is_numeric_field(self, field: WebElement) -> bool:
|
||||
field_type = field.get_attribute('type').lower()
|
||||
is_numeric = 'numeric' in field_type or ('id' in field.get_attribute("id") and 'numeric' in field.get_attribute("id"))
|
||||
logger.debug("Field is numeric: %s", is_numeric)
|
||||
field_id = field.get_attribute("id").lower()
|
||||
is_numeric = 'numeric' in field_id or field_type == 'number' or ('text' == field_type and 'numeric' in field_id)
|
||||
logger.debug("Field type: %s, Field ID: %s, Is numeric: %s", field_type, field_id, is_numeric)
|
||||
return is_numeric
|
||||
|
||||
def _enter_text(self, element: WebElement, text: str) -> None:
|
||||
|
|
|
|||
|
|
@ -85,12 +85,24 @@ class LinkedInJobManager:
|
|||
self.next_job_page(position, location_url, job_page_number)
|
||||
time.sleep(random.uniform(1.5, 3.5))
|
||||
utils.printyellow("Starting the application process for this page...")
|
||||
|
||||
# Проверка на наличие вакансий на странице
|
||||
try:
|
||||
jobs = self.get_jobs_from_page()
|
||||
if not jobs:
|
||||
utils.printyellow("No more jobs found on this page. Exiting loop.")
|
||||
break
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to retrieve jobs: {e}")
|
||||
break # Выходим из цикла, если не удалось получить вакансии
|
||||
|
||||
try:
|
||||
self.apply_jobs()
|
||||
except Exception as e:
|
||||
logger.error("Error during job application: %s", e)
|
||||
utils.printred(f"Error during job application: {e}")
|
||||
continue
|
||||
|
||||
utils.printyellow("Applying to jobs on this page has been completed!")
|
||||
|
||||
time_left = minimum_page_time - time.time()
|
||||
|
|
@ -122,6 +134,47 @@ class LinkedInJobManager:
|
|||
time.sleep(sleep_time)
|
||||
page_sleep += 1
|
||||
|
||||
|
||||
def get_jobs_from_page(self):
|
||||
"""
|
||||
Функция для получения списка вакансий на текущей странице.
|
||||
Если вакансии не найдены, возвращает пустой список.
|
||||
"""
|
||||
try:
|
||||
# Проверка на отсутствие вакансий
|
||||
no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand')
|
||||
if 'No matching jobs found' in no_jobs_element.text or 'unfortunately, things aren' in self.driver.page_source.lower():
|
||||
utils.printyellow("No matching jobs found on this page.")
|
||||
logger.debug("No matching jobs found on this page, skipping.")
|
||||
return [] # Возвращаем пустой список, если нет вакансий
|
||||
|
||||
except NoSuchElementException:
|
||||
pass # Если элемент не найден, продолжаем поиск вакансий
|
||||
|
||||
# Поиск контейнера результатов с вакансиями
|
||||
try:
|
||||
job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list")
|
||||
utils.scroll_slow(self.driver, job_results)
|
||||
utils.scroll_slow(self.driver, job_results, step=300, reverse=True)
|
||||
|
||||
# Поиск элементов списка вакансий
|
||||
job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item')
|
||||
if not job_list_elements:
|
||||
utils.printyellow("No job class elements found on page.")
|
||||
logger.debug("No job class elements found on page, skipping.")
|
||||
return []
|
||||
|
||||
# Возвращаем список найденных вакансий
|
||||
return job_list_elements
|
||||
|
||||
except NoSuchElementException:
|
||||
logger.debug("No job results found on the page.")
|
||||
return [] # Если не найден контейнер с результатами, возвращаем пустой список
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error while fetching job elements: {e}")
|
||||
return []
|
||||
|
||||
def apply_jobs(self):
|
||||
try:
|
||||
no_jobs_element = self.driver.find_element(By.CLASS_NAME, 'jobs-search-two-pane__no-results-banner--expand')
|
||||
|
|
|
|||
15
src/utils.py
15
src/utils.py
|
|
@ -10,6 +10,11 @@ import logging
|
|||
logging.basicConfig(level=logging.DEBUG, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Отключаем логирование для selenium и urllib3
|
||||
logging.getLogger("selenium.webdriver.remote.remote_connection").setLevel(logging.WARNING)
|
||||
logging.getLogger("urllib3").setLevel(logging.WARNING)
|
||||
logging.getLogger("httpcore").setLevel(logging.WARNING)
|
||||
|
||||
|
||||
chromeProfilePath = os.path.join(os.getcwd(), "chrome_profile", "linkedin_profile")
|
||||
|
||||
|
|
@ -31,7 +36,7 @@ def is_scrollable(element):
|
|||
logger.debug("Element scrollable check: scrollHeight=%s, clientHeight=%s, scrollable=%s", scroll_height, client_height, scrollable)
|
||||
return scrollable
|
||||
|
||||
def scroll_slow(driver, scrollable_element, start=0, end=3600, step=100, reverse=False):
|
||||
def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse=False):
|
||||
logger.debug("Starting slow scroll: start=%d, end=%d, step=%d, reverse=%s", start, end, step, reverse)
|
||||
if reverse:
|
||||
start, end = end, start
|
||||
|
|
@ -39,6 +44,14 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=100, reverse
|
|||
if step == 0:
|
||||
logger.error("Step value cannot be zero.")
|
||||
raise ValueError("Step cannot be zero.")
|
||||
|
||||
max_scroll_height = int(scrollable_element.get_attribute("scrollHeight"))
|
||||
logger.debug("Max scroll height of the element: %d", max_scroll_height)
|
||||
|
||||
if end > max_scroll_height:
|
||||
logger.warning("End value exceeds the scroll height. Adjusting end to %d", max_scroll_height)
|
||||
end = max_scroll_height
|
||||
|
||||
script_scroll_to = "arguments[0].scrollTop = arguments[1];"
|
||||
try:
|
||||
if scrollable_element.is_displayed():
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue