Merge remote-tracking branch 'upstream/v3' into v3

This commit is contained in:
Manu Altieri 2024-09-10 21:24:34 +02:00
commit 40cf3e4d34
13 changed files with 725 additions and 308 deletions

165
.gitignore vendored
View file

@ -1,14 +1,155 @@
*.csv # Byte-compiled / optimized / DLL files
__pycache__/** __pycache__/
.idea/** *.py[cod]
open_ai_calls.log *$py.class
test*
openaiSelenium* # C extensions
open_ai_calls.json *.so
_*
# Distribution / packaging
.Python
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
pip-wheel-metadata/
share/python-wheels/
*.egg-info/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py,cover
.hypothesis/
.pytest_cache/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
_build/
# PyBuilder
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
.python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenvs dependency resolution may lead to different
# Pipfile.lock files generated on each colleagues machine.
# Thus, uncomment the following line if the pipenv environment is expected to be identical
# across all environments.
#Pipfile.lock
# PEP 582; used by e.g. github.com/David-OConnor/pyflow
__pypackages__/
# Celery stuff
celerybeat-schedule
celerybeat.pid
# SageMath parsed files
*.sage.py
# Environments
.env
.venv .venv
generated_cv* env/
.vscode venv/
chrome_profile ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.mypy_cache/
# PyCharm and all JetBrains IDEs
# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839
.idea/
*.iml
# Visual Studio Code
.vscode/
# Visual Studio 2015/2017/2019/2022
.vs/
*.opendb
*.VC.db
# User-specific files
*.suo
*.user
*.userosscache
*.sln.docstates
# Mono Auto Generated Files
mono_crash.*
# Project Specific
data_folder/output/*
generated_cv/*
chrome_profile/*
answers.json answers.json
data*

View file

@ -133,6 +133,11 @@ LinkedIn_AIHawk steps in as a game-changing solution to these challenges. It's n
source virtual/bin/activate source virtual/bin/activate
``` ```
or for Windows-based machines -
```bash
.\virtual\Scripts\activate
```
5. **Install the required packages:** 5. **Install the required packages:**
```bash ```bash
pip install -r requirements.txt pip install -r requirements.txt
@ -148,7 +153,7 @@ This file contains sensitive information. Never share or commit this file to ver
- Replace with your LinkedIn account email address - Replace with your LinkedIn account email address
- `password: [Your LinkedIn password]` - `password: [Your LinkedIn password]`
- Replace with your LinkedIn account password - Replace with your LinkedIn account password
- `llm_api_key: [Your OpenAI or Ollama API key]` - `llm_api_key: [Your OpenAI or Ollama API key or Gemini API key]`
- Replace with your OpenAI API key for GPT integration - Replace with your OpenAI API key for GPT integration
- To obtain an API key, follow the tutorial at: https://medium.com/@lorenzozar/how-to-get-your-own-openai-api-key-f4d44e60c327 - To obtain an API key, follow the tutorial at: https://medium.com/@lorenzozar/how-to-get-your-own-openai-api-key-f4d44e60c327
- Note: You need to add credit to your OpenAI account to use the API. You can add credit by visiting the [OpenAI billing dashboard](https://platform.openai.com/account/billing). - Note: You need to add credit to your OpenAI account to use the API. You can add credit by visiting the [OpenAI billing dashboard](https://platform.openai.com/account/billing).
@ -157,6 +162,7 @@ This file contains sensitive information. Never share or commit this file to ver
`{'error': {'message': 'Rate limit reached for gpt-4o-mini in organization <org> on requests per day (RPD): Limit 200, Used 200, Requested 1.}}` `{'error': {'message': 'Rate limit reached for gpt-4o-mini in organization <org> on requests per day (RPD): Limit 200, Used 200, Requested 1.}}`
OpenAI will update your account automatically, but it might take some time, ranging from a couple of hours to a few days. OpenAI will update your account automatically, but it might take some time, ranging from a couple of hours to a few days.
You can find more about your organization limits on the [official page](https://platform.openai.com/settings/organization/limits). You can find more about your organization limits on the [official page](https://platform.openai.com/settings/organization/limits).
- For obtaining Gemini API key visit [Google AI for Devs](https://ai.google.dev/gemini-api/docs/api-key)
### 2. config.yaml ### 2. config.yaml
@ -220,17 +226,19 @@ This file defines your job search parameters and bot behavior. Each section cont
#### 2.1 config.yaml - Customize LLM model endpoint #### 2.1 config.yaml - Customize LLM model endpoint
- `llm_model_type`: - `llm_model_type`:
- Choose the model type, supported: openai / ollama / claude - Choose the model type, supported: openai / ollama / claude / gemini
- `llm_model`: - `llm_model`:
- Choose the LLM model, currently supported: - Choose the LLM model, currently supported:
- openai: gpt-4o - openai: gpt-4o
- ollama: llama2, mistral:v0.3 - ollama: llama2, mistral:v0.3
- claude: any model - claude: any model
- gemini: any model
- `llm_api_url`: - `llm_api_url`:
- Link of the API endpoint for the LLM model - Link of the API endpoint for the LLM model
- openai: https://api.pawan.krd/cosmosrp/v1 - openai: https://api.pawan.krd/cosmosrp/v1
- ollama: http://127.0.0.1:11434/ - ollama: http://127.0.0.1:11434/
- claude: https://api.anthropic.com/v1 - claude: https://api.anthropic.com/v1
- gemini: no api_url
- Note: To run local Ollama, follow the guidelines here: [Guide to Ollama deployment](https://github.com/ollama/ollama) - Note: To run local Ollama, follow the guidelines here: [Guide to Ollama deployment](https://github.com/ollama/ollama)
### 3. plain_text_resume.yaml ### 3. plain_text_resume.yaml
@ -272,7 +280,8 @@ Each section has specific fields to fill out:
- This section outlines your academic background, including degrees earned and relevant coursework. - This section outlines your academic background, including degrees earned and relevant coursework.
- **degree**: The type of degree obtained (e.g., Bachelor's Degree, Master's Degree). - **degree**: The type of degree obtained (e.g., Bachelor's Degree, Master's Degree).
- **university**: The name of the university or institution where you studied. - **university**: The name of the university or institution where you studied.
- **gpa**: Your Grade Point Average or equivalent measure of academic performance. - **final_evaluation_grade**: Your Grade Point Average or equivalent measure of academic performance.
- **start_date**: The start year of your studies.
- **graduation_year**: The year you graduated. - **graduation_year**: The year you graduated.
- **field_of_study**: The major or focus area of your studies. - **field_of_study**: The major or focus area of your studies.
- **exam**: A list of courses or subjects taken along with their respective grades. - **exam**: A list of courses or subjects taken along with their respective grades.
@ -280,11 +289,12 @@ Each section has specific fields to fill out:
- Example: - Example:
```yaml ```yaml
education_details: education_details:
- degree: "Bachelor's Degree" - education_level: "Bachelor's Degree"
university: "University of Example" institution: "University of Example"
gpa: "3.8/4"
graduation_year: "2022"
field_of_study: "Software Engineering" field_of_study: "Software Engineering"
final_evaluation_grade: "4/4"
start_date: "2021"
year_of_completion: "2023"
exam: exam:
Algorithms: "A" Algorithms: "A"
Data Structures: "B+" Data Structures: "B+"
@ -354,7 +364,8 @@ Each section has specific fields to fill out:
- `certifications:` - `certifications:`
- Include any professional certifications you have earned. - Include any professional certifications you have earned.
- **certification_name**: The name of the certification. - name: "PMP"
description: "Certification for project management professionals, issued by the Project Management Institute (PMI)"
- Example: - Example:
```yaml ```yaml

View file

@ -1,6 +1,6 @@
remote: [true/false] remote: [true/false]
experienceLevel: experience_level:
internship: [true/false] internship: [true/false]
entry: [true/false] entry: [true/false]
associate: [true/false] associate: [true/false]
@ -31,18 +31,22 @@ locations:
- Country1 - Country1
- Country2 - Country2
applyOnceAtCompany: [true/false] apply_once_at_company: [true/false]
distance: 100 distance: 100
companyBlacklist: company_blacklist:
- Company1 - Company1
- Company2 - Company2
titleBlacklist: title_blacklist:
- word1 - word1
- word2 - word2
job_applicants_threshold:
min_applicants: 0
max_applicants: 100
llm_model_type: openai llm_model_type: openai
llm_model: gpt-4o llm_model: gpt-4o
llm_api_url: https://api.pawan.krd/cosmosrp/v1 llm_api_url: https://api.pawan.krd/cosmosrp/v1

View file

@ -1,7 +1,7 @@
personal_information: personal_information:
name: "[Your Name]" name: "[Your Name]"
surname: "[Your Surname]" surname: "[Your Surname]"
date_of_birth: "[DD/MM/YYYY]" date_of_birth: "[Your Date of Birth]"
country: "[Your Country]" country: "[Your Country]"
city: "[Your City]" city: "[Your City]"
address: "[Your Address]" address: "[Your Address]"
@ -12,74 +12,92 @@ personal_information:
linkedin: "[Your LinkedIn Profile URL]" linkedin: "[Your LinkedIn Profile URL]"
education_details: education_details:
- degree: "[Your Degree]" - education_level: "[Your Education Level]"
university: "[Your University]" institution: "[Your Institution]"
gpa: "[Your GPA]"
graduation_year: "[Year of Graduation]"
field_of_study: "[Your Field of Study]" field_of_study: "[Your Field of Study]"
final_evaluation_grade: "[Your Final Evaluation Grade]"
start_date: "[Start Date]"
year_of_completion: "[Year of Completion]"
exam: exam:
[Course Name 1]: "[Grade]" exam_name_1: "[Grade]"
[Course Name 2]: "[Grade]" exam_name_2: "[Grade]"
[Course Name 3]: "[Grade]" exam_name_3: "[Grade]"
[Course Name 4]: "[Grade]" exam_name_4: "[Grade]"
[Course Name 5]: "[Grade]" exam_name_5: "[Grade]"
exam_name_6: "[Grade]"
experience_details: experience_details:
- position: "[Your Job Title]" - position: "[Your Position]"
company: "[Company Name]" company: "[Company Name]"
employment_period: "[Start Date] - [End Date]" employment_period: "[Employment Period]"
location: "[Location]" location: "[Location]"
industry: "[Industry]" industry: "[Industry]"
key_responsibilities: key_responsibilities:
- responsibility_1: "[Key Responsibility 1]" - responsibility_1: "[Responsibility Description]"
- responsibility_2: "[Key Responsibility 2]" - responsibility_2: "[Responsibility Description]"
- responsibility_3: "[Key Responsibility 3]" - responsibility_3: "[Responsibility Description]"
skills_acquired: skills_acquired:
- "[Skill 1]" - "[Skill]"
- "[Skill 2]" - "[Skill]"
- "[Skill 3]" - "[Skill]"
- position: "[Your Position]"
company: "[Company Name]"
employment_period: "[Employment Period]"
location: "[Location]"
industry: "[Industry]"
key_responsibilities:
- responsibility_1: "[Responsibility Description]"
- responsibility_2: "[Responsibility Description]"
- responsibility_3: "[Responsibility Description]"
skills_acquired:
- "[Skill]"
- "[Skill]"
- "[Skill]"
projects: projects:
- name: "[Project Name]" - name: "[Project Name]"
description: "[Brief Description of the Project]" description: "[Project Description]"
link: "[Project URL]" link: "[Project Link]"
- name: "[Project Name]" - name: "[Project Name]"
description: "[Brief Description of the Project]" description: "[Project Description]"
link: "[Project URL]" link: "[Project Link]"
achievements: achievements:
- name: "[Achievement Title]" - name: "[Achievement Name]"
description: "[Brief Description of the Achievement]" description: "[Achievement Description]"
- name: "[Achievement Title]" - name: "[Achievement Name]"
description: "[Brief Description of the Achievement]" description: "[Achievement Description]"
certifications: certifications:
- "[Certification Name]" - name: "[Certification Name]"
description: "[Certification Description]"
- name: "[Certification Name]"
description: "[Certification Description]"
languages: languages:
- language: "[Language Name]" - language: "[Language]"
proficiency: "[Proficiency Level]" proficiency: "[Proficiency Level]"
- language: "[Language Name]" - language: "[Language]"
proficiency: "[Proficiency Level]" proficiency: "[Proficiency Level]"
interests: interests:
- "[Interest 1]" - "[Interest]"
- "[Interest 2]" - "[Interest]"
- "[Interest 3]" - "[Interest]"
- "[Interest 4]"
- "[Interest 5]"
availability: availability:
notice_period: "[Notice Period]" notice_period: "[Notice Period]"
salary_expectations: salary_expectations:
salary_range_usd: "[Expected Salary Range in USD]" salary_range_usd: "[Salary Range]"
self_identification: self_identification:
gender: "[Gender]" gender: "[Gender]"
pronouns: "[Pronouns]" pronouns: "[Pronouns]"
veteran: "[Veteran Status]" veteran: "[Yes/No]"
disability: "[Disability Status]" disability: "[Yes/No]"
ethnicity: "[Ethnicity]" ethnicity: "[Ethnicity]"
legal_authorization: legal_authorization:

View file

@ -1,6 +1,6 @@
remote: true remote: true
experienceLevel: experience_level:
internship: true internship: true
entry: true entry: true
associate: true associate: true
@ -29,15 +29,21 @@ positions:
locations: locations:
- USA - USA
applyOnceAtCompany: [true/false] apply_once_at_company: [true/false]
distance: 100 distance: 100
companyBlacklist: company_blacklist:
- Noir - Noir
- Crossover - Crossover
titleBlacklist: title_blacklist:
- word1
- word2
job_applicants_threshold:
min_applicants: 0
max_applicants: 100
llm_model_type: openai llm_model_type: openai
llm_model: 'gpt-4o' llm_model: 'gpt-4o'

View file

@ -1,118 +1,124 @@
personal_information: personal_information:
name: "Liam" name: "Giovanni"
surname: "Murphy" surname: "Bianchi"
date_of_birth: "15/08/1995" date_of_birth: "12/02/1988"
country: "Ireland" country: "Italy"
city: "Galway" city: "Rome"
address: "Galway City Center" address: "Via Nazionale, 45"
phone_prefix: "+353" phone_prefix: "+39"
phone: "871234567" phone: "3345678901"
email: "liam.murphy@gmail.com" email: "giovanni.bianchi@example.com"
github: "https://github.com/liam-murphy" github: "https://github.com/giovanni-bianchi"
linkedin: "https://www.linkedin.com/in/liam-murphy/" linkedin: "https://www.linkedin.com/in/giovanni-bianchi/"
education_details: education_details:
- degree: "Bachelor's Degree" - education_level: "Master's Degree"
university: "National University of Ireland, Galway" institution: "University of Rome"
gpa: "4/4" field_of_study: "Computer Engineering"
graduation_year: "2020" final_evaluation_grade: "110/110"
field_of_study: "Computer Science" start_date: "2011"
year_of_completion: "2013"
exam: exam:
Information Theory and Inference: "4" Computer Networks: "30/30"
Algorithm Analysis and Design: "4" Advanced Algorithms: "30/30"
Object-Oriented Languages and Programming: "4" Database Systems: "30/30"
Linear Algebra and Numerical Analysis: "4" Embedded Systems: "30/30"
Database: "4" Artificial Intelligence: "30/30"
experience_details: experience_details:
- position: "Co-Founder & Software Engineer" - position: "Senior Software Engineer"
company: "CryptoWave Solutions" company: "TechSolutions"
employment_period: "03/2021 - Present" employment_period: "01/2018 - Present"
location: "Ireland" location: "Rome, Italy"
industry: "Blockchain Technology" industry: "Software Development"
key_responsibilities: key_responsibilities:
- responsibility_1: "Co-founded and led a startup specializing in app and software development with a focus on blockchain technology" - responsibility_1: "Led a team of developers in designing and implementing enterprise software solutions"
- responsibility_2: "Provided blockchain consultations for 10+ companies, enhancing their software capabilities with secure, decentralized solutions" - responsibility_2: "Architected scalable systems to handle high-volume data processing"
- responsibility_3: "Developed blockchain applications, integrated cutting-edge technology to meet client needs and drive industry innovation" - responsibility_3: "Optimized application performance and reduced downtime by 20%"
skills_acquired: skills_acquired:
- "Blockchain development" - "Software architecture"
- "Software engineering" - "Team leadership"
- "Consultancy" - "Performance optimization"
- position: "Research Intern" - position: "Software Developer"
company: "National University of Ireland, Galway" company: "Innovatech"
employment_period: "11/2022 - 03/2023" employment_period: "06/2015 - 12/2017"
location: "Galway, Ireland" location: "Milan, Italy"
industry: "IoT Security Research" industry: "Technology"
key_responsibilities: key_responsibilities:
- responsibility_1: "Conducted in-depth research on IoT security, focusing on binary instrumentation and runtime monitoring" - responsibility_1: "Developed and maintained web applications using modern technologies"
- responsibility_2: "Performed in-depth study of the MQTT protocol and Falco" - responsibility_2: "Collaborated with UX/UI designers to enhance user experience"
- responsibility_3: "Developed multiple software components including MQTT packet analysis library, Falco adapter, and RML monitor in Prolog" - responsibility_3: "Implemented automated testing procedures to ensure code quality"
- responsibility_4: "Authored thesis 'Binary Instrumentation for Runtime Monitoring of Internet of Things Systems Using Falco'"
skills_acquired: skills_acquired:
- "IoT security" - "Web development"
- "Binary instrumentation" - "User experience design"
- "MQTT protocol" - "Automated testing"
- "Prolog programming"
- position: "Software Engineer" - position: "Junior Developer"
company: "University Hospital Galway" company: "StartUp Hub"
employment_period: "05/2022 - 11/2022" employment_period: "01/2014 - 05/2015"
location: "Galway, Ireland" location: "Florence, Italy"
industry: "Healthcare IT" industry: "Startups"
key_responsibilities: key_responsibilities:
- responsibility_1: "Integrated and enforced robust security protocols" - responsibility_1: "Assisted in the development of mobile applications and web platforms"
- responsibility_2: "Developed and maintained a critical software tool for password validation used by over 1,600 employees" - responsibility_2: "Participated in code reviews and contributed to software design discussions"
- responsibility_3: "Played an integral role in the hospital's cybersecurity team" - responsibility_3: "Resolved bugs and implemented feature enhancements"
skills_acquired: skills_acquired:
- "Cybersecurity" - "Mobile app development"
- "Software development" - "Code reviews"
- "Password validation" - "Bug fixing"
projects: projects:
- name: "JobBot" - name: "E-Commerce Platform"
description: "AI-driven tool to automate and personalize job applications on LinkedIn, gained over 3000 stars on GitHub, improving efficiency and reducing application time" description: "Developed a scalable e-commerce platform with advanced features like real-time inventory tracking and user analytics"
link: "https://github.com/liam-murphy/jobbot" link: "https://github.com/giovanni-bianchi/ecommerce-platform"
- name: "mqtt-packet-parser" - name: "Smart Home Automation"
description: "Developed a Node.js module for parsing MQTT packets, improved parsing efficiency by 40%" description: "Created a smart home automation system integrating various IoT devices for remote control and monitoring"
link: "https://github.com/liam-murphy/mqtt-packet-parser" link: "https://github.com/giovanni-bianchi/smart-home-automation"
achievements: achievements:
- name: "Winner of an Irish public competition" - name: "Top Innovator Award"
description: "Won first place in a public competition with a perfect score of 70/70, securing a Software Developer position at University Hospital Galway" description: "Recognized for innovative solutions and contributions to high-impact projects at TechSolutions"
- name: "Galway Merit Scholarship" - name: "Best Young Developer"
description: "Awarded annually from 2018 to 2020 in recognition of academic excellence and contribution" description: "Awarded for outstanding performance and contributions during the first three years at Innovatech"
- name: "GitHub Recognition"
description: "Gained over 3000 stars on GitHub with JobBot project"
certifications: certifications:
- "C1" - name: "Certified Ethical Hacker (CEH)"
description: "Certification for expertise in ethical hacking and cybersecurity practices"
- name: "AWS Certified DevOps Engineer"
description: "Certification for DevOps practices and using AWS for cloud services"
- name: "Microsoft Certified: Azure Solutions Architect Expert"
description: "Certification for designing and implementing Azure solutions"
- name: "Certified Kubernetes Administrator (CKA)"
description: "Certification for managing and orchestrating Kubernetes clusters"
- name: "Certified Data Privacy Professional (CDPP)"
description: "Certification for ensuring data privacy and compliance with regulations"
languages: languages:
- language: "English" - language: "Italian"
proficiency: "Native" proficiency: "Native"
- language: "Spanish" - language: "English"
proficiency: "Professional" proficiency: "Fluent"
interests: interests:
- "Full-Stack Development" - "Cloud Computing"
- "Software Architecture" - "Cybersecurity"
- "IoT system design and development" - "IoT Development"
- "Artificial Intelligence" - "Artificial Intelligence"
- "Cloud Technologies" - "Data Privacy"
availability: availability:
notice_period: "immediately" notice_period: "2 months"
salary_expectations: salary_expectations:
salary_range_usd: "100000" salary_range_usd: "90000 - 110000"
self_identification: self_identification:
gender: "Male" gender: "Male"
pronouns: "He" pronouns: "He/Him"
veteran: "No" veteran: "No"
disability: "No" disability: "No"
ethnicity: "white" ethnicity: "White"
legal_authorization: legal_authorization:
eu_work_authorization: "Yes" eu_work_authorization: "Yes"

View file

@ -9,7 +9,7 @@ from selenium.webdriver.chrome.service import Service as ChromeService
from webdriver_manager.chrome import ChromeDriverManager from webdriver_manager.chrome import ChromeDriverManager
from selenium.common.exceptions import WebDriverException, TimeoutException from selenium.common.exceptions import WebDriverException, TimeoutException
from lib_resume_builder_AIHawk import Resume,StyleManager,FacadeManager,ResumeGenerator from lib_resume_builder_AIHawk import Resume,StyleManager,FacadeManager,ResumeGenerator
from src.utils import chromeBrowserOptions from src.utils import chrome_browser_options
from src.gpt import GPTAnswerer from src.gpt import GPTAnswerer
from src.linkedIn_authenticator import LinkedInAuthenticator from src.linkedIn_authenticator import LinkedInAuthenticator
from src.linkedIn_bot_facade import LinkedInBotFacade from src.linkedIn_bot_facade import LinkedInBotFacade
@ -149,7 +149,7 @@ class FileManager:
def init_browser() -> webdriver.Chrome: def init_browser() -> webdriver.Chrome:
try: try:
options = chromeBrowserOptions() options = chrome_browser_options()
service = ChromeService(ChromeDriverManager().install()) service = ChromeService(ChromeDriverManager().install())
return webdriver.Chrome(service=service, options=options) return webdriver.Chrome(service=service, options=options)
except Exception as e: except Exception as e:

View file

@ -14,3 +14,12 @@ click
git+https://github.com/feder-cr/lib_resume_builder_AIHawk.git git+https://github.com/feder-cr/lib_resume_builder_AIHawk.git
linkedin-api linkedin-api
pdfminer.six==20221105 pdfminer.six==20221105
inputimeout==1.0.4
langchain-ollama==0.1.3
langchain-anthropic==0.1.3
langchain-google-genai==1.0.10
jsonschema==4.23.0
jsonschema-specifications==2023.12.1
httpx~=0.27.2
python-dotenv~=1.0.1
PyYAML~=6.0.2

View file

@ -3,11 +3,11 @@ import os
import re import re
import textwrap import textwrap
import time import time
from datetime import datetime
from abc import ABC, abstractmethod from abc import ABC, abstractmethod
from typing import Dict, List, Union from datetime import datetime
from pathlib import Path from pathlib import Path
from typing import Dict, List from typing import Dict, List
from typing import Union
import httpx import httpx
from Levenshtein import distance from Levenshtein import distance
@ -16,18 +16,19 @@ from langchain_core.messages.ai import AIMessage
from langchain_core.output_parsers import StrOutputParser from langchain_core.output_parsers import StrOutputParser
from langchain_core.prompt_values import StringPromptValue from langchain_core.prompt_values import StringPromptValue
from langchain_core.prompts import ChatPromptTemplate from langchain_core.prompts import ChatPromptTemplate
from langchain_openai import ChatOpenAI
import src.strings as strings import src.strings as strings
from src.utils import logger from src.utils import logger
load_dotenv() load_dotenv()
class AIModel(ABC): class AIModel(ABC):
@abstractmethod @abstractmethod
def invoke(self, prompt: str) -> str: def invoke(self, prompt: str) -> str:
pass pass
class OpenAIModel(AIModel): class OpenAIModel(AIModel):
def __init__(self, api_key: str, llm_model: str, llm_api_url: str): def __init__(self, api_key: str, llm_model: str, llm_api_url: str):
from langchain_openai import ChatOpenAI from langchain_openai import ChatOpenAI
@ -39,16 +40,18 @@ class OpenAIModel(AIModel):
response = self.model.invoke(prompt) response = self.model.invoke(prompt)
return response return response
class ClaudeModel(AIModel): class ClaudeModel(AIModel):
def __init__(self, api_key: str, llm_model: str, llm_api_url: str): def __init__(self, api_key: str, llm_model: str, llm_api_url: str):
from langchain_anthropic import ChatAnthropic from langchain_anthropic import ChatAnthropic
self.model = ChatAnthropic(model=llm_model, api_key=api_key, self.model = ChatAnthropic(model=llm_model, api_key=api_key,
temperature=0.4, base_url=llm_api_url) temperature=0.4, base_url=llm_api_url)
def invoke(self, prompt: str) -> str: def invoke(self, prompt: str) -> str:
response = self.model.invoke(prompt) response = self.model.invoke(prompt)
return response return response
class OllamaModel(AIModel): class OllamaModel(AIModel):
def __init__(self, api_key: str, llm_model: str, llm_api_url: str): def __init__(self, api_key: str, llm_model: str, llm_api_url: str):
from langchain_ollama import ChatOllama from langchain_ollama import ChatOllama
@ -58,6 +61,17 @@ class OllamaModel(AIModel):
response = self.model.invoke(prompt) response = self.model.invoke(prompt)
return response return response
class GeminiModel(AIModel):
def __init__(self, api_key:str, llm_model: str, llm_api_url: str):
from langchain_google_genai import ChatGoogleGenerativeAI
self.model = ChatGoogleGenerativeAI(model=llm_model, google_api_key=api_key)
def invoke(self, prompt: str) -> str:
response = self.model.invoke(prompt)
return response
class AIAdapter: class AIAdapter:
def __init__(self, config: dict, api_key: str): def __init__(self, config: dict, api_key: str):
self.model = self._create_model(config, api_key) self.model = self._create_model(config, api_key)
@ -66,7 +80,8 @@ class AIAdapter:
llm_model_type = config['llm_model_type'] llm_model_type = config['llm_model_type']
llm_model = config['llm_model'] llm_model = config['llm_model']
llm_api_url = config['llm_api_url'] llm_api_url = config['llm_api_url']
print('Using {0} with {1} from {2}'.format(llm_model_type, llm_model, llm_api_url)) print('Using {0} with {1} from {2}'.format(
llm_model_type, llm_model, llm_api_url))
if llm_model_type == "openai": if llm_model_type == "openai":
return OpenAIModel(api_key, llm_model, llm_api_url) return OpenAIModel(api_key, llm_model, llm_api_url)
@ -74,16 +89,18 @@ class AIAdapter:
return ClaudeModel(api_key, llm_model, llm_api_url) return ClaudeModel(api_key, llm_model, llm_api_url)
elif llm_model_type == "ollama": elif llm_model_type == "ollama":
return OllamaModel(api_key, llm_model, llm_api_url) return OllamaModel(api_key, llm_model, llm_api_url)
elif llm_model_type == "gemini":
return GeminiModel(api_key, llm_model, llm_api_url)
else: else:
raise ValueError(f"Unsupported model type: {model_type}") raise ValueError(f"Unsupported model type: {llm_model_type}")
def invoke(self, prompt: str) -> str: def invoke(self, prompt: str) -> str:
return self.model.invoke(prompt) return self.model.invoke(prompt)
class LLMLogger: class LLMLogger:
def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel, GeminiModel]):
def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]):
self.llm = llm self.llm = llm
logger.debug("LLMLogger successfully initialized with LLM: %s", llm) logger.debug("LLMLogger successfully initialized with LLM: %s", llm)
@ -95,7 +112,8 @@ class LLMLogger:
logger.debug("Parsed reply received: %s", parsed_reply) logger.debug("Parsed reply received: %s", parsed_reply)
try: try:
calls_log = os.path.join(Path("data_folder/output"), "open_ai_calls.json") calls_log = os.path.join(
Path("data_folder/output"), "open_ai_calls.json")
logger.debug("Logging path determined: %s", calls_log) logger.debug("Logging path determined: %s", calls_log)
except Exception as e: except Exception as e:
logger.error("Error determining the log path: %s", str(e)) logger.error("Error determining the log path: %s", str(e))
@ -114,18 +132,22 @@ class LLMLogger:
} }
logger.debug("Prompts converted to dictionary: %s", prompts) logger.debug("Prompts converted to dictionary: %s", prompts)
except Exception as e: except Exception as e:
logger.error("Error converting prompts to dictionary: %s", str(e)) logger.error(
"Error converting prompts to dictionary: %s", str(e))
raise raise
else: else:
logger.debug("Prompts are of unknown type, attempting default conversion") logger.debug(
"Prompts are of unknown type, attempting default conversion")
try: try:
prompts = { prompts = {
f"prompt_{i + 1}": prompt.content f"prompt_{i + 1}": prompt.content
for i, prompt in enumerate(prompts.messages) for i, prompt in enumerate(prompts.messages)
} }
logger.debug("Prompts converted to dictionary using default method: %s", prompts) logger.debug(
"Prompts converted to dictionary using default method: %s", prompts)
except Exception as e: except Exception as e:
logger.error("Error converting prompts using default method: %s", str(e)) logger.error(
"Error converting prompts using default method: %s", str(e))
raise raise
try: try:
@ -140,7 +162,8 @@ class LLMLogger:
output_tokens = token_usage["output_tokens"] output_tokens = token_usage["output_tokens"]
input_tokens = token_usage["input_tokens"] input_tokens = token_usage["input_tokens"]
total_tokens = token_usage["total_tokens"] total_tokens = token_usage["total_tokens"]
logger.debug("Token usage - Input: %d, Output: %d, Total: %d", input_tokens, output_tokens, total_tokens) logger.debug("Token usage - Input: %d, Output: %d, Total: %d",
input_tokens, output_tokens, total_tokens)
except KeyError as e: except KeyError as e:
logger.error("KeyError in parsed_reply structure: %s", str(e)) logger.error("KeyError in parsed_reply structure: %s", str(e))
raise raise
@ -155,7 +178,8 @@ class LLMLogger:
try: try:
prompt_price_per_token = 0.00000015 prompt_price_per_token = 0.00000015
completion_price_per_token = 0.0000006 completion_price_per_token = 0.0000006
total_cost = (input_tokens * prompt_price_per_token) + (output_tokens * completion_price_per_token) total_cost = (input_tokens * prompt_price_per_token) + \
(output_tokens * completion_price_per_token)
logger.debug("Total cost calculated: %f", total_cost) logger.debug("Total cost calculated: %f", total_cost)
except Exception as e: except Exception as e:
logger.error("Error calculating total cost: %s", str(e)) logger.error("Error calculating total cost: %s", str(e))
@ -174,12 +198,14 @@ class LLMLogger:
} }
logger.debug("Log entry created: %s", log_entry) logger.debug("Log entry created: %s", log_entry)
except KeyError as e: except KeyError as e:
logger.error("Error creating log entry: missing key %s in parsed_reply", str(e)) logger.error(
"Error creating log entry: missing key %s in parsed_reply", str(e))
raise raise
try: try:
with open(calls_log, "a", encoding="utf-8") as f: with open(calls_log, "a", encoding="utf-8") as f:
json_string = json.dumps(log_entry, ensure_ascii=False, indent=4) json_string = json.dumps(
log_entry, ensure_ascii=False, indent=4)
f.write(json_string + "\n") f.write(json_string + "\n")
logger.debug("Log entry written to file: %s", calls_log) logger.debug("Log entry written to file: %s", calls_log)
except Exception as e: except Exception as e:
@ -189,25 +215,25 @@ class LLMLogger:
class LoggerChatModel: class LoggerChatModel:
def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel, GeminiModel]):
def __init__(self, llm: Union[OpenAIModel, OllamaModel, ClaudeModel]):
self.llm = llm self.llm = llm
logger.debug("LoggerChatModel successfully initialized with LLM: %s", llm) logger.debug(
"LoggerChatModel successfully initialized with LLM: %s", llm)
def __call__(self, messages: List[Dict[str, str]]) -> str: def __call__(self, messages: List[Dict[str, str]]) -> str:
logger.debug("Entering __call__ method with messages: %s", messages) logger.debug("Entering __call__ method with messages: %s", messages)
while True: while True:
try: try:
logger.debug("Attempting to call the LLM with messages") logger.debug("Attempting to call the LLM with messages")
reply = self.llm(messages)
reply = self.llm.invoke(messages)
logger.debug("LLM response received: %s", reply) logger.debug("LLM response received: %s", reply)
parsed_reply = self.parse_llmresult(reply) parsed_reply = self.parse_llmresult(reply)
logger.debug("Parsed LLM reply: %s", parsed_reply) logger.debug("Parsed LLM reply: %s", parsed_reply)
LLMLogger.log_request(prompts=messages, parsed_reply=parsed_reply) LLMLogger.log_request(
prompts=messages, parsed_reply=parsed_reply)
logger.debug("Request successfully logged") logger.debug("Request successfully logged")
return reply return reply
@ -243,11 +269,11 @@ class LoggerChatModel:
except Exception as e: except Exception as e:
logger.error("Unexpected error occurred: %s", str(e)) logger.error("Unexpected error occurred: %s", str(e))
logger.info("Waiting for 30 seconds before retrying due to an unexpected error.") logger.info(
"Waiting for 30 seconds before retrying due to an unexpected error.")
time.sleep(30) time.sleep(30)
continue continue
def parse_llmresult(self, llmresult: AIMessage) -> Dict[str, Dict]: def parse_llmresult(self, llmresult: AIMessage) -> Dict[str, Dict]:
logger.debug("Parsing LLM result: %s", llmresult) logger.debug("Parsing LLM result: %s", llmresult)
@ -277,11 +303,13 @@ class LoggerChatModel:
return parsed_result return parsed_result
except KeyError as e: except KeyError as e:
logger.error("KeyError while parsing LLM result: missing key %s", str(e)) logger.error(
"KeyError while parsing LLM result: missing key %s", str(e))
raise raise
except Exception as e: except Exception as e:
logger.error("Unexpected error while parsing LLM result: %s", str(e)) logger.error(
"Unexpected error while parsing LLM result: %s", str(e))
raise raise
@ -297,7 +325,8 @@ class GPTAnswerer:
@staticmethod @staticmethod
def find_best_match(text: str, options: list[str]) -> str: def find_best_match(text: str, options: list[str]) -> str:
logger.debug("Finding best match for text: '%s' in options: %s", text, options) logger.debug(
"Finding best match for text: '%s' in options: %s", text, options)
distances = [ distances = [
(option, distance(text.lower(), option.lower())) for option in options (option, distance(text.lower(), option.lower())) for option in options
] ]
@ -323,10 +352,12 @@ class GPTAnswerer:
def set_job(self, job): def set_job(self, job):
logger.debug("Setting job: %s", job) logger.debug("Setting job: %s", job)
self.job = job self.job = job
self.job.set_summarize_job_description(self.summarize_job_description(self.job.description)) self.job.set_summarize_job_description(
self.summarize_job_description(self.job.description))
def set_job_application_profile(self, job_application_profile): def set_job_application_profile(self, job_application_profile):
logger.debug("Setting job application profile: %s", job_application_profile) logger.debug("Setting job application profile: %s",
job_application_profile)
self.job_application_profile = job_application_profile self.job_application_profile = job_application_profile
def summarize_job_description(self, text: str) -> str: def summarize_job_description(self, text: str) -> str:
@ -334,7 +365,8 @@ class GPTAnswerer:
strings.summarize_prompt_template = self._preprocess_template_string( strings.summarize_prompt_template = self._preprocess_template_string(
strings.summarize_prompt_template strings.summarize_prompt_template
) )
prompt = ChatPromptTemplate.from_template(strings.summarize_prompt_template) prompt = ChatPromptTemplate.from_template(
strings.summarize_prompt_template)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output = chain.invoke({"text": text}) output = chain.invoke({"text": text})
logger.debug("Summary generated: %s", output) logger.debug("Summary generated: %s", output)
@ -454,33 +486,40 @@ class GPTAnswerer:
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output = chain.invoke({"question": question}) output = chain.invoke({"question": question})
match = re.search(r"(Personal information|Self Identification|Legal Authorization|Work Preferences|Education Details|Experience Details|Projects|Availability|Salary Expectations|Certifications|Languages|Interests|Cover letter)", output, re.IGNORECASE) match = re.search(
r"(Personal information|Self Identification|Legal Authorization|Work Preferences|Education Details|Experience Details|Projects|Availability|Salary Expectations|Certifications|Languages|Interests|Cover letter)",
output, re.IGNORECASE)
if not match: if not match:
raise ValueError("Could not extract section name from the response.") raise ValueError(
"Could not extract section name from the response.")
section_name = match.group(1).lower().replace(" ", "_") section_name = match.group(1).lower().replace(" ", "_")
if section_name == "cover_letter": if section_name == "cover_letter":
chain = chains.get(section_name) chain = chains.get(section_name)
output = chain.invoke({"resume": self.resume, "job_description": self.job_description}) output = chain.invoke(
{"resume": self.resume, "job_description": self.job_description})
logger.debug("Cover letter generated: %s", output) logger.debug("Cover letter generated: %s", output)
return output return output
resume_section = getattr(self.resume, section_name, None) or getattr(self.job_application_profile, section_name, resume_section = getattr(self.resume, section_name, None) or getattr(self.job_application_profile, section_name,
None) None)
if resume_section is None: if resume_section is None:
logger.error("Section '%s' not found in either resume or job_application_profile.", section_name) logger.error(
"Section '%s' not found in either resume or job_application_profile.", section_name)
raise ValueError(f"Section '{section_name}' not found in either resume or job_application_profile.") raise ValueError(f"Section '{section_name}' not found in either resume or job_application_profile.")
chain = chains.get(section_name) chain = chains.get(section_name)
if chain is None: if chain is None:
logger.error("Chain not defined for section '%s'", section_name) logger.error("Chain not defined for section '%s'", section_name)
raise ValueError(f"Chain not defined for section '{section_name}'") raise ValueError(f"Chain not defined for section '{section_name}'")
output = chain.invoke({"resume_section": resume_section, "question": question}) output = chain.invoke(
{"resume_section": resume_section, "question": question})
logger.debug("Question answered: %s", output) logger.debug("Question answered: %s", output)
return output return output
def answer_question_numeric(self, question: str, default_experience: int = 3) -> int: def answer_question_numeric(self, question: str, default_experience: int = 3) -> int:
logger.debug("Answering numeric question: %s", question) logger.debug("Answering numeric question: %s", question)
func_template = self._preprocess_template_string(strings.numeric_question_template) func_template = self._preprocess_template_string(
strings.numeric_question_template)
prompt = ChatPromptTemplate.from_template(func_template) prompt = ChatPromptTemplate.from_template(func_template)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output_str = chain.invoke( output_str = chain.invoke(
@ -491,7 +530,8 @@ class GPTAnswerer:
output = self.extract_number_from_string(output_str) output = self.extract_number_from_string(output_str)
logger.debug("Extracted number: %d", output) logger.debug("Extracted number: %d", output)
except ValueError: except ValueError:
logger.warning("Failed to extract number, using default experience: %d", default_experience) logger.warning(
"Failed to extract number, using default experience: %d", default_experience)
output = default_experience output = default_experience
return output return output
@ -507,17 +547,20 @@ class GPTAnswerer:
def answer_question_from_options(self, question: str, options: list[str]) -> str: def answer_question_from_options(self, question: str, options: list[str]) -> str:
logger.debug("Answering question from options: %s", question) logger.debug("Answering question from options: %s", question)
func_template = self._preprocess_template_string(strings.options_template) func_template = self._preprocess_template_string(
strings.options_template)
prompt = ChatPromptTemplate.from_template(func_template) prompt = ChatPromptTemplate.from_template(func_template)
chain = prompt | self.llm_cheap | StrOutputParser() chain = prompt | self.llm_cheap | StrOutputParser()
output_str = chain.invoke({"resume": self.resume, "question": question, "options": options}) output_str = chain.invoke(
{"resume": self.resume, "question": question, "options": options})
logger.debug("Raw output for options question: %s", output_str) logger.debug("Raw output for options question: %s", output_str)
best_option = self.find_best_match(output_str, options) best_option = self.find_best_match(output_str, options)
logger.debug("Best option determined: %s", best_option) logger.debug("Best option determined: %s", best_option)
return best_option return best_option
def resume_or_cover(self, phrase: str) -> str: def resume_or_cover(self, phrase: str) -> str:
logger.debug("Determining if phrase refers to resume or cover letter: %s", phrase) logger.debug(
"Determining if phrase refers to resume or cover letter: %s", phrase)
prompt_template = """ prompt_template = """
Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter. Given the following phrase, respond with only 'resume' if the phrase is about a resume, or 'cover' if it's about a cover letter.
If the phrase contains only one word 'upload', consider it as 'cover'. If the phrase contains only one word 'upload', consider it as 'cover'.

View file

@ -8,7 +8,7 @@ import traceback
from typing import List, Optional, Any, Tuple from typing import List, Optional, Any, Tuple
from httpx import HTTPStatusError from httpx import HTTPStatusError
from reportlab.lib.pagesizes import letter from reportlab.lib.pagesizes import A4
from reportlab.pdfgen import canvas from reportlab.pdfgen import canvas
from selenium.common.exceptions import NoSuchElementException, TimeoutException from selenium.common.exceptions import NoSuchElementException, TimeoutException
from selenium.webdriver import ActionChains from selenium.webdriver import ActionChains
@ -37,7 +37,6 @@ class LinkedInEasyApplier:
logger.debug("LinkedInEasyApplier initialized successfully") logger.debug("LinkedInEasyApplier initialized successfully")
def _load_questions_from_json(self) -> List[dict]: def _load_questions_from_json(self) -> List[dict]:
output_file = 'answers.json' output_file = 'answers.json'
logger.debug("Loading questions from JSON file: %s", output_file) logger.debug("Loading questions from JSON file: %s", output_file)
@ -60,10 +59,8 @@ class LinkedInEasyApplier:
logger.error("Error loading questions data from JSON file: %s", tb_str) logger.error("Error loading questions data from JSON file: %s", tb_str)
raise Exception(f"Error loading questions data from JSON file: \nTraceback:\n{tb_str}") raise Exception(f"Error loading questions data from JSON file: \nTraceback:\n{tb_str}")
def check_for_premium_redirect(self, job: Any, max_attempts=3): def check_for_premium_redirect(self, job: Any, max_attempts=3):
"""Проверяет, был ли выполнен редирект на страницу LinkedIn Premium.
В случае редиректа возвращает пользователя на исходную страницу вакансии."""
current_url = self.driver.current_url current_url = self.driver.current_url
attempts = 0 attempts = 0
@ -80,7 +77,6 @@ class LinkedInEasyApplier:
raise Exception( raise Exception(
f"Redirected to LinkedIn Premium page and failed to return after {max_attempts} attempts. Job application aborted.") f"Redirected to LinkedIn Premium page and failed to return after {max_attempts} attempts. Job application aborted.")
def job_apply(self, job: Any): def job_apply(self, job: Any):
logger.debug("Starting job application for job: %s", job) logger.debug("Starting job application for job: %s", job)
@ -167,7 +163,7 @@ class LinkedInEasyApplier:
logger.debug(f"Attempting search using {method['description']}") logger.debug(f"Attempting search using {method['description']}")
if method.get('find_elements'): if method.get('find_elements'):
# Поиск всех кнопок "Easy Apply"
buttons = self.driver.find_elements(By.XPATH, method['xpath']) buttons = self.driver.find_elements(By.XPATH, method['xpath'])
if buttons: if buttons:
for index, button in enumerate(buttons): for index, button in enumerate(buttons):
@ -209,7 +205,6 @@ class LinkedInEasyApplier:
logger.error("No clickable 'Easy Apply' button found after 2 attempts. Page source:\n%s", page_source) logger.error("No clickable 'Easy Apply' button found after 2 attempts. Page source:\n%s", page_source)
raise Exception("No clickable 'Easy Apply' button found") raise Exception("No clickable 'Easy Apply' button found")
def _get_job_description(self) -> str: def _get_job_description(self) -> str:
logger.debug("Getting job description") logger.debug("Getting job description")
try: try:
@ -514,11 +509,47 @@ class LinkedInEasyApplier:
file_path_pdf = os.path.join(folder_path, f"Cover_Letter_{timestamp}.pdf") file_path_pdf = os.path.join(folder_path, f"Cover_Letter_{timestamp}.pdf")
logger.debug(f"Generated file path for cover letter: {file_path_pdf}") logger.debug(f"Generated file path for cover letter: {file_path_pdf}")
c = canvas.Canvas(file_path_pdf, pagesize=letter) c = canvas.Canvas(file_path_pdf, pagesize=A4)
_, height = letter page_width, page_height = A4
text_object = c.beginText(100, height - 100) text_object = c.beginText(50, page_height - 50)
text_object.setFont("Helvetica", 12) text_object.setFont("Helvetica", 12)
text_object.textLines(cover_letter_text)
max_width = page_width - 100
bottom_margin = 50
available_height = page_height - bottom_margin - 50
def split_text_by_width(text, font, font_size, max_width):
wrapped_lines = []
for line in text.splitlines():
if utils.stringWidth(line, font, font_size) > max_width:
words = line.split()
new_line = ""
for word in words:
if utils.stringWidth(new_line + word + " ", font, font_size) <= max_width:
new_line += word + " "
else:
wrapped_lines.append(new_line.strip())
new_line = word + " "
wrapped_lines.append(new_line.strip())
else:
wrapped_lines.append(line)
return wrapped_lines
lines = split_text_by_width(cover_letter_text, "Helvetica", 12, max_width)
for line in lines:
text_height = text_object.getY()
if text_height > bottom_margin:
text_object.textLine(line)
else:
c.drawText(text_object)
c.showPage()
text_object = c.beginText(50, page_height - 50)
text_object.setFont("Helvetica", 12)
text_object.textLine(line)
c.drawText(text_object) c.drawText(text_object)
c.save() c.save()
logger.debug(f"Cover letter successfully generated and saved to: {file_path_pdf}") logger.debug(f"Cover letter successfully generated and saved to: {file_path_pdf}")
@ -632,7 +663,6 @@ class LinkedInEasyApplier:
for item in self.all_data: for item in self.all_data:
logger.debug( logger.debug(
f"Comparing sanitized stored question: '{self._sanitize_text(item['question'])}' and type: '{item.get('type')}' with current question: '{self._sanitize_text(question_text)}' and type: '{question_type}'") f"Comparing sanitized stored question: '{self._sanitize_text(item['question'])}' and type: '{item.get('type')}' with current question: '{self._sanitize_text(question_text)}' and type: '{question_type}'")
@ -659,7 +689,6 @@ class LinkedInEasyApplier:
answer = self.gpt_answerer.answer_question_textual_wide_range(question_text) answer = self.gpt_answerer.answer_question_textual_wide_range(question_text)
logger.debug(f"Generated textual answer: {answer}") logger.debug(f"Generated textual answer: {answer}")
self._save_questions_to_json({'type': question_type, 'question': question_text, 'answer': answer}) self._save_questions_to_json({'type': question_type, 'question': question_text, 'answer': answer})
self._enter_text(text_field, answer) self._enter_text(text_field, answer)
logger.debug("Entered new answer into the textbox and saved it to JSON.") logger.debug("Entered new answer into the textbox and saved it to JSON.")
@ -692,7 +721,6 @@ class LinkedInEasyApplier:
logger.debug("Entered existing date answer") logger.debug("Entered existing date answer")
return True return True
self._save_questions_to_json({'type': 'date', 'question': question_text, 'answer': answer_text}) self._save_questions_to_json({'type': 'date', 'question': question_text, 'answer': answer_text})
self._enter_text(date_field, answer_text) self._enter_text(date_field, answer_text)
logger.debug("Entered new date answer") logger.debug("Entered new date answer")
@ -701,12 +729,12 @@ class LinkedInEasyApplier:
def _find_and_handle_dropdown_question(self, section: WebElement) -> bool: def _find_and_handle_dropdown_question(self, section: WebElement) -> bool:
try: try:
question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element') question = section.find_element(By.CLASS_NAME, 'jobs-easy-apply-form-element')
question_text = question.find_element(By.TAG_NAME, 'label').text.lower()
logger.debug(f"Processing dropdown or combobox question: {question_text}")
dropdowns = question.find_elements(By.TAG_NAME, 'select') dropdowns = question.find_elements(By.TAG_NAME, 'select')
if not dropdowns:
dropdowns = section.find_elements(By.CSS_SELECTOR, '[data-test-text-entity-list-form-select]')
if dropdowns: if dropdowns:
dropdown = dropdowns[0] dropdown = dropdowns[0]
select = Select(dropdown) select = Select(dropdown)
@ -714,6 +742,9 @@ class LinkedInEasyApplier:
logger.debug(f"Dropdown options found: {options}") logger.debug(f"Dropdown options found: {options}")
question_text = question.find_element(By.TAG_NAME, 'label').text.lower()
logger.debug(f"Processing dropdown or combobox question: {question_text}")
current_selection = select.first_selected_option.text current_selection = select.first_selected_option.text
logger.debug(f"Current selection: {current_selection}") logger.debug(f"Current selection: {current_selection}")
@ -738,9 +769,15 @@ class LinkedInEasyApplier:
logger.debug(f"Selected new dropdown answer: {answer}") logger.debug(f"Selected new dropdown answer: {answer}")
return True return True
return False else:
logger.debug(f"No dropdown found. Logging elements for debugging.")
elements = section.find_elements(By.XPATH, ".//*")
logger.debug(f"Elements found: {[element.tag_name for element in elements]}")
return False
except Exception as e: except Exception as e:
logger.warning(f"Failed to handle dropdown or combobox question: {e}") logger.warning(f"Failed to handle dropdown or combobox question: {e}", exc_info=True)
return False return False
def _is_numeric_field(self, field: WebElement) -> bool: def _is_numeric_field(self, field: WebElement) -> bool:

View file

@ -5,6 +5,7 @@ import time
from itertools import product from itertools import product
from pathlib import Path from pathlib import Path
from inputimeout import inputimeout, TimeoutOccurred
from selenium.common.exceptions import NoSuchElementException from selenium.common.exceptions import NoSuchElementException
from selenium.webdriver.common.by import By from selenium.webdriver.common.by import By
@ -45,13 +46,18 @@ class LinkedInJobManager:
def set_parameters(self, parameters): def set_parameters(self, parameters):
logger.debug("Setting parameters for LinkedInJobManager") logger.debug("Setting parameters for LinkedInJobManager")
self.company_blacklist = parameters.get('companyBlacklist', []) or [] self.company_blacklist = parameters.get('company_blacklist', []) or []
self.title_blacklist = parameters.get('titleBlacklist', []) or [] self.title_blacklist = parameters.get('title_blacklist', []) or []
self.positions = parameters.get('positions', []) self.positions = parameters.get('positions', [])
self.locations = parameters.get('locations', []) self.locations = parameters.get('locations', [])
self.apply_once_at_company = parameters.get('applyOnceAtCompany', False) self.apply_once_at_company = parameters.get('apply_once_at_company', False)
self.base_search_url = self.get_base_search_url(parameters) self.base_search_url = self.get_base_search_url(parameters)
self.seen_jobs = [] self.seen_jobs = []
job_applicants_threshold = parameters.get('job_applicants_threshold', {})
self.min_applicants = job_applicants_threshold.get('min_applicants', 0)
self.max_applicants = job_applicants_threshold.get('max_applicants', float('inf'))
resume_path = parameters.get('uploads', {}).get('resume', None) resume_path = parameters.get('uploads', {}).get('resume', None)
self.resume_path = Path(resume_path) if resume_path and Path(resume_path).exists() else None self.resume_path = Path(resume_path) if resume_path and Path(resume_path).exists() else None
self.output_file_directory = Path(parameters['outputFileDirectory']) self.output_file_directory = Path(parameters['outputFileDirectory'])
@ -109,32 +115,80 @@ class LinkedInJobManager:
utils.printyellow("Applying to jobs on this page has been completed!") utils.printyellow("Applying to jobs on this page has been completed!")
time_left = minimum_page_time - time.time() time_left = minimum_page_time - time.time()
# Ask user if they want to skip waiting, with timeout
if time_left > 0: if time_left > 0:
utils.printyellow(f"Sleeping for {time_left} seconds.") try:
logger.debug("Sleeping for %d seconds", time_left) user_input = inputimeout(
time.sleep(time_left) prompt=f"Sleeping for {time_left} seconds. Press 'y' to skip waiting. Timeout 60 seconds : ",
minimum_page_time = time.time() + minimum_time timeout=60).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {time_left} seconds as user chose not to skip.")
utils.printyellow(f"Sleeping for {time_left} seconds.")
time.sleep(time_left)
minimum_page_time = time.time() + minimum_time
if page_sleep % 5 == 0: if page_sleep % 5 == 0:
sleep_time = random.randint(5, 34) sleep_time = random.randint(5, 34)
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") try:
logger.debug("Sleeping for %d seconds", sleep_time) user_input = inputimeout(
time.sleep(sleep_time) prompt=f"Sleeping for {sleep_time / 60} minutes. Press 'y' to skip waiting. Timeout 60 seconds : ",
timeout=60).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {sleep_time} seconds.")
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.")
time.sleep(sleep_time)
page_sleep += 1 page_sleep += 1
except Exception as e: except Exception as e:
logger.error("Unexpected error during job search: %s", e) logger.error("Unexpected error during job search: %s", e)
utils.printred(f"Unexpected error: {e}") utils.printred(f"Unexpected error: {e}")
continue continue
time_left = minimum_page_time - time.time() time_left = minimum_page_time - time.time()
if time_left > 0: if time_left > 0:
utils.printyellow(f"Sleeping for {time_left} seconds.") try:
logger.debug("Sleeping for %d seconds", time_left) user_input = inputimeout(
time.sleep(time_left) prompt=f"Sleeping for {time_left} seconds. Press 'y' to skip waiting. Timeout 60 seconds : ",
minimum_page_time = time.time() + minimum_time timeout=60).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {time_left} seconds as user chose not to skip.")
utils.printyellow(f"Sleeping for {time_left} seconds.")
time.sleep(time_left)
minimum_page_time = time.time() + minimum_time
if page_sleep % 5 == 0: if page_sleep % 5 == 0:
sleep_time = random.randint(50, 90) sleep_time = random.randint(50, 90)
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.") try:
logger.debug("Sleeping for %d seconds", sleep_time) user_input = inputimeout(
time.sleep(sleep_time) prompt=f"Sleeping for {sleep_time / 60} minutes. Press 'y' to skip waiting: ",
timeout=60).strip().lower()
except TimeoutOccurred:
user_input = '' # No input after timeout
if user_input == 'y':
logger.debug("User chose to skip waiting.")
utils.printyellow("User skipped waiting.")
else:
logger.debug(f"Sleeping for {sleep_time} seconds.")
utils.printyellow(f"Sleeping for {sleep_time / 60} minutes.")
time.sleep(sleep_time)
page_sleep += 1 page_sleep += 1
def get_jobs_from_page(self): def get_jobs_from_page(self):
@ -183,16 +237,82 @@ class LinkedInJobManager:
pass pass
job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list") job_results = self.driver.find_element(By.CLASS_NAME, "jobs-search-results-list")
utils.scroll_slow(self.driver, job_results) # utils.scroll_slow(self.driver, job_results)
utils.scroll_slow(self.driver, job_results, step=300, reverse=True) # utils.scroll_slow(self.driver, job_results, step=300, reverse=True)
job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[ job_list_elements = self.driver.find_elements(By.CLASS_NAME, 'scaffold-layout__list-container')[
0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item') 0].find_elements(By.CLASS_NAME, 'jobs-search-results__list-item')
if not job_list_elements: if not job_list_elements:
utils.printyellow("No job class elements found on page, moving to next page.") utils.printyellow("No job class elements found on page, moving to next page.")
logger.debug("No job class elements found on page, skipping") logger.debug("No job class elements found on page, skipping")
return return
job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements] job_list = [Job(*self.extract_job_information_from_tile(job_element)) for job_element in job_list_elements]
for job in job_list: for job in job_list:
try:
logger.debug(f"Starting applicant count search for job: {job.title} at {job.company}")
# Find all job insight elements
job_insight_elements = self.driver.find_elements(By.CLASS_NAME,
"job-details-jobs-unified-top-card__job-insight")
logger.debug(f"Found {len(job_insight_elements)} job insight elements")
# Initialize applicants_count as None
applicants_count = None
# Iterate over each job insight element to find the one containing the word "applicant"
for element in job_insight_elements:
logger.debug(f"Checking element text: {element.text}")
if "applicant" in element.text.lower():
# Found an element containing "applicant"
applicants_text = element.text.strip()
logger.debug(f"Applicants text found: {applicants_text}")
# Extract numeric digits from the text (e.g., "70 applicants" -> "70")
applicants_count = ''.join(filter(str.isdigit, applicants_text))
logger.debug(f"Extracted applicants count: {applicants_count}")
if applicants_count:
if "over" in applicants_text.lower():
applicants_count = int(applicants_count) + 1 # Handle "over X applicants"
logger.debug(f"Applicants count adjusted for 'over': {applicants_count}")
else:
applicants_count = int(applicants_count) # Convert the extracted number to an integer
break
# Check if applicants_count is valid (not None) before performing comparisons
if applicants_count is not None:
# Perform the threshold check for applicants count
if applicants_count < self.min_applicants or applicants_count > self.max_applicants:
utils.printyellow(
f"Skipping {job.title} at {job.company} due to applicants count: {applicants_count}")
logger.debug(f"Skipping {job.title} at {job.company}, applicants count: {applicants_count}")
self.write_to_file(job, "skipped_due_to_applicants")
continue # Skip this job if applicants count is outside the threshold
else:
logger.debug(f"Applicants count {applicants_count} is within the threshold")
else:
# If no applicants count was found, log a warning but continue the process
logger.warning(
f"Applicants count not found for {job.title} at {job.company}, continuing with application.")
except NoSuchElementException:
# Log a warning if the job insight elements are not found, but do not stop the job application process
logger.warning(
f"Applicants count elements not found for {job.title} at {job.company}, continuing with application.")
except ValueError as e:
# Handle errors when parsing the applicants count
logger.error(f"Error parsing applicants count for {job.title} at {job.company}: {e}")
except Exception as e:
# Catch any other exceptions to ensure the process continues
logger.error(
f"Unexpected error during applicants count processing for {job.title} at {job.company}: {e}")
# Continue with the job application process regardless of the applicants count check
logger.debug(f"Continuing with job application for {job.title} at {job.company}")
if self.is_blacklisted(job.title, job.company, job.link): if self.is_blacklisted(job.title, job.company, job.link):
utils.printyellow(f"Blacklisted {job.title} at {job.company}, skipping...") utils.printyellow(f"Blacklisted {job.title} at {job.company}, skipping...")
logger.debug("Job blacklisted: %s at %s", job.title, job.company) logger.debug("Job blacklisted: %s at %s", job.title, job.company)
@ -250,7 +370,7 @@ class LinkedInJobManager:
url_parts = [] url_parts = []
if parameters['remote']: if parameters['remote']:
url_parts.append("f_CF=f_WRA") url_parts.append("f_CF=f_WRA")
experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experienceLevel', {}).items()) if experience_levels = [str(i + 1) for i, (level, v) in enumerate(parameters.get('experience_level', {}).items()) if
v] v]
if experience_levels: if experience_levels:
url_parts.append(f"f_E={','.join(experience_levels)}") url_parts.append(f"f_E={','.join(experience_levels)}")
@ -307,10 +427,8 @@ class LinkedInJobManager:
title_blacklisted = any(word in job_title_words for word in self.title_blacklist) title_blacklisted = any(word in job_title_words for word in self.title_blacklist)
company_blacklisted = company.strip().lower() in (word.strip().lower() for word in self.company_blacklist) company_blacklisted = company.strip().lower() in (word.strip().lower() for word in self.company_blacklist)
link_seen = link in self.seen_jobs link_seen = link in self.seen_jobs
is_blacklisted = title_blacklisted or company_blacklisted or link_seen is_blacklisted = title_blacklisted or company_blacklisted or link_seen
logger.debug("Job blacklisted status: %s", is_blacklisted) logger.debug("Job blacklisted status: %s", is_blacklisted)
return is_blacklisted
return title_blacklisted or company_blacklisted or link_seen return title_blacklisted or company_blacklisted or link_seen
@ -333,9 +451,9 @@ class LinkedInJobManager:
existing_data = json.load(f) existing_data = json.load(f)
for applied_job in existing_data: for applied_job in existing_data:
if applied_job['company'].strip().lower() == company.strip().lower(): if applied_job['company'].strip().lower() == company.strip().lower():
utils.printyellow(f"Already applied at {company} (once per company policy), skipping...") utils.printyellow(
f"Already applied at {company} (once per company policy), skipping...")
return True return True
except json.JSONDecodeError: except json.JSONDecodeError:
continue continue
return False return False

View file

@ -1,13 +1,21 @@
<<<<<<< HEAD
from typing import Dict, List from typing import Dict, List
from linkedin_api import Linkedin from linkedin_api import Linkedin
from typing import Optional, Union, Literal from typing import Optional, Union, Literal
from urllib.parse import quote, urlencode, parse_qs, urlparse from urllib.parse import quote, urlencode, parse_qs, urlparse
=======
>>>>>>> upstream/v3
import logging import logging
import json from typing import Dict, List
from typing import Optional, Union, Literal
from urllib.parse import urlencode
from linkedin_api import Linkedin
# set log to all debug # set log to all debug
logging.basicConfig(level=logging.INFO) logging.basicConfig(level=logging.INFO)
class LinkedInEvolvedAPI(Linkedin): class LinkedInEvolvedAPI(Linkedin):
already_applied_jobs: List[str] = [] already_applied_jobs: List[str] = []
@ -15,44 +23,44 @@ class LinkedInEvolvedAPI(Linkedin):
super().__init__(username, password) super().__init__(username, password)
def search_jobs( def search_jobs(
self, self,
keywords: Optional[str] = None, keywords: Optional[str] = None,
companies: Optional[List[str]] = None, companies: Optional[List[str]] = None,
experience: Optional[ experience: Optional[
List[ List[
Union[ Union[
Literal["1"], Literal["1"],
Literal["2"], Literal["2"],
Literal["3"], Literal["3"],
Literal["4"], Literal["4"],
Literal["5"], Literal["5"],
Literal["6"], Literal["6"],
]
] ]
] ] = None,
] = None, job_type: Optional[
job_type: Optional[ List[
List[ Union[
Union[ Literal["F"],
Literal["F"], Literal["C"],
Literal["C"], Literal["P"],
Literal["P"], Literal["T"],
Literal["T"], Literal["I"],
Literal["I"], Literal["V"],
Literal["V"], Literal["O"],
Literal["O"], ]
] ]
] ] = None,
] = None, job_title: Optional[List[str]] = None,
job_title: Optional[List[str]] = None, industries: Optional[List[str]] = None,
industries: Optional[List[str]] = None, location_name: Optional[str] = None,
location_name: Optional[str] = None, remote: Optional[List[Union[Literal["1"], Literal["2"], Literal["3"]]]] = None,
remote: Optional[List[Union[Literal["1"], Literal["2"], Literal["3"]]]] = None, listed_at: None | int = None,
listed_at: None | int = None, distance: Optional[int] = None,
distance: Optional[int] = None, easy_apply: Optional[bool] = True,
easy_apply: Optional[bool] = True, limit=-1,
limit=-1, offset=0,
offset=0, **kwargs,
**kwargs,
) -> List[Dict]: ) -> List[Dict]:
"""Perform a LinkedIn search for jobs. """Perform a LinkedIn search for jobs.
@ -159,8 +167,8 @@ class LinkedInEvolvedAPI(Linkedin):
break break
results.extend(new_data) results.extend(new_data)
if ( if (
(-1 < limit <= len(results)) (-1 < limit <= len(results))
or len(results) / count >= Linkedin._MAX_REPEATED_REQUESTS or len(results) / count >= Linkedin._MAX_REPEATED_REQUESTS
) or len(elements) == 0: ) or len(elements) == 0:
break break
@ -168,7 +176,7 @@ class LinkedInEvolvedAPI(Linkedin):
return results return results
def get_fields_for_easy_apply(self,job_id: str) -> List[Dict]: def get_fields_for_easy_apply(self, job_id: str) -> List[Dict]:
"""Get fields needed for easy apply jobs. """Get fields needed for easy apply jobs.
:param job_id: Job ID :param job_id: Job ID
@ -182,13 +190,11 @@ class LinkedInEvolvedAPI(Linkedin):
headers: Dict[str, str] = self._headers() headers: Dict[str, str] = self._headers()
headers["Accept"] = "application/vnd.linkedin.normalized+json+2.1" headers["Accept"] = "application/vnd.linkedin.normalized+json+2.1"
headers["csrf-token"] = cookies["JSESSIONID"].replace('"', "") headers["csrf-token"] = cookies["JSESSIONID"].replace('"', "")
headers["Cookie"] = cookie_str headers["Cookie"] = cookie_str
headers["Connection"] = "keep-alive" headers["Connection"] = "keep-alive"
default_params = { default_params = {
"decorationId": "com.linkedin.voyager.dash.deco.jobs.OnsiteApplyApplication-67", "decorationId": "com.linkedin.voyager.dash.deco.jobs.OnsiteApplyApplication-67",
"jobPostingUrn": f"urn:li:fsd_jobPosting:{job_id}", "jobPostingUrn": f"urn:li:fsd_jobPosting:{job_id}",
@ -253,7 +259,7 @@ class LinkedInEvolvedAPI(Linkedin):
return form_components return form_components
def apply_to_job(self,job_id: str, fields: dict, followCompany: bool = True) -> bool: def apply_to_job(self, job_id: str, fields: dict, followCompany: bool = True) -> bool:
return False return False
# ToDo: Implement apply to job parser first # ToDo: Implement apply to job parser first
@ -269,7 +275,7 @@ class LinkedInEvolvedAPI(Linkedin):
# EXAMPLE OF WORKING PAYLOAD # EXAMPLE OF WORKING PAYLOAD
# 4005350454 is job_id, so need to be replaced with the job_id # 4005350454 is job_id, so need to be replaced with the job_id
#{ # {
# "followCompany": true, # "followCompany": true,
# "responses": [ # "responses": [
# { # {
@ -349,9 +355,10 @@ class LinkedInEvolvedAPI(Linkedin):
# } # }
# ], # ],
# "trackingId": "" # "trackingId": ""
#} # }
# Push the commit to the repository and create a pull request to the v3 branch. # Push the commit to the repository and create a pull request to the v3 branch.
<<<<<<< HEAD
def create_request_pdf(self, filename: str) -> str | None: def create_request_pdf(self, filename: str) -> str | None:
""" """
@ -469,10 +476,13 @@ class LinkedInEvolvedAPI(Linkedin):
with open(file_path, 'rb') as file: with open(file_path, 'rb') as file:
binary_data = file.read() binary_data = file.read()
return binary_data return binary_data
=======
>>>>>>> upstream/v3
def set_job_as_applied(self, job_id: str) -> None: def set_job_as_applied(self, job_id: str) -> None:
self.already_applied_jobs.append(job_id) self.already_applied_jobs.append(job_id)
<<<<<<< HEAD
def upload_linkedin_resume(self, cv_path: str) -> str | bool: def upload_linkedin_resume(self, cv_path: str) -> str | bool:
url = self.create_request_pdf("resume.pdf") url = self.create_request_pdf("resume.pdf")
if url: if url:
@ -491,6 +501,14 @@ if __name__ == "__main__":
api: LinkedInEvolvedAPI = LinkedInEvolvedAPI(username="", password="") api: LinkedInEvolvedAPI = LinkedInEvolvedAPI(username="", password="")
jobs = api.search_jobs(keywords="Frontend Developer", location_name="Italia", limit=100, easy_apply=True, offset=1, listed_at=None) jobs = api.search_jobs(keywords="Frontend Developer", location_name="Italia", limit=100, easy_apply=True, offset=1, listed_at=None)
=======
## EXAMPLE USAGE
if __name__ == "__main__":
api: LinkedInEvolvedAPI = LinkedInEvolvedAPI(username="", password="")
jobs = api.search_jobs(keywords="Frontend Developer", location_name="Italia", limit=100, easy_apply=True, offset=1,
listed_at=None)
>>>>>>> upstream/v3
for job in jobs: for job in jobs:
job_id: str = job["job_id"] job_id: str = job["job_id"]
@ -514,7 +532,3 @@ if __name__ == "__main__":
print(field) print(field)
break break

View file

@ -8,7 +8,7 @@ from selenium import webdriver
log_file = "app_log.log" log_file = "app_log.log"
logging.basicConfig( logging.basicConfig(
level=logging.DEBUG, level=logging.INFO,
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s', format='%(asctime)s - %(name)s - %(levelname)s - %(message)s',
handlers=[ handlers=[
logging.FileHandler(log_file, mode='a', encoding='utf-8'), logging.FileHandler(log_file, mode='a', encoding='utf-8'),
@ -22,7 +22,7 @@ file_handler = logging.FileHandler(log_file, mode='a', encoding='utf-8')
formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s') formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s')
file_handler.setFormatter(formatter) file_handler.setFormatter(formatter)
logger.addHandler(file_handler) logger.addHandler(file_handler)
logger.setLevel(logging.DEBUG) logger.setLevel(logging.INFO)
chromeProfilePath = os.path.join(os.getcwd(), "chrome_profile", "linkedin_profile") chromeProfilePath = os.path.join(os.getcwd(), "chrome_profile", "linkedin_profile")
@ -90,7 +90,13 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse
return return
position = start position = start
previous_position = None # Tracking the previous position to avoid duplicate scrolls
while (step > 0 and position < end) or (step < 0 and position > end): while (step > 0 and position < end) or (step < 0 and position > end):
if position == previous_position:
# Avoid re-scrolling to the same position
logger.debug("Stopping scroll as position hasn't changed: %d", position)
break
try: try:
driver.execute_script(script_scroll_to, scrollable_element, position) driver.execute_script(script_scroll_to, scrollable_element, position)
logger.debug("Scrolled to position: %d", position) logger.debug("Scrolled to position: %d", position)
@ -98,11 +104,15 @@ def scroll_slow(driver, scrollable_element, start=0, end=3600, step=300, reverse
logger.error("Error during scrolling: %s", e) logger.error("Error during scrolling: %s", e)
print(f"Error during scrolling: {e}") print(f"Error during scrolling: {e}")
previous_position = position
position += step position += step
# Decrease the step but ensure it doesn't reverse direction
step = max(10, abs(step) - 10) * (-1 if reverse else 1) step = max(10, abs(step) - 10) * (-1 if reverse else 1)
time.sleep(random.uniform(0.6, 1.5)) time.sleep(random.uniform(0.6, 1.5))
# Ensure the final scroll position is correct
driver.execute_script(script_scroll_to, scrollable_element, end) driver.execute_script(script_scroll_to, scrollable_element, end)
logger.debug("Scrolled to final position: %d", end) logger.debug("Scrolled to final position: %d", end)
time.sleep(0.5) time.sleep(0.5)