diff --git a/.github/workflows/deploy-job-php.yml b/.github/workflows/deploy-job-php.yml
index b7835ef..85aac1e 100644
--- a/.github/workflows/deploy-job-php.yml
+++ b/.github/workflows/deploy-job-php.yml
@@ -9,7 +9,7 @@ jobs:
steps:
- name: Checkout code
- uses: actions/checkout@v3
+ uses: actions/checkout@v7
- name: Deploy via SFTP
uses: wlixcc/SFTP-Deploy-Action@v1.2.4
diff --git a/.github/workflows/deploy-to-db.yml b/.github/workflows/deploy-to-db.yml
index c06ad12..91daf46 100644
--- a/.github/workflows/deploy-to-db.yml
+++ b/.github/workflows/deploy-to-db.yml
@@ -20,10 +20,10 @@ jobs:
steps:
- name: Checkout repository
- uses: actions/checkout@v3
+ uses: actions/checkout@v7
- name: Set up Python
- uses: actions/setup-python@v4
+ uses: actions/setup-python@v7
with:
python-version: '3.10'
cache: 'pip'
diff --git a/.github/workflows/python-package-conda.yml b/.github/workflows/python-package-conda.yml
index 67a279e..8a25851 100644
--- a/.github/workflows/python-package-conda.yml
+++ b/.github/workflows/python-package-conda.yml
@@ -2,27 +2,42 @@ name: Job Scraping and Firebase Processing Pipeline
on:
schedule:
- - cron: '0 0 * * *' # Run daily at midnight UTC
+ # Not on the full hour: GitHub delays or drops scheduled runs at :00 when it is busy.
+ # The second run catches up if the first one failed or was dropped.
+ - cron: '17 5 * * *'
+ - cron: '47 15 * * *'
workflow_dispatch: # Allow manual triggering
+permissions:
+ contents: write
+
+# Never run two scrapers at once (e.g. a manual run while the scheduled one is running)
+concurrency:
+ group: job-scraper
+ cancel-in-progress: false
+
jobs:
scrape-and-process:
runs-on: ubuntu-latest
+ timeout-minutes: 45
+ env:
+ PYTHONUNBUFFERED: '1'
steps:
- - uses: actions/checkout@v3
+ - uses: actions/checkout@v7
with:
- fetch-depth: 0
- persist-credentials: true
+ # Latest commit of the branch, not the trigger commit (matters when a run was queued)
+ ref: ${{ github.ref_name }}
- name: Set up Python
- uses: actions/setup-python@v4
+ uses: actions/setup-python@v7
with:
- python-version: '3.x'
+ python-version: '3.14'
+ cache: pip
- name: Install dependencies
run: |
python -m pip install --upgrade pip
- pip install requests beautifulsoup4 firebase-admin
+ pip install -r requirements.txt
- name: Run job scraper
run: python job_scraper.py
@@ -35,21 +50,32 @@ jobs:
# FIREBASE_SERVICE_ACCOUNT: ${{ secrets.FIREBASE_SERVICE_ACCOUNT }}
# run: python firebase_importer.py
+ # Notify before committing: if Telegram fails, nothing is committed and the next run
+ # sends these jobs again. Only on main: test runs on other branches neither notify nor commit.
- name: Send Telegram Notification
+ if: github.ref == 'refs/heads/main'
env:
TELEGRAM_BOT_TOKEN: ${{ secrets.TELEGRAM_BOT_TOKEN }}
TELEGRAM_CHAT_ID: ${{ secrets.TELEGRAM_CHAT_ID }}
run: python telegram.py
- name: Commit and push if changes
+ if: github.ref == 'refs/heads/main'
run: |
git config --local user.email "action@github.com"
git config --local user.name "GitHub Action"
git add jobs_all_processed.json new_jobs.json
- git diff --quiet && git diff --staged --quiet || (git commit -m "Update job data" && git push)
-
+ git diff --staged --quiet && exit 0
+ git commit -m "Update job data"
+ for attempt in 1 2 3; do
+ git push && exit 0
+ sleep 10
+ git pull --rebase origin main || { git rebase --abort; exit 1; }
+ done
+ exit 1
+
# Trigger the database import workflow
-
+
#- name: Trigger DB Import Workflow
# if: success()
# uses: peter-evans/repository-dispatch@v2
diff --git a/job_scraper.py b/job_scraper.py
index 42c2024..dceb119 100644
--- a/job_scraper.py
+++ b/job_scraper.py
@@ -1,8 +1,13 @@
import requests
+from requests.adapters import HTTPAdapter
+from urllib3.util.retry import Retry
from bs4 import BeautifulSoup
import json
from datetime import datetime
import os
+import random
+import re
+import sys
import time # Import the time module for adding delay
# JSON configuration
@@ -33,27 +38,87 @@
}
}
+BASE_URL = 'https://govjobs.public.lu'
+START_URL = 'https://govjobs.public.lu/fr/rechercher-parmi-offres-emploi.html'
+
+# Network settings. govjobs.public.lu refuses TCP connections ("Connection refused") when a
+# client opens too many of them, so all requests share one keep-alive session. The server
+# closes idle connections after 5 seconds, so the delays between requests stay below that.
+REQUEST_TIMEOUT = (10, 30) # (connect, read) in seconds
+LISTING_DELAY = 2 # seconds between listing pages (plus up to 1s jitter)
+DETAIL_DELAY = 2 # seconds between detail pages (plus up to 1s jitter)
+MAX_PAGES = 100 # safety net against endless pagination
+MAX_DETAIL_FETCHES = 100 # per run; the rest is fetched on the next run
+MAX_DETAIL_FAILURES = 3 # consecutive failed detail pages before details are paused for this run
+TIME_BUDGET = 20 * 60 # seconds; afterwards the scraper stops and saves what it has
+RETRY_STATUSES = (429, 500, 502, 503, 504)
+
+
+class CappedRetry(Retry):
+ """Retry that never waits longer than 60 seconds, even if the server sends a larger Retry-After."""
+
+ def get_retry_after(self, response):
+ retry_after = super().get_retry_after(response)
+ return None if retry_after is None else min(retry_after, 60)
+
+
+def create_session():
+ # Refused connections and 429/5xx answers are retried after ~0, 6, 12, 24 and 48 seconds
+ retry = CappedRetry(
+ total=5,
+ connect=5,
+ read=2,
+ status=3,
+ backoff_factor=3,
+ backoff_max=60,
+ backoff_jitter=1,
+ status_forcelist=RETRY_STATUSES,
+ allowed_methods=frozenset({'GET'}),
+ raise_on_status=False,
+ )
+ session = requests.Session()
+ session.mount('https://', HTTPAdapter(max_retries=retry))
+ session.headers.update({
+ 'User-Agent': 'GovScrapperJobMail/1.0 (+https://github.com/masterries/GovScrapperJobMail)',
+ 'Accept': 'text/html,application/xhtml+xml',
+ 'Accept-Language': 'fr,en;q=0.8',
+ })
+ return session
+
+
+session = create_session()
+
+
+def polite_sleep(seconds):
+ time.sleep(seconds + random.uniform(0, 1))
+
+
def scrape_job_details(url):
- """Scrape detailed information from the job's detail page"""
+ """Scrape detailed information from the job's detail page.
+
+ Raises requests.RequestException when the page could not be fetched (refused connection,
+ timeout, 429/5xx after all retries) so the caller can back off.
+ """
print(f'Scraping details: {url}')
job_details = {}
-
+
+ response = session.get(url, timeout=REQUEST_TIMEOUT)
+ if response.status_code in (404, 410):
+ print(f"Failed to get details from {url}: Status code {response.status_code}")
+ return job_details
+ response.raise_for_status()
+
try:
- response = requests.get(url)
- if response.status_code != 200:
- print(f"Failed to get details from {url}: Status code {response.status_code}")
- return job_details
-
soup = BeautifulSoup(response.content, 'html.parser')
-
+
# Find the main content div that contains the detailed job description
page_text_div = soup.find('div', class_='page-text')
if not page_text_div:
return job_details
-
+
# Extract all text content from the page-text div
job_details['Full Description'] = page_text_div.get_text(separator=' ', strip=True)
-
+
# Try to extract structured data from the page-text div
# Look for headers/titles followed by content
headers = page_text_div.find_all(['h2', 'h3', 'h4', 'strong', 'b'])
@@ -68,10 +133,10 @@ def scrape_job_details(url):
break
if sibling.name == 'p' or sibling.name == 'ul' or sibling.name == 'div':
content.append(sibling.get_text(strip=True))
-
+
if content:
job_details[header_text] = ' '.join(content)
-
+
# Try to extract any table data if present
tables = page_text_div.find_all('table')
for i, table in enumerate(tables):
@@ -84,108 +149,188 @@ def scrape_job_details(url):
job_details[row_data[0]] = row_data[1]
elif row_data:
table_data.append(row_data)
-
+
if table_data and i == 0:
job_details['Table Data'] = table_data
-
+
except Exception as e:
- print(f"Error scraping details from {url}: {str(e)}")
-
+ print(f"Error parsing details from {url}: {str(e)}")
+
return job_details
-def scrape_jobs():
- base_url = 'https://govjobs.public.lu'
- start_url = 'https://govjobs.public.lu/fr/rechercher-parmi-offres-emploi.html'
- all_jobs = []
- page_number = 0
-
- # Load existing jobs to check if we need to fetch details
- existing_jobs_dict = {}
- if os.path.exists('jobs_all_processed.json'):
- try:
- with open('jobs_all_processed.json', 'r', encoding='utf-8') as f:
- existing_jobs = json.load(f)
- existing_jobs_dict = {job['Link']: job for job in existing_jobs if 'Link' in job}
- print(f"Loaded {len(existing_jobs_dict)} existing jobs for reference")
- except Exception as e:
- print(f"Error loading existing jobs: {str(e)}")
-
- while True:
- url = f'{start_url}?b={page_number * 20}'
+def parse_result_count(soup):
+ """Number of jobs the search page says it has, or None if it can't be read."""
+ count_tag = soup.find(class_='search-meta-count')
+ if not count_tag:
+ return None
+ match = re.search(r'\d+', count_tag.get_text().replace(' ', '').replace('\xa0', ''))
+ return int(match.group()) if match else None
+
+def parse_listing_article(article):
+ job = {}
+ title_tag = article.find('h2', class_='article-title')
+ a_tag = title_tag.find('a') if title_tag else None
+ if not a_tag or not a_tag.get('href'):
+ return None
+
+ job['Titel'] = a_tag.text.strip()
+ link = a_tag['href']
+ if link.startswith('//'):
+ link = 'https:' + link
+ elif link.startswith('/'):
+ link = BASE_URL + link
+ else:
+ link = BASE_URL + '/' + link.lstrip('/')
+ job['Link'] = link
+
+ footer = article.find('footer', class_='article-metas')
+ if footer:
+ meta_list = footer.find('ul', class_='list--inline list--dotted')
+ if meta_list:
+ for item in meta_list.find_all('li'):
+ text = item.get_text(separator=' ').strip()
+ key, value = text.split(': ', 1) if ': ' in text else (text, '')
+ job[key] = value
+
+ custom_list = article.find('ul', class_='nude article-custom')
+ if custom_list:
+ for item in custom_list.find_all('li'):
+ span, b_tag = item.find('span'), item.find('b')
+ if span and b_tag:
+ job[span.text.strip()] = b_tag.text.strip()
+
+ return job
+
+def scrape_listing(deadline):
+ """Collect all jobs from the search result pages.
+
+ Returns (jobs, complete); complete is False when the crawl had to stop early.
+ """
+ jobs = []
+ seen_links = set()
+ expected_total = None
+
+ for page_number in range(MAX_PAGES):
+ if time.monotonic() > deadline:
+ print('::warning::Time budget exceeded while crawling the listing pages')
+ return jobs, False
+ if page_number:
+ polite_sleep(LISTING_DELAY)
+
+ url = f'{START_URL}?b={page_number * 20}'
print(f'Scraping page: {url}')
+ try:
+ response = session.get(url, timeout=REQUEST_TIMEOUT)
+ response.raise_for_status()
+ except requests.RequestException as e:
+ print(f'::warning::Listing crawl aborted at {url}: {e}')
+ return jobs, False
- response = requests.get(url)
soup = BeautifulSoup(response.content, 'html.parser')
+ if expected_total is None:
+ expected_total = parse_result_count(soup)
articles = soup.find_all('article', class_='article search-result search-result--job')
if not articles:
+ if not soup.find('ol', class_='search-results'):
+ # Not a regular (empty) result page, e.g. a maintenance or error page
+ print(f'::warning::Unexpected page without search results at {url}')
+ return jobs, False
break
+ new_on_page = 0
for article in articles:
- job = {}
- title_tag = article.find('h2', class_='article-title')
- if title_tag and title_tag.find('a'):
- a_tag = title_tag.find('a')
- job['Titel'] = a_tag.text.strip()
- link = a_tag['href']
- if link.startswith('//'):
- link = 'https:' + link
- elif link.startswith('/'):
- link = base_url + link
- else:
- link = base_url + '/' + link.lstrip('/')
- job['Link'] = link
-
- footer = article.find('footer', class_='article-metas')
- if footer:
- meta_list = footer.find('ul', class_='list--inline list--dotted')
- if meta_list:
- for item in meta_list.find_all('li'):
- text = item.get_text(separator=' ').strip()
- key, value = text.split(': ', 1) if ': ' in text else (text, '')
- job[key] = value
-
- custom_list = article.find('ul', class_='nude article-custom')
- if custom_list:
- for item in custom_list.find_all('li'):
- span, b_tag = item.find('span'), item.find('b')
- if span and b_tag:
- job[span.text.strip()] = b_tag.text.strip()
-
- # Get detailed information from the job page if needed
- if 'Link' in job:
- existing_job = existing_jobs_dict.get(job['Link'])
-
- # Check if we need to fetch details
- needs_details = True
- if existing_job:
- # Check if the existing job already has detailed info
- if 'Full Description' in existing_job or any(key.startswith('Section:') for key in existing_job):
- needs_details = False
- print(f"Using cached details for: {job.get('Titel', job['Link'])}")
-
- if needs_details:
- # Add a delay between requests to avoid overloading the server
- time.sleep(1)
- print(f"Fetching details for: {job.get('Titel', job['Link'])}")
- job_details = scrape_job_details(job['Link'])
-
- # Merge the details with the main job data
- for key, value in job_details.items():
- if key not in job: # Don't overwrite existing data
- job[key] = value
- elif existing_job:
- # Copy the detailed fields from the existing job
- for key, value in existing_job.items():
- if key not in job and key != 'adding_date':
- job[key] = value
-
- all_jobs.append(job)
-
- page_number += 1
- time.sleep(2) # Add a 2-second delay between page scrapes
-
- return process_jobs(all_jobs)
+ job = parse_listing_article(article)
+ if not job:
+ print(f'Skipping a search result without link on {url}')
+ continue
+ if job['Link'] in seen_links:
+ continue # the listing shifted while crawling
+ seen_links.add(job['Link'])
+ jobs.append(job)
+ new_on_page += 1
+
+ if new_on_page == 0:
+ break # the site repeats pages we already have
+ else:
+ print(f'::warning::Stopped after {MAX_PAGES} listing pages')
+ return jobs, False
+
+ if expected_total is not None and len(jobs) < expected_total:
+ print(f'::warning::The listing shows {expected_total} jobs but only {len(jobs)} were collected')
+ return jobs, True
+
+def add_job_details(jobs, existing_jobs_dict, deadline):
+ """Add the detail page data to the listed jobs, from the cache where possible.
+
+ Returns False when some missing details could not be fetched in this run. Those jobs are
+ saved without 'Full Description' and fetched again on the next run.
+ """
+ complete = True
+ fetches = 0
+ consecutive_failures = 0
+
+ for job in jobs:
+ existing_job = existing_jobs_dict.get(job['Link'])
+
+ # Check if the existing job already has detailed info
+ if existing_job and ('Full Description' in existing_job or any(key.startswith('Section:') for key in existing_job)):
+ print(f"Using cached details for: {job.get('Titel', job['Link'])}")
+ # Copy the detailed fields from the existing job
+ for key, value in existing_job.items():
+ if key not in job and key != 'adding_date':
+ job[key] = value
+ continue
+
+ if not complete:
+ continue
+ if fetches >= MAX_DETAIL_FETCHES or time.monotonic() > deadline:
+ print('::warning::Detail fetch limit reached; the remaining details are fetched on the next run')
+ complete = False
+ continue
+
+ # Add a delay between requests to avoid overloading the server
+ polite_sleep(DETAIL_DELAY)
+ print(f"Fetching details for: {job.get('Titel', job['Link'])}")
+ fetches += 1
+ try:
+ job_details = scrape_job_details(job['Link'])
+ except requests.RequestException as e:
+ print(f"Error scraping details from {job['Link']}: {str(e)}")
+ consecutive_failures += 1
+ if consecutive_failures >= MAX_DETAIL_FAILURES:
+ print(f'::warning::{consecutive_failures} detail pages failed in a row; the remaining details are fetched on the next run')
+ complete = False
+ continue
+ consecutive_failures = 0
+
+ # Merge the details with the main job data
+ for key, value in job_details.items():
+ if key not in job: # Don't overwrite existing data
+ job[key] = value
+
+ return complete
+
+def scrape_jobs():
+ """Scrape the listing, then the missing detail pages.
+
+ Returns (processed_jobs, complete).
+ """
+ deadline = time.monotonic() + TIME_BUDGET
+
+ # Load existing jobs to check if we need to fetch details
+ existing_jobs_dict = {}
+ if os.path.exists('jobs_all_processed.json'):
+ with open('jobs_all_processed.json', 'r', encoding='utf-8') as f:
+ existing_jobs = json.load(f)
+ existing_jobs_dict = {job['Link']: job for job in existing_jobs if 'Link' in job}
+ print(f"Loaded {len(existing_jobs_dict)} existing jobs for reference")
+
+ # Listing first, so a blocked detail page can't cost us the list of new jobs
+ jobs, listing_complete = scrape_listing(deadline)
+ details_complete = add_job_details(jobs, existing_jobs_dict, deadline)
+
+ return process_jobs(jobs), listing_complete and details_complete
def process_jobs(jobs):
processed_jobs = []
@@ -194,17 +339,17 @@ def process_jobs(jobs):
for fr_key, value in job.items():
en_key = json_config['column_mapping'].get(fr_key, fr_key)
processed_job[en_key] = value
-
+
for group_name, fields in json_config['group_fields'].items():
group_value = next((processed_job[field] for field in fields if field in processed_job), None)
if group_value:
processed_job[group_name] = group_value
for field in fields:
processed_job.pop(field, None)
-
+
processed_job['adding_date'] = datetime.now().isoformat()
processed_jobs.append(processed_job)
-
+
return processed_jobs
def update_json(new_jobs, filename='jobs_all_processed.json'):
@@ -216,7 +361,7 @@ def update_json(new_jobs, filename='jobs_all_processed.json'):
existing_links = {job['Link'] for job in existing_jobs}
existing_jobs_dict = {job['Link']: job for job in existing_jobs}
-
+
updated_jobs = existing_jobs.copy()
new_jobs_added = []
@@ -226,17 +371,18 @@ def update_json(new_jobs, filename='jobs_all_processed.json'):
updated_jobs.append(job)
new_jobs_added.append(job)
existing_links.add(job['Link'])
+ existing_jobs_dict[job['Link']] = job
else:
# The job exists, but check if we need to update with new detail info
existing_job = existing_jobs_dict[job['Link']]
-
+
# Check if the job has any new fields from the detail page that the existing one doesn't
has_new_details = False
for key, value in job.items():
if key not in existing_job and key != 'adding_date':
has_new_details = True
break
-
+
if has_new_details:
# Update the existing job with new details while preserving the original adding_date
original_date = existing_job.get('adding_date')
@@ -260,52 +406,42 @@ def test_scrape_details(num_jobs=2):
"""
Test function to scrape only a limited number of jobs with their details
for testing purposes.
-
+
Args:
num_jobs: Number of jobs to scrape (default 2)
"""
- base_url = 'https://govjobs.public.lu'
- start_url = 'https://govjobs.public.lu/fr/rechercher-parmi-offres-emploi.html'
test_jobs = []
-
+
# Get the first page only
- print(f'Test scraping: {start_url}')
- response = requests.get(start_url)
+ print(f'Test scraping: {START_URL}')
+ response = session.get(START_URL, timeout=REQUEST_TIMEOUT)
+ response.raise_for_status()
soup = BeautifulSoup(response.content, 'html.parser')
articles = soup.find_all('article', class_='article search-result search-result--job')
-
+
# Limit to the specified number of jobs
articles = articles[:num_jobs]
-
+
for article in articles:
- job = {}
- title_tag = article.find('h2', class_='article-title')
- if title_tag and title_tag.find('a'):
- a_tag = title_tag.find('a')
- job['Title'] = a_tag.text.strip()
- link = a_tag['href']
- if link.startswith('//'):
- link = 'https:' + link
- elif link.startswith('/'):
- link = base_url + link
- else:
- link = base_url + '/' + link.lstrip('/')
- job['Link'] = link
-
+ job = parse_listing_article(article)
+ if job:
+ job['Title'] = job.pop('Titel')
+
# Get the job details
print(f'Testing detail scraping for: {job["Title"]}')
+ polite_sleep(DETAIL_DELAY)
job_details = scrape_job_details(job['Link'])
-
+
# Merge the details with the basic job data
job.update(job_details)
-
+
test_jobs.append(job)
-
+
# Save the test results to a file
test_file = 'test_job_details.json'
with open(test_file, 'w', encoding='utf-8') as f:
json.dump(test_jobs, f, ensure_ascii=False, indent=4)
-
+
print(f"Test completed. {len(test_jobs)} jobs processed and saved to {test_file}")
return test_jobs
@@ -313,16 +449,23 @@ def main():
# Uncomment the test function to run in test mode
# test_scrape_details(2)
# return
-
- scraped_jobs = scrape_jobs()
+
+ scraped_jobs, complete = scrape_jobs()
+ if not scraped_jobs:
+ print('::error::No jobs could be scraped, nothing was saved')
+ sys.exit(1)
+
new_jobs = update_json(scraped_jobs)
save_new_jobs(new_jobs)
print(f"Scraping completed. {len(scraped_jobs)} jobs processed.")
print(f"{len(new_jobs)} new jobs added to 'jobs_all_processed.json' and saved to 'new_jobs.json'")
+ if not complete:
+ # Partial data is still saved and committed; the rest is picked up on the next run
+ print('::warning::Scraping was incomplete, the missing jobs/details are fetched on the next run')
if __name__ == "__main__":
# For testing, uncomment this line:
# test_scrape_details(2)
-
+
# For regular operation, keep this line:
- main()
\ No newline at end of file
+ main()
diff --git a/requirements.txt b/requirements.txt
index fad3350..70974c3 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -1,2 +1,3 @@
-tqdm
-firebase_admin
\ No newline at end of file
+requests>=2.32,<3
+urllib3>=2,<3
+beautifulsoup4>=4.12,<5
diff --git a/telegram.py b/telegram.py
index 420af5b..b1d3777 100644
--- a/telegram.py
+++ b/telegram.py
@@ -1,38 +1,66 @@
import json
import requests
-import re
+import html
import os
+import sys
+import time
from datetime import datetime, timedelta
import glob
-def send_telegram_message(bot_token, chat_id, message):
+def send_telegram_message(bot_token, chat_id, message, attempts=4):
url = f"https://api.telegram.org/bot{bot_token}/sendMessage"
payload = {
"chat_id": chat_id,
"text": message,
- "parse_mode": "Markdown"
+ "parse_mode": "HTML"
}
- response = requests.post(url, json=payload)
- return response.json()
+ result = {}
+ for attempt in range(1, attempts + 1):
+ try:
+ response = requests.post(url, json=payload, timeout=(10, 30))
+ result = response.json()
+ except (requests.RequestException, ValueError) as e:
+ # Don't leak the bot token, which is part of the URL, into the log
+ result = {"ok": False, "description": str(e).replace(bot_token, '***')}
+ response = None
-def escape_markdown(text):
- """Escape special characters for Markdown in Telegram."""
- return re.sub(r'([*_`\[\]()])', r'\\\1', str(text))
+ if result.get('ok'):
+ return result
+ if response is not None and response.status_code < 500 and response.status_code != 429:
+ return result # e.g. a formatting error, retrying won't help
+ if attempt < attempts:
+ retry_after = result.get('parameters', {}).get('retry_after')
+ time.sleep(min(retry_after, 60) if retry_after else 5 * attempt)
+ return result
-def format_job_message(jobs):
- message = f"*New Relevant Jobs Found - {len(jobs)} jobs*\n\n"
+def format_job_blocks(jobs):
+ blocks = []
for i, job in enumerate(jobs, 1):
- title = escape_markdown(job.get('Title', 'No Title'))
- link = escape_markdown(job.get('Link', '#'))
- education = escape_markdown(job.get('Education Level', 'Not specified'))
- category = escape_markdown(job.get('Job Category', 'Not specified'))
- group = escape_markdown(job.get('Group Classification', 'Not specified'))
-
- message += f"{i}. [{title}]({link})\n"
- message += f" Education Level: {education}\n"
- message += f" Job Category: {category}\n"
- message += f" Group Classification: {group}\n\n"
- return message
+ title = html.escape(str(job.get('Title', 'No Title')))
+ link = html.escape(str(job.get('Link', '#')), quote=True)
+ education = html.escape(str(job.get('Education Level', 'Not specified')))
+ category = html.escape(str(job.get('Job Category', 'Not specified')))
+ group = html.escape(str(job.get('Group Classification', 'Not specified')))
+
+ block = f'{i}. {title}\n'
+ block += f" Education Level: {education}\n"
+ block += f" Job Category: {category}\n"
+ block += f" Group Classification: {group}\n\n"
+ blocks.append(block)
+ return blocks
+
+def split_messages(header, blocks, max_length=4000):
+ """Pack whole job blocks into messages, so a link is never cut in half."""
+ messages = []
+ current = header
+ for block in blocks:
+ if len(current) + len(block) > max_length:
+ messages.append(current)
+ current = ''
+ current += block
+ if current:
+ messages.append(current)
+ return messages
def get_latest_jobs_file(directory='relevant_jobs'):
pattern = os.path.join(directory, 'relevant_jobs_*.json')
@@ -61,7 +89,7 @@ def main():
return
try:
- with open(latest_jobs_file, 'r') as f:
+ with open(latest_jobs_file, 'r', encoding='utf-8') as f:
jobs = json.load(f)
except json.JSONDecodeError:
print(f"Error: Unable to parse JSON from {latest_jobs_file}.")
@@ -71,18 +99,23 @@ def main():
print("No jobs found in the file. Skipping notification.")
return
- message = format_job_message(jobs)
+ header = f"New Relevant Jobs Found - {len(jobs)} jobs\n\n"
+ # Telegram's max message length is 4096, we leave some buffer
+ messages = split_messages(header, format_job_blocks(jobs))
- # Split message if it's too long
- max_length = 4000 # Telegram's max message length is 4096, we leave some buffer
- messages = [message[i:i+max_length] for i in range(0, len(message), max_length)]
-
- for msg in messages:
+ failed = 0
+ for i, msg in enumerate(messages):
+ if i:
+ time.sleep(1) # stay below Telegram's flood limits
response = send_telegram_message(bot_token, chat_id, msg)
if response.get('ok'):
print(f"Message sent successfully. Using file: {latest_jobs_file}")
else:
+ failed += 1
print(f"Failed to send message. Error: {response.get('description')}")
+ if failed:
+ sys.exit(f"{failed} of {len(messages)} Telegram messages could not be sent.")
+
if __name__ == "__main__":
- main()
\ No newline at end of file
+ main()