diff --git a/.github/workflows/deploy-job-php.yml b/.github/workflows/deploy-job-php.yml index b7835ef..85aac1e 100644 --- a/.github/workflows/deploy-job-php.yml +++ b/.github/workflows/deploy-job-php.yml @@ -9,7 +9,7 @@ jobs: steps: - name: Checkout code - uses: actions/checkout@v3 + uses: actions/checkout@v7 - name: Deploy via SFTP uses: wlixcc/SFTP-Deploy-Action@v1.2.4 diff --git a/.github/workflows/deploy-to-db.yml b/.github/workflows/deploy-to-db.yml index c06ad12..91daf46 100644 --- a/.github/workflows/deploy-to-db.yml +++ b/.github/workflows/deploy-to-db.yml @@ -20,10 +20,10 @@ jobs: steps: - name: Checkout repository - uses: actions/checkout@v3 + uses: actions/checkout@v7 - name: Set up Python - uses: actions/setup-python@v4 + uses: actions/setup-python@v7 with: python-version: '3.10' cache: 'pip' diff --git a/.github/workflows/python-package-conda.yml b/.github/workflows/python-package-conda.yml index 67a279e..8a25851 100644 --- a/.github/workflows/python-package-conda.yml +++ b/.github/workflows/python-package-conda.yml @@ -2,27 +2,42 @@ name: Job Scraping and Firebase Processing Pipeline on: schedule: - - cron: '0 0 * * *' # Run daily at midnight UTC + # Not on the full hour: GitHub delays or drops scheduled runs at :00 when it is busy. + # The second run catches up if the first one failed or was dropped. + - cron: '17 5 * * *' + - cron: '47 15 * * *' workflow_dispatch: # Allow manual triggering +permissions: + contents: write + +# Never run two scrapers at once (e.g. a manual run while the scheduled one is running) +concurrency: + group: job-scraper + cancel-in-progress: false + jobs: scrape-and-process: runs-on: ubuntu-latest + timeout-minutes: 45 + env: + PYTHONUNBUFFERED: '1' steps: - - uses: actions/checkout@v3 + - uses: actions/checkout@v7 with: - fetch-depth: 0 - persist-credentials: true + # Latest commit of the branch, not the trigger commit (matters when a run was queued) + ref: ${{ github.ref_name }} - name: Set up Python - uses: actions/setup-python@v4 + uses: actions/setup-python@v7 with: - python-version: '3.x' + python-version: '3.14' + cache: pip - name: Install dependencies run: | python -m pip install --upgrade pip - pip install requests beautifulsoup4 firebase-admin + pip install -r requirements.txt - name: Run job scraper run: python job_scraper.py @@ -35,21 +50,32 @@ jobs: # FIREBASE_SERVICE_ACCOUNT: ${{ secrets.FIREBASE_SERVICE_ACCOUNT }} # run: python firebase_importer.py + # Notify before committing: if Telegram fails, nothing is committed and the next run + # sends these jobs again. Only on main: test runs on other branches neither notify nor commit. - name: Send Telegram Notification + if: github.ref == 'refs/heads/main' env: TELEGRAM_BOT_TOKEN: ${{ secrets.TELEGRAM_BOT_TOKEN }} TELEGRAM_CHAT_ID: ${{ secrets.TELEGRAM_CHAT_ID }} run: python telegram.py - name: Commit and push if changes + if: github.ref == 'refs/heads/main' run: | git config --local user.email "action@github.com" git config --local user.name "GitHub Action" git add jobs_all_processed.json new_jobs.json - git diff --quiet && git diff --staged --quiet || (git commit -m "Update job data" && git push) - + git diff --staged --quiet && exit 0 + git commit -m "Update job data" + for attempt in 1 2 3; do + git push && exit 0 + sleep 10 + git pull --rebase origin main || { git rebase --abort; exit 1; } + done + exit 1 + # Trigger the database import workflow - + #- name: Trigger DB Import Workflow # if: success() # uses: peter-evans/repository-dispatch@v2 diff --git a/job_scraper.py b/job_scraper.py index 42c2024..dceb119 100644 --- a/job_scraper.py +++ b/job_scraper.py @@ -1,8 +1,13 @@ import requests +from requests.adapters import HTTPAdapter +from urllib3.util.retry import Retry from bs4 import BeautifulSoup import json from datetime import datetime import os +import random +import re +import sys import time # Import the time module for adding delay # JSON configuration @@ -33,27 +38,87 @@ } } +BASE_URL = 'https://govjobs.public.lu' +START_URL = 'https://govjobs.public.lu/fr/rechercher-parmi-offres-emploi.html' + +# Network settings. govjobs.public.lu refuses TCP connections ("Connection refused") when a +# client opens too many of them, so all requests share one keep-alive session. The server +# closes idle connections after 5 seconds, so the delays between requests stay below that. +REQUEST_TIMEOUT = (10, 30) # (connect, read) in seconds +LISTING_DELAY = 2 # seconds between listing pages (plus up to 1s jitter) +DETAIL_DELAY = 2 # seconds between detail pages (plus up to 1s jitter) +MAX_PAGES = 100 # safety net against endless pagination +MAX_DETAIL_FETCHES = 100 # per run; the rest is fetched on the next run +MAX_DETAIL_FAILURES = 3 # consecutive failed detail pages before details are paused for this run +TIME_BUDGET = 20 * 60 # seconds; afterwards the scraper stops and saves what it has +RETRY_STATUSES = (429, 500, 502, 503, 504) + + +class CappedRetry(Retry): + """Retry that never waits longer than 60 seconds, even if the server sends a larger Retry-After.""" + + def get_retry_after(self, response): + retry_after = super().get_retry_after(response) + return None if retry_after is None else min(retry_after, 60) + + +def create_session(): + # Refused connections and 429/5xx answers are retried after ~0, 6, 12, 24 and 48 seconds + retry = CappedRetry( + total=5, + connect=5, + read=2, + status=3, + backoff_factor=3, + backoff_max=60, + backoff_jitter=1, + status_forcelist=RETRY_STATUSES, + allowed_methods=frozenset({'GET'}), + raise_on_status=False, + ) + session = requests.Session() + session.mount('https://', HTTPAdapter(max_retries=retry)) + session.headers.update({ + 'User-Agent': 'GovScrapperJobMail/1.0 (+https://github.com/masterries/GovScrapperJobMail)', + 'Accept': 'text/html,application/xhtml+xml', + 'Accept-Language': 'fr,en;q=0.8', + }) + return session + + +session = create_session() + + +def polite_sleep(seconds): + time.sleep(seconds + random.uniform(0, 1)) + + def scrape_job_details(url): - """Scrape detailed information from the job's detail page""" + """Scrape detailed information from the job's detail page. + + Raises requests.RequestException when the page could not be fetched (refused connection, + timeout, 429/5xx after all retries) so the caller can back off. + """ print(f'Scraping details: {url}') job_details = {} - + + response = session.get(url, timeout=REQUEST_TIMEOUT) + if response.status_code in (404, 410): + print(f"Failed to get details from {url}: Status code {response.status_code}") + return job_details + response.raise_for_status() + try: - response = requests.get(url) - if response.status_code != 200: - print(f"Failed to get details from {url}: Status code {response.status_code}") - return job_details - soup = BeautifulSoup(response.content, 'html.parser') - + # Find the main content div that contains the detailed job description page_text_div = soup.find('div', class_='page-text') if not page_text_div: return job_details - + # Extract all text content from the page-text div job_details['Full Description'] = page_text_div.get_text(separator=' ', strip=True) - + # Try to extract structured data from the page-text div # Look for headers/titles followed by content headers = page_text_div.find_all(['h2', 'h3', 'h4', 'strong', 'b']) @@ -68,10 +133,10 @@ def scrape_job_details(url): break if sibling.name == 'p' or sibling.name == 'ul' or sibling.name == 'div': content.append(sibling.get_text(strip=True)) - + if content: job_details[header_text] = ' '.join(content) - + # Try to extract any table data if present tables = page_text_div.find_all('table') for i, table in enumerate(tables): @@ -84,108 +149,188 @@ def scrape_job_details(url): job_details[row_data[0]] = row_data[1] elif row_data: table_data.append(row_data) - + if table_data and i == 0: job_details['Table Data'] = table_data - + except Exception as e: - print(f"Error scraping details from {url}: {str(e)}") - + print(f"Error parsing details from {url}: {str(e)}") + return job_details -def scrape_jobs(): - base_url = 'https://govjobs.public.lu' - start_url = 'https://govjobs.public.lu/fr/rechercher-parmi-offres-emploi.html' - all_jobs = [] - page_number = 0 - - # Load existing jobs to check if we need to fetch details - existing_jobs_dict = {} - if os.path.exists('jobs_all_processed.json'): - try: - with open('jobs_all_processed.json', 'r', encoding='utf-8') as f: - existing_jobs = json.load(f) - existing_jobs_dict = {job['Link']: job for job in existing_jobs if 'Link' in job} - print(f"Loaded {len(existing_jobs_dict)} existing jobs for reference") - except Exception as e: - print(f"Error loading existing jobs: {str(e)}") - - while True: - url = f'{start_url}?b={page_number * 20}' +def parse_result_count(soup): + """Number of jobs the search page says it has, or None if it can't be read.""" + count_tag = soup.find(class_='search-meta-count') + if not count_tag: + return None + match = re.search(r'\d+', count_tag.get_text().replace(' ', '').replace('\xa0', '')) + return int(match.group()) if match else None + +def parse_listing_article(article): + job = {} + title_tag = article.find('h2', class_='article-title') + a_tag = title_tag.find('a') if title_tag else None + if not a_tag or not a_tag.get('href'): + return None + + job['Titel'] = a_tag.text.strip() + link = a_tag['href'] + if link.startswith('//'): + link = 'https:' + link + elif link.startswith('/'): + link = BASE_URL + link + else: + link = BASE_URL + '/' + link.lstrip('/') + job['Link'] = link + + footer = article.find('footer', class_='article-metas') + if footer: + meta_list = footer.find('ul', class_='list--inline list--dotted') + if meta_list: + for item in meta_list.find_all('li'): + text = item.get_text(separator=' ').strip() + key, value = text.split(': ', 1) if ': ' in text else (text, '') + job[key] = value + + custom_list = article.find('ul', class_='nude article-custom') + if custom_list: + for item in custom_list.find_all('li'): + span, b_tag = item.find('span'), item.find('b') + if span and b_tag: + job[span.text.strip()] = b_tag.text.strip() + + return job + +def scrape_listing(deadline): + """Collect all jobs from the search result pages. + + Returns (jobs, complete); complete is False when the crawl had to stop early. + """ + jobs = [] + seen_links = set() + expected_total = None + + for page_number in range(MAX_PAGES): + if time.monotonic() > deadline: + print('::warning::Time budget exceeded while crawling the listing pages') + return jobs, False + if page_number: + polite_sleep(LISTING_DELAY) + + url = f'{START_URL}?b={page_number * 20}' print(f'Scraping page: {url}') + try: + response = session.get(url, timeout=REQUEST_TIMEOUT) + response.raise_for_status() + except requests.RequestException as e: + print(f'::warning::Listing crawl aborted at {url}: {e}') + return jobs, False - response = requests.get(url) soup = BeautifulSoup(response.content, 'html.parser') + if expected_total is None: + expected_total = parse_result_count(soup) articles = soup.find_all('article', class_='article search-result search-result--job') if not articles: + if not soup.find('ol', class_='search-results'): + # Not a regular (empty) result page, e.g. a maintenance or error page + print(f'::warning::Unexpected page without search results at {url}') + return jobs, False break + new_on_page = 0 for article in articles: - job = {} - title_tag = article.find('h2', class_='article-title') - if title_tag and title_tag.find('a'): - a_tag = title_tag.find('a') - job['Titel'] = a_tag.text.strip() - link = a_tag['href'] - if link.startswith('//'): - link = 'https:' + link - elif link.startswith('/'): - link = base_url + link - else: - link = base_url + '/' + link.lstrip('/') - job['Link'] = link - - footer = article.find('footer', class_='article-metas') - if footer: - meta_list = footer.find('ul', class_='list--inline list--dotted') - if meta_list: - for item in meta_list.find_all('li'): - text = item.get_text(separator=' ').strip() - key, value = text.split(': ', 1) if ': ' in text else (text, '') - job[key] = value - - custom_list = article.find('ul', class_='nude article-custom') - if custom_list: - for item in custom_list.find_all('li'): - span, b_tag = item.find('span'), item.find('b') - if span and b_tag: - job[span.text.strip()] = b_tag.text.strip() - - # Get detailed information from the job page if needed - if 'Link' in job: - existing_job = existing_jobs_dict.get(job['Link']) - - # Check if we need to fetch details - needs_details = True - if existing_job: - # Check if the existing job already has detailed info - if 'Full Description' in existing_job or any(key.startswith('Section:') for key in existing_job): - needs_details = False - print(f"Using cached details for: {job.get('Titel', job['Link'])}") - - if needs_details: - # Add a delay between requests to avoid overloading the server - time.sleep(1) - print(f"Fetching details for: {job.get('Titel', job['Link'])}") - job_details = scrape_job_details(job['Link']) - - # Merge the details with the main job data - for key, value in job_details.items(): - if key not in job: # Don't overwrite existing data - job[key] = value - elif existing_job: - # Copy the detailed fields from the existing job - for key, value in existing_job.items(): - if key not in job and key != 'adding_date': - job[key] = value - - all_jobs.append(job) - - page_number += 1 - time.sleep(2) # Add a 2-second delay between page scrapes - - return process_jobs(all_jobs) + job = parse_listing_article(article) + if not job: + print(f'Skipping a search result without link on {url}') + continue + if job['Link'] in seen_links: + continue # the listing shifted while crawling + seen_links.add(job['Link']) + jobs.append(job) + new_on_page += 1 + + if new_on_page == 0: + break # the site repeats pages we already have + else: + print(f'::warning::Stopped after {MAX_PAGES} listing pages') + return jobs, False + + if expected_total is not None and len(jobs) < expected_total: + print(f'::warning::The listing shows {expected_total} jobs but only {len(jobs)} were collected') + return jobs, True + +def add_job_details(jobs, existing_jobs_dict, deadline): + """Add the detail page data to the listed jobs, from the cache where possible. + + Returns False when some missing details could not be fetched in this run. Those jobs are + saved without 'Full Description' and fetched again on the next run. + """ + complete = True + fetches = 0 + consecutive_failures = 0 + + for job in jobs: + existing_job = existing_jobs_dict.get(job['Link']) + + # Check if the existing job already has detailed info + if existing_job and ('Full Description' in existing_job or any(key.startswith('Section:') for key in existing_job)): + print(f"Using cached details for: {job.get('Titel', job['Link'])}") + # Copy the detailed fields from the existing job + for key, value in existing_job.items(): + if key not in job and key != 'adding_date': + job[key] = value + continue + + if not complete: + continue + if fetches >= MAX_DETAIL_FETCHES or time.monotonic() > deadline: + print('::warning::Detail fetch limit reached; the remaining details are fetched on the next run') + complete = False + continue + + # Add a delay between requests to avoid overloading the server + polite_sleep(DETAIL_DELAY) + print(f"Fetching details for: {job.get('Titel', job['Link'])}") + fetches += 1 + try: + job_details = scrape_job_details(job['Link']) + except requests.RequestException as e: + print(f"Error scraping details from {job['Link']}: {str(e)}") + consecutive_failures += 1 + if consecutive_failures >= MAX_DETAIL_FAILURES: + print(f'::warning::{consecutive_failures} detail pages failed in a row; the remaining details are fetched on the next run') + complete = False + continue + consecutive_failures = 0 + + # Merge the details with the main job data + for key, value in job_details.items(): + if key not in job: # Don't overwrite existing data + job[key] = value + + return complete + +def scrape_jobs(): + """Scrape the listing, then the missing detail pages. + + Returns (processed_jobs, complete). + """ + deadline = time.monotonic() + TIME_BUDGET + + # Load existing jobs to check if we need to fetch details + existing_jobs_dict = {} + if os.path.exists('jobs_all_processed.json'): + with open('jobs_all_processed.json', 'r', encoding='utf-8') as f: + existing_jobs = json.load(f) + existing_jobs_dict = {job['Link']: job for job in existing_jobs if 'Link' in job} + print(f"Loaded {len(existing_jobs_dict)} existing jobs for reference") + + # Listing first, so a blocked detail page can't cost us the list of new jobs + jobs, listing_complete = scrape_listing(deadline) + details_complete = add_job_details(jobs, existing_jobs_dict, deadline) + + return process_jobs(jobs), listing_complete and details_complete def process_jobs(jobs): processed_jobs = [] @@ -194,17 +339,17 @@ def process_jobs(jobs): for fr_key, value in job.items(): en_key = json_config['column_mapping'].get(fr_key, fr_key) processed_job[en_key] = value - + for group_name, fields in json_config['group_fields'].items(): group_value = next((processed_job[field] for field in fields if field in processed_job), None) if group_value: processed_job[group_name] = group_value for field in fields: processed_job.pop(field, None) - + processed_job['adding_date'] = datetime.now().isoformat() processed_jobs.append(processed_job) - + return processed_jobs def update_json(new_jobs, filename='jobs_all_processed.json'): @@ -216,7 +361,7 @@ def update_json(new_jobs, filename='jobs_all_processed.json'): existing_links = {job['Link'] for job in existing_jobs} existing_jobs_dict = {job['Link']: job for job in existing_jobs} - + updated_jobs = existing_jobs.copy() new_jobs_added = [] @@ -226,17 +371,18 @@ def update_json(new_jobs, filename='jobs_all_processed.json'): updated_jobs.append(job) new_jobs_added.append(job) existing_links.add(job['Link']) + existing_jobs_dict[job['Link']] = job else: # The job exists, but check if we need to update with new detail info existing_job = existing_jobs_dict[job['Link']] - + # Check if the job has any new fields from the detail page that the existing one doesn't has_new_details = False for key, value in job.items(): if key not in existing_job and key != 'adding_date': has_new_details = True break - + if has_new_details: # Update the existing job with new details while preserving the original adding_date original_date = existing_job.get('adding_date') @@ -260,52 +406,42 @@ def test_scrape_details(num_jobs=2): """ Test function to scrape only a limited number of jobs with their details for testing purposes. - + Args: num_jobs: Number of jobs to scrape (default 2) """ - base_url = 'https://govjobs.public.lu' - start_url = 'https://govjobs.public.lu/fr/rechercher-parmi-offres-emploi.html' test_jobs = [] - + # Get the first page only - print(f'Test scraping: {start_url}') - response = requests.get(start_url) + print(f'Test scraping: {START_URL}') + response = session.get(START_URL, timeout=REQUEST_TIMEOUT) + response.raise_for_status() soup = BeautifulSoup(response.content, 'html.parser') articles = soup.find_all('article', class_='article search-result search-result--job') - + # Limit to the specified number of jobs articles = articles[:num_jobs] - + for article in articles: - job = {} - title_tag = article.find('h2', class_='article-title') - if title_tag and title_tag.find('a'): - a_tag = title_tag.find('a') - job['Title'] = a_tag.text.strip() - link = a_tag['href'] - if link.startswith('//'): - link = 'https:' + link - elif link.startswith('/'): - link = base_url + link - else: - link = base_url + '/' + link.lstrip('/') - job['Link'] = link - + job = parse_listing_article(article) + if job: + job['Title'] = job.pop('Titel') + # Get the job details print(f'Testing detail scraping for: {job["Title"]}') + polite_sleep(DETAIL_DELAY) job_details = scrape_job_details(job['Link']) - + # Merge the details with the basic job data job.update(job_details) - + test_jobs.append(job) - + # Save the test results to a file test_file = 'test_job_details.json' with open(test_file, 'w', encoding='utf-8') as f: json.dump(test_jobs, f, ensure_ascii=False, indent=4) - + print(f"Test completed. {len(test_jobs)} jobs processed and saved to {test_file}") return test_jobs @@ -313,16 +449,23 @@ def main(): # Uncomment the test function to run in test mode # test_scrape_details(2) # return - - scraped_jobs = scrape_jobs() + + scraped_jobs, complete = scrape_jobs() + if not scraped_jobs: + print('::error::No jobs could be scraped, nothing was saved') + sys.exit(1) + new_jobs = update_json(scraped_jobs) save_new_jobs(new_jobs) print(f"Scraping completed. {len(scraped_jobs)} jobs processed.") print(f"{len(new_jobs)} new jobs added to 'jobs_all_processed.json' and saved to 'new_jobs.json'") + if not complete: + # Partial data is still saved and committed; the rest is picked up on the next run + print('::warning::Scraping was incomplete, the missing jobs/details are fetched on the next run') if __name__ == "__main__": # For testing, uncomment this line: # test_scrape_details(2) - + # For regular operation, keep this line: - main() \ No newline at end of file + main() diff --git a/requirements.txt b/requirements.txt index fad3350..70974c3 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,2 +1,3 @@ -tqdm -firebase_admin \ No newline at end of file +requests>=2.32,<3 +urllib3>=2,<3 +beautifulsoup4>=4.12,<5 diff --git a/telegram.py b/telegram.py index 420af5b..b1d3777 100644 --- a/telegram.py +++ b/telegram.py @@ -1,38 +1,66 @@ import json import requests -import re +import html import os +import sys +import time from datetime import datetime, timedelta import glob -def send_telegram_message(bot_token, chat_id, message): +def send_telegram_message(bot_token, chat_id, message, attempts=4): url = f"https://api.telegram.org/bot{bot_token}/sendMessage" payload = { "chat_id": chat_id, "text": message, - "parse_mode": "Markdown" + "parse_mode": "HTML" } - response = requests.post(url, json=payload) - return response.json() + result = {} + for attempt in range(1, attempts + 1): + try: + response = requests.post(url, json=payload, timeout=(10, 30)) + result = response.json() + except (requests.RequestException, ValueError) as e: + # Don't leak the bot token, which is part of the URL, into the log + result = {"ok": False, "description": str(e).replace(bot_token, '***')} + response = None -def escape_markdown(text): - """Escape special characters for Markdown in Telegram.""" - return re.sub(r'([*_`\[\]()])', r'\\\1', str(text)) + if result.get('ok'): + return result + if response is not None and response.status_code < 500 and response.status_code != 429: + return result # e.g. a formatting error, retrying won't help + if attempt < attempts: + retry_after = result.get('parameters', {}).get('retry_after') + time.sleep(min(retry_after, 60) if retry_after else 5 * attempt) + return result -def format_job_message(jobs): - message = f"*New Relevant Jobs Found - {len(jobs)} jobs*\n\n" +def format_job_blocks(jobs): + blocks = [] for i, job in enumerate(jobs, 1): - title = escape_markdown(job.get('Title', 'No Title')) - link = escape_markdown(job.get('Link', '#')) - education = escape_markdown(job.get('Education Level', 'Not specified')) - category = escape_markdown(job.get('Job Category', 'Not specified')) - group = escape_markdown(job.get('Group Classification', 'Not specified')) - - message += f"{i}. [{title}]({link})\n" - message += f" Education Level: {education}\n" - message += f" Job Category: {category}\n" - message += f" Group Classification: {group}\n\n" - return message + title = html.escape(str(job.get('Title', 'No Title'))) + link = html.escape(str(job.get('Link', '#')), quote=True) + education = html.escape(str(job.get('Education Level', 'Not specified'))) + category = html.escape(str(job.get('Job Category', 'Not specified'))) + group = html.escape(str(job.get('Group Classification', 'Not specified'))) + + block = f'{i}. {title}\n' + block += f" Education Level: {education}\n" + block += f" Job Category: {category}\n" + block += f" Group Classification: {group}\n\n" + blocks.append(block) + return blocks + +def split_messages(header, blocks, max_length=4000): + """Pack whole job blocks into messages, so a link is never cut in half.""" + messages = [] + current = header + for block in blocks: + if len(current) + len(block) > max_length: + messages.append(current) + current = '' + current += block + if current: + messages.append(current) + return messages def get_latest_jobs_file(directory='relevant_jobs'): pattern = os.path.join(directory, 'relevant_jobs_*.json') @@ -61,7 +89,7 @@ def main(): return try: - with open(latest_jobs_file, 'r') as f: + with open(latest_jobs_file, 'r', encoding='utf-8') as f: jobs = json.load(f) except json.JSONDecodeError: print(f"Error: Unable to parse JSON from {latest_jobs_file}.") @@ -71,18 +99,23 @@ def main(): print("No jobs found in the file. Skipping notification.") return - message = format_job_message(jobs) + header = f"New Relevant Jobs Found - {len(jobs)} jobs\n\n" + # Telegram's max message length is 4096, we leave some buffer + messages = split_messages(header, format_job_blocks(jobs)) - # Split message if it's too long - max_length = 4000 # Telegram's max message length is 4096, we leave some buffer - messages = [message[i:i+max_length] for i in range(0, len(message), max_length)] - - for msg in messages: + failed = 0 + for i, msg in enumerate(messages): + if i: + time.sleep(1) # stay below Telegram's flood limits response = send_telegram_message(bot_token, chat_id, msg) if response.get('ok'): print(f"Message sent successfully. Using file: {latest_jobs_file}") else: + failed += 1 print(f"Failed to send message. Error: {response.get('description')}") + if failed: + sys.exit(f"{failed} of {len(messages)} Telegram messages could not be sent.") + if __name__ == "__main__": - main() \ No newline at end of file + main()