JazzHR Jobs API.
Pull structured job data — titles, salaries, departments, and locations — from thousands of small-business boards on applytojob.com using plain HTTP and HTML parsing, no API keys required.
Try the API.
Test Jobs and Feed endpoints against https://connect.jobo.world with live request/response examples, then copy ready-to-use curl commands.
What's in every response.
Data fields, real-world applications, and the companies already running on JazzHR.
- Full Job Descriptions
- Salary & Compensation
- Department & Team Info
- Employment Type
- Experience Level
- Structured Locations
- 01SMB Job Market Monitoring
- 02Salary & Compensation Benchmarking
- 03Small-Business Talent Sourcing
- 04Multi-Company Job Aggregation
How to scrape JazzHR.
Step-by-step guide to extracting jobs from JazzHR-powered career pages—endpoints, authentication, and working code.
import requests
import xml.etree.ElementTree as ET
def discover_jazzhr_companies():
companies = set()
# Parse all 5 sitemap feeds (indices 0-4)
for i in range(5):
url = f"https://app.jazz.co/feeds/google/xml/{i}"
response = requests.get(url, timeout=30)
root = ET.fromstring(response.content)
# Extract company slugs from URLs
for url_elem in root.findall('.//{*}loc'):
job_url = url_elem.text
# URL format: https://{company}.applytojob.com/apply/{JOB_ID}/...
if 'applytojob.com' in job_url:
slug = job_url.split('//')[1].split('.')[0]
companies.add(slug)
return list(companies)
companies = discover_jazzhr_companies()
print(f"Found {len(companies)} companies")import requests
from bs4 import BeautifulSoup
company_slug = "agil3tech"
url = f"https://{company_slug}.applytojob.com/apply"
# Always use HTTPS (some links may use HTTP which can timeout)
response = requests.get(url, timeout=15, allow_redirects=False)
# Check if company is valid (invalid companies redirect away)
if response.status_code in (301, 302):
redirect_location = response.headers.get('Location', '')
if 'info.jazzhr.com' in redirect_location:
print(f"Company '{company_slug}' does not exist")
exit()
soup = BeautifulSoup(response.text, 'html.parser')
print(f"Successfully fetched job board for {company_slug}")from bs4 import BeautifulSoup
def parse_job_listings(soup):
jobs = []
# Primary selector: list-group structure
job_items = soup.select('li.list-group-item')
for item in job_items:
# Extract job URL and title
title_elem = item.select_one('h4.list-group-item-heading a')
if not title_elem:
title_elem = item.select_one('a[href*="/apply/"]')
if title_elem:
job_url = title_elem.get('href', '')
title = title_elem.get_text(strip=True)
# Extract location and department
info_items = item.select('ul.list-inline.list-group-item-text li')
location = info_items[0].get_text(strip=True) if len(info_items) > 0 else None
department = info_items[1].get_text(strip=True) if len(info_items) > 1 else None
# Extract job ID from URL
import re
job_id_match = re.search(r'/apply/([A-Za-z0-9]{8,})', job_url)
job_id = job_id_match.group(1) if job_id_match else None
jobs.append({
'id': job_id,
'title': title,
'url': job_url,
'location': location,
'department': department,
})
return jobs
jobs = parse_job_listings(soup)
print(f"Found {len(jobs)} jobs")import requests
from bs4 import BeautifulSoup
import re
def fetch_job_details(job_url):
# Normalize HTTP to HTTPS
if job_url.startswith('http://'):
job_url = job_url.replace('http://', 'https://')
response = requests.get(job_url, timeout=15)
soup = BeautifulSoup(response.text, 'html.parser')
details = {}
# Title (h2 on resumator themes, else h1)
title_elem = soup.select_one('div.job-header h2, div.job-header h1, h1')
details['title'] = title_elem.get_text(strip=True) if title_elem else None
# Location
location_elem = soup.select_one("div.job-attributes-container div[title='Location']")
details['location'] = location_elem.get_text(strip=True) if location_elem else None
# Description (resumator id first, then legacy id)
desc_elem = soup.select_one('#resumator-job-description, #job-description, .job-details .description')
details['description'] = desc_elem.get_text(strip=True) if desc_elem else None
# Compensation (dedicated field)
comp_elem = soup.select_one("div.job-attributes-container div[title='Compensation'], "
"div.job-attributes-container div[title='Salary'], "
"#resumator-job-salary")
details['compensation'] = comp_elem.get_text(strip=True) if comp_elem else None
# If no dedicated compensation field, search in description
if not details['compensation'] and details['description']:
salary_match = re.search(r'\$[\d,]+(?:\s*-\s*\$?[\d,]+)?(?:\s*(?:per|/)\s*\w+)?',
details['description'])
if salary_match:
details['compensation'] = salary_match.group()
return details
# Fetch details for first job
if jobs:
job_details = fetch_job_details(jobs[0]['url'])
print(job_details)import requests
import time
from bs4 import BeautifulSoup
def scrape_jazzhr_company(company_slug, delay_seconds=1):
base_url = f"https://{company_slug}.applytojob.com/apply"
try:
# Fetch listings
response = requests.get(base_url, timeout=15, allow_redirects=False)
# Validate company exists
if response.status_code in (301, 302):
return {'error': 'Company not found', 'jobs': []}
if response.status_code != 200:
return {'error': f'HTTP {response.status_code}', 'jobs': []}
soup = BeautifulSoup(response.text, 'html.parser')
jobs = parse_job_listings(soup)
# Fetch details for each job with delay
for i, job in enumerate(jobs):
try:
details = fetch_job_details(job['url'])
job.update(details)
print(f"Processed {i+1}/{len(jobs)}: {job['title']}")
except requests.RequestException as e:
job['error'] = str(e)
# Rate limiting delay
if i < len(jobs) - 1:
time.sleep(delay_seconds)
return {'company': company_slug, 'jobs': jobs}
except requests.RequestException as e:
return {'error': str(e), 'jobs': []}
# Scrape a company
result = scrape_jazzhr_company("agil3tech", delay_seconds=1.5)
print(f"Total jobs scraped: {len(result['jobs'])}")Invalid or inactive company subdomains redirect away from applytojob.com. Use allow_redirects=False and check for 301/302 redirects to info.jazzhr.com to detect companies that no longer exist.
Canonical JazzHR job codes are short alphanumeric tokens (roughly 8-24 chars) found on the tenant board. The 40+ character hex tokens in the app.jazz.co XML feed rotate between crawls and expire quickly (HTTP 410), so never treat them as stable IDs. Always scrape the board's /apply page for canonical codes.
Some job links use HTTP instead of HTTPS which can hang or timeout. Always normalize URLs by converting http:// to https:// for applytojob.com domains.
JazzHR ships several board themes. Newer 'resumator' themes render the title in div.job-header h2 (not h1) and the body in #resumator-job-description (not #job-description). Include both variants as fallback selectors, and fall back to og:title for the title.
Some jobs include salary in the description text instead of a dedicated field. If the Compensation/Salary selector returns nothing, search the description with a salary regex before treating the job as having no pay data.
Boards return HTTP 403/429 under heavy load. Add ~500 ms delays, cap concurrent detail fetches at ~3, and rotate user agents or residential proxies for large-scale runs.
- 1Use the XML feeds (indices 0-4) to discover companies, then scrape each board for canonical listings
- 2Always normalize HTTP URLs to HTTPS for applytojob.com
- 3Check for redirects to info.jazzhr.com to detect invalid companies
- 4Add ~500 ms delays and cap concurrent detail fetches to avoid 403/429
- 5Add fallback selectors (h2/h1 titles, resumator vs legacy description ids) for theme variations
- 6Search description text for compensation when the dedicated field is empty
One endpoint. All JazzHR jobs. No scraping, no sessions, no maintenance.
Get API accesscurl "https://connect.jobo.world/api/jobs?sources=jazzhr" \
-H "X-Api-Key: YOUR_KEY" Access JazzHR
job data today.
One API call. Structured data. No scraping infrastructure to build or maintain — start with the free tier and scale as you grow.