initial public release
This commit is contained in:
@@ -0,0 +1,668 @@
|
||||
package scan
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// maxResponseBytes caps any single upstream response we'll parse.
|
||||
// Guards against OOM from maliciously-large or buggy feeds.
|
||||
const maxResponseBytes = 25 * 1024 * 1024 // 25 MB. Bumped from 5 MB: OpenAI's Ashby board is ~12 MB.
|
||||
|
||||
// DetectAdapter resolves a Company into an API endpoint, or returns ok=false.
|
||||
func DetectAdapter(c Company) (AdapterInfo, bool) {
|
||||
// Explicit Greenhouse API override.
|
||||
if c.API != "" && strings.Contains(c.API, "greenhouse") {
|
||||
return AdapterInfo{Type: AdapterGreenhouse, URL: c.API}, true
|
||||
}
|
||||
// Explicit Workday config block.
|
||||
if c.Workday != nil && c.Workday.Host != "" && c.Workday.Tenant != "" && c.Workday.Site != "" {
|
||||
host := strings.TrimRight(c.Workday.Host, "/")
|
||||
return AdapterInfo{
|
||||
Type: AdapterWorkday,
|
||||
URL: fmt.Sprintf("%s/wday/cxs/%s/%s/jobs", host, c.Workday.Tenant, c.Workday.Site),
|
||||
Host: host,
|
||||
Site: c.Workday.Site,
|
||||
}, true
|
||||
}
|
||||
|
||||
// Any explicit API URL that isn't greenhouse — infer from domain.
|
||||
if c.API != "" {
|
||||
u := c.API
|
||||
switch {
|
||||
case strings.Contains(u, "boards-api.greenhouse.io"):
|
||||
return AdapterInfo{Type: AdapterGreenhouse, URL: u}, true
|
||||
case strings.Contains(u, "api.lever.co"):
|
||||
return AdapterInfo{Type: AdapterLever, URL: u}, true
|
||||
case strings.Contains(u, "api.ashbyhq.com"):
|
||||
return AdapterInfo{Type: AdapterAshby, URL: u}, true
|
||||
case strings.Contains(u, ".bamboohr.com"):
|
||||
return AdapterInfo{Type: AdapterBambooHR, URL: u}, true
|
||||
case strings.Contains(u, ".teamtailor.com"):
|
||||
return AdapterInfo{Type: AdapterTeamtailor, URL: u}, true
|
||||
}
|
||||
}
|
||||
|
||||
url := c.CareersURL
|
||||
if m := reAshby.FindStringSubmatch(url); len(m) == 2 {
|
||||
return AdapterInfo{
|
||||
Type: AdapterAshby,
|
||||
URL: fmt.Sprintf("https://api.ashbyhq.com/posting-api/job-board/%s?includeCompensation=true", m[1]),
|
||||
}, true
|
||||
}
|
||||
if m := reLever.FindStringSubmatch(url); len(m) == 2 {
|
||||
return AdapterInfo{
|
||||
Type: AdapterLever,
|
||||
URL: fmt.Sprintf("https://api.lever.co/v0/postings/%s", m[1]),
|
||||
}, true
|
||||
}
|
||||
if m := reGreenhouse.FindStringSubmatch(url); len(m) == 2 {
|
||||
return AdapterInfo{
|
||||
Type: AdapterGreenhouse,
|
||||
URL: fmt.Sprintf("https://boards-api.greenhouse.io/v1/boards/%s/jobs", m[1]),
|
||||
}, true
|
||||
}
|
||||
if m := reBambooHR.FindStringSubmatch(url); len(m) == 2 {
|
||||
return AdapterInfo{
|
||||
Type: AdapterBambooHR,
|
||||
URL: fmt.Sprintf("https://%s.bamboohr.com/careers/list", m[1]),
|
||||
}, true
|
||||
}
|
||||
if m := reTeamtailor.FindStringSubmatch(url); len(m) == 2 {
|
||||
return AdapterInfo{
|
||||
Type: AdapterTeamtailor,
|
||||
URL: fmt.Sprintf("https://%s.teamtailor.com/jobs.rss", m[1]),
|
||||
}, true
|
||||
}
|
||||
|
||||
// scan_method: linkedin — browser-based search; ScanQuery holds keywords.
|
||||
if c.ScanMethod == "linkedin" {
|
||||
return AdapterInfo{Type: AdapterLinkedIn}, true
|
||||
}
|
||||
|
||||
return AdapterInfo{Type: AdapterUnknown}, false
|
||||
}
|
||||
|
||||
var (
|
||||
reAshby = regexp.MustCompile(`jobs\.ashbyhq\.com/([^/?#]+)`)
|
||||
reLever = regexp.MustCompile(`jobs\.lever\.co/([^/?#]+)`)
|
||||
reGreenhouse = regexp.MustCompile(`job-boards(?:\.eu)?\.greenhouse\.io/([^/?#]+)`)
|
||||
reBambooHR = regexp.MustCompile(`https?://([^./]+)\.bamboohr\.com`)
|
||||
reTeamtailor = regexp.MustCompile(`https?://([^./]+)\.teamtailor\.com`)
|
||||
)
|
||||
|
||||
// Fetch runs the right HTTP dance per adapter and returns normalized jobs.
|
||||
func Fetch(ctx context.Context, client *http.Client, c Company, info AdapterInfo) ([]Job, error) {
|
||||
switch info.Type {
|
||||
case AdapterGreenhouse:
|
||||
return fetchGreenhouse(ctx, client, info.URL, c.Name)
|
||||
case AdapterAshby:
|
||||
return fetchAshby(ctx, client, info.URL, c.Name)
|
||||
case AdapterLever:
|
||||
return fetchLever(ctx, client, info.URL, c.Name)
|
||||
case AdapterWorkday:
|
||||
return fetchWorkday(ctx, client, info, c.Name)
|
||||
case AdapterBambooHR:
|
||||
return fetchBambooHR(ctx, client, info.URL, c.Name)
|
||||
case AdapterTeamtailor:
|
||||
return fetchTeamtailor(ctx, client, info.URL, c.Name)
|
||||
case AdapterRemotive:
|
||||
return fetchRemotive(ctx, client, info.URL)
|
||||
case AdapterRemoteOK:
|
||||
return fetchRemoteOK(ctx, client, info.URL, info.UserAgent)
|
||||
case AdapterUSAJobs:
|
||||
return fetchUSAJobs(ctx, client, info)
|
||||
case AdapterRSS:
|
||||
return fetchRSS(ctx, client, info.URL, info.DefaultCompany)
|
||||
case AdapterLinkedIn:
|
||||
return fetchLinkedIn(ctx, c.ScanQuery, "", c.Name)
|
||||
default:
|
||||
return nil, fmt.Errorf("unsupported adapter type: %s", info.Type)
|
||||
}
|
||||
}
|
||||
|
||||
// DetectAggregator resolves an Aggregator into a scan-ready AdapterInfo.
|
||||
// Returns ok=false for unknown or misconfigured aggregator types.
|
||||
func DetectAggregator(a Aggregator) (AdapterInfo, bool) {
|
||||
limit := a.Limit
|
||||
if limit <= 0 {
|
||||
limit = 50
|
||||
}
|
||||
switch strings.ToLower(a.Type) {
|
||||
case "remotive":
|
||||
q := url.QueryEscape(a.Query)
|
||||
return AdapterInfo{
|
||||
Type: AdapterRemotive,
|
||||
URL: fmt.Sprintf("https://remotive.com/api/remote-jobs?search=%s&limit=%d", q, limit),
|
||||
}, true
|
||||
case "remoteok":
|
||||
tag := a.Tag
|
||||
if tag == "" {
|
||||
tag = "security"
|
||||
}
|
||||
return AdapterInfo{
|
||||
Type: AdapterRemoteOK,
|
||||
URL: fmt.Sprintf("https://remoteok.io/api?tags=%s", url.QueryEscape(tag)),
|
||||
UserAgent: a.UserAgent,
|
||||
}, true
|
||||
case "usajobs":
|
||||
apiKey := a.APIKey
|
||||
if apiKey == "" {
|
||||
apiKey = os.Getenv("USAJOBS_API_KEY")
|
||||
}
|
||||
ua := a.UserAgent
|
||||
if ua == "" {
|
||||
ua = os.Getenv("USAJOBS_USER_AGENT")
|
||||
}
|
||||
return AdapterInfo{
|
||||
Type: AdapterUSAJobs,
|
||||
URL: fmt.Sprintf("https://data.usajobs.gov/api/search?Keyword=%s&ResultsPerPage=%d", url.QueryEscape(a.Query), limit),
|
||||
APIKey: apiKey,
|
||||
UserAgent: ua,
|
||||
}, true
|
||||
case "rss":
|
||||
if a.URL == "" {
|
||||
return AdapterInfo{Type: AdapterUnknown}, false
|
||||
}
|
||||
return AdapterInfo{
|
||||
Type: AdapterRSS,
|
||||
URL: a.URL,
|
||||
DefaultCompany: a.Company,
|
||||
}, true
|
||||
}
|
||||
return AdapterInfo{Type: AdapterUnknown}, false
|
||||
}
|
||||
|
||||
func getJSON(ctx context.Context, client *http.Client, url string, headers map[string]string, out interface{}) error {
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for k, v := range headers {
|
||||
req.Header.Set(k, v)
|
||||
}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
return fmt.Errorf("HTTP %d", resp.StatusCode)
|
||||
}
|
||||
return json.NewDecoder(io.LimitReader(resp.Body, maxResponseBytes)).Decode(out)
|
||||
}
|
||||
|
||||
// Greenhouse: { jobs: [ { title, absolute_url, location:{name} } ] }
|
||||
func fetchGreenhouse(ctx context.Context, client *http.Client, url, company string) ([]Job, error) {
|
||||
var payload struct {
|
||||
Jobs []struct {
|
||||
Title string `json:"title"`
|
||||
AbsoluteURL string `json:"absolute_url"`
|
||||
Location struct {
|
||||
Name string `json:"name"`
|
||||
} `json:"location"`
|
||||
} `json:"jobs"`
|
||||
}
|
||||
if err := getJSON(ctx, client, url, nil, &payload); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Job, 0, len(payload.Jobs))
|
||||
for _, j := range payload.Jobs {
|
||||
out = append(out, Job{
|
||||
Title: j.Title,
|
||||
URL: j.AbsoluteURL,
|
||||
Company: company,
|
||||
Location: j.Location.Name,
|
||||
Source: "greenhouse",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Ashby (posting-api): { jobs: [ { title, jobUrl, location } ] }
|
||||
func fetchAshby(ctx context.Context, client *http.Client, url, company string) ([]Job, error) {
|
||||
var payload struct {
|
||||
Jobs []struct {
|
||||
Title string `json:"title"`
|
||||
JobURL string `json:"jobUrl"`
|
||||
Location string `json:"location"`
|
||||
} `json:"jobs"`
|
||||
}
|
||||
if err := getJSON(ctx, client, url, nil, &payload); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Job, 0, len(payload.Jobs))
|
||||
for _, j := range payload.Jobs {
|
||||
out = append(out, Job{
|
||||
Title: j.Title,
|
||||
URL: j.JobURL,
|
||||
Company: company,
|
||||
Location: j.Location,
|
||||
Source: "ashby",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Lever: [ { text, hostedUrl, categories:{location} } ]
|
||||
func fetchLever(ctx context.Context, client *http.Client, url, company string) ([]Job, error) {
|
||||
var payload []struct {
|
||||
Text string `json:"text"`
|
||||
HostedURL string `json:"hostedUrl"`
|
||||
Categories struct {
|
||||
Location string `json:"location"`
|
||||
} `json:"categories"`
|
||||
}
|
||||
if err := getJSON(ctx, client, url, nil, &payload); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Job, 0, len(payload))
|
||||
for _, j := range payload {
|
||||
out = append(out, Job{
|
||||
Title: j.Text,
|
||||
URL: j.HostedURL,
|
||||
Company: company,
|
||||
Location: j.Categories.Location,
|
||||
Source: "lever",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Workday: POST with pagination.
|
||||
func fetchWorkday(ctx context.Context, client *http.Client, info AdapterInfo, company string) ([]Job, error) {
|
||||
const limit = 20
|
||||
const safetyCap = 1000
|
||||
var all []Job
|
||||
offset := 0
|
||||
for offset < safetyCap {
|
||||
body := fmt.Sprintf(`{"appliedFacets":{},"limit":%d,"offset":%d,"searchText":""}`, limit, offset)
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, info.URL, bytes.NewBufferString(body))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode >= 400 {
|
||||
resp.Body.Close()
|
||||
return nil, fmt.Errorf("HTTP %d", resp.StatusCode)
|
||||
}
|
||||
var payload struct {
|
||||
JobPostings []struct {
|
||||
Title string `json:"title"`
|
||||
ExternalPath string `json:"externalPath"`
|
||||
LocationsText string `json:"locationsText"`
|
||||
} `json:"jobPostings"`
|
||||
}
|
||||
if err := json.NewDecoder(io.LimitReader(resp.Body, maxResponseBytes)).Decode(&payload); err != nil {
|
||||
resp.Body.Close()
|
||||
return nil, err
|
||||
}
|
||||
resp.Body.Close()
|
||||
for _, j := range payload.JobPostings {
|
||||
// Workday sometimes returns odd externalPath values. Reject any
|
||||
// that isn't a clean absolute segment (no traversal, no scheme).
|
||||
if !strings.HasPrefix(j.ExternalPath, "/") || strings.Contains(j.ExternalPath, "..") {
|
||||
continue
|
||||
}
|
||||
url := fmt.Sprintf("%s/en-US/%s%s", info.Host, info.Site, j.ExternalPath)
|
||||
all = append(all, Job{
|
||||
Title: j.Title,
|
||||
URL: url,
|
||||
Company: company,
|
||||
Location: j.LocationsText,
|
||||
Source: "workday",
|
||||
})
|
||||
}
|
||||
if len(payload.JobPostings) < limit {
|
||||
break
|
||||
}
|
||||
offset += limit
|
||||
}
|
||||
return all, nil
|
||||
}
|
||||
|
||||
// BambooHR list: { result: [ { id, jobOpeningName, jobOpeningShareUrl|null, locationCity } ] }
|
||||
func fetchBambooHR(ctx context.Context, client *http.Client, url, company string) ([]Job, error) {
|
||||
var payload struct {
|
||||
Result []struct {
|
||||
ID int `json:"id"`
|
||||
JobOpeningName string `json:"jobOpeningName"`
|
||||
JobOpeningShareURL string `json:"jobOpeningShareUrl"`
|
||||
LocationCity string `json:"locationCity"`
|
||||
} `json:"result"`
|
||||
}
|
||||
if err := getJSON(ctx, client, url, map[string]string{"Accept": "application/json"}, &payload); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
host := url // list endpoint lives on the same subdomain
|
||||
if i := strings.Index(host, "/careers"); i > 0 {
|
||||
host = host[:i]
|
||||
}
|
||||
out := make([]Job, 0, len(payload.Result))
|
||||
for _, j := range payload.Result {
|
||||
link := j.JobOpeningShareURL
|
||||
if link == "" {
|
||||
link = fmt.Sprintf("%s/careers/%d/detail", host, j.ID)
|
||||
}
|
||||
out = append(out, Job{
|
||||
Title: j.JobOpeningName,
|
||||
URL: link,
|
||||
Company: company,
|
||||
Location: j.LocationCity,
|
||||
Source: "bamboohr",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Teamtailor RSS: simple XML feed.
|
||||
func fetchTeamtailor(ctx context.Context, client *http.Client, url, company string) ([]Job, error) {
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Accept", "application/rss+xml, application/xml")
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("HTTP %d", resp.StatusCode)
|
||||
}
|
||||
body, err := io.ReadAll(io.LimitReader(resp.Body, maxResponseBytes))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Minimal RSS 2.0 parse. Disable entity expansion to avoid XXE.
|
||||
return parseRSSItems(body, company)
|
||||
}
|
||||
|
||||
// Minimal RSS <item><title>…</title><link>…</link></item> extractor with XXE off.
|
||||
func parseRSSItems(body []byte, company string) ([]Job, error) {
|
||||
type item struct {
|
||||
Title string `xml:"title"`
|
||||
Link string `xml:"link"`
|
||||
}
|
||||
type rss struct {
|
||||
Channel struct {
|
||||
Items []item `xml:"item"`
|
||||
} `xml:"channel"`
|
||||
}
|
||||
dec := newSafeXMLDecoder(bytes.NewReader(body))
|
||||
var feed rss
|
||||
if err := dec.Decode(&feed); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Job, 0, len(feed.Channel.Items))
|
||||
for _, it := range feed.Channel.Items {
|
||||
out = append(out, Job{
|
||||
Title: strings.TrimSpace(it.Title),
|
||||
URL: strings.TrimSpace(it.Link),
|
||||
Company: company,
|
||||
Source: "teamtailor",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Remotive: { jobs: [ { title, url, company_name, candidate_required_location } ] }
|
||||
func fetchRemotive(ctx context.Context, client *http.Client, url string) ([]Job, error) {
|
||||
var payload struct {
|
||||
Jobs []struct {
|
||||
Title string `json:"title"`
|
||||
URL string `json:"url"`
|
||||
CompanyName string `json:"company_name"`
|
||||
Location string `json:"candidate_required_location"`
|
||||
JobType string `json:"job_type"`
|
||||
} `json:"jobs"`
|
||||
}
|
||||
if err := getJSON(ctx, client, url, nil, &payload); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Job, 0, len(payload.Jobs))
|
||||
for _, j := range payload.Jobs {
|
||||
loc := j.Location
|
||||
if loc == "" {
|
||||
loc = j.JobType
|
||||
}
|
||||
out = append(out, Job{
|
||||
Title: j.Title,
|
||||
URL: j.URL,
|
||||
Company: j.CompanyName,
|
||||
Location: loc,
|
||||
Source: "remotive",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// safeHeaderValue rejects values containing CRLF or NUL that could split HTTP
|
||||
// headers. Go's net/http panics on these, but we prefer a clear error.
|
||||
func safeHeaderValue(v string) error {
|
||||
if strings.ContainsAny(v, "\r\n\x00") {
|
||||
return fmt.Errorf("header value contains invalid control characters")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// RemoteOK: array where first element is legal header — skip entries without
|
||||
// (id AND position|title). Bare fetch is rejected with 403 so we send a UA.
|
||||
func fetchRemoteOK(ctx context.Context, client *http.Client, url, userAgent string) ([]Job, error) {
|
||||
ua := userAgent
|
||||
if ua == "" {
|
||||
ua = "apex-scanner/1.0 (+https://github.com/cobr-ai/apex)"
|
||||
}
|
||||
if err := safeHeaderValue(ua); err != nil {
|
||||
return nil, fmt.Errorf("remoteok: user-agent: %w", err)
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("User-Agent", ua)
|
||||
req.Header.Set("Accept", "application/json")
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("HTTP %d", resp.StatusCode)
|
||||
}
|
||||
var entries []struct {
|
||||
ID interface{} `json:"id"` // string or int depending on the posting
|
||||
Position string `json:"position"`
|
||||
Title string `json:"title"`
|
||||
URL string `json:"url"`
|
||||
Slug string `json:"slug"`
|
||||
Company string `json:"company"`
|
||||
Location string `json:"location"`
|
||||
}
|
||||
if err := json.NewDecoder(io.LimitReader(resp.Body, maxResponseBytes)).Decode(&entries); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Job, 0, len(entries))
|
||||
for _, j := range entries {
|
||||
if j.ID == nil {
|
||||
continue
|
||||
}
|
||||
title := j.Position
|
||||
if title == "" {
|
||||
title = j.Title
|
||||
}
|
||||
if title == "" {
|
||||
continue
|
||||
}
|
||||
jobURL := j.URL
|
||||
if jobURL == "" && j.Slug != "" {
|
||||
jobURL = "https://remoteok.io/remote-jobs/" + j.Slug
|
||||
}
|
||||
loc := j.Location
|
||||
if loc == "" {
|
||||
loc = "Remote"
|
||||
}
|
||||
out = append(out, Job{
|
||||
Title: title,
|
||||
URL: jobURL,
|
||||
Company: j.Company,
|
||||
Location: loc,
|
||||
Source: "remoteok",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// USAJobs: paginated SearchResult.SearchResultItems[].MatchedObjectDescriptor.
|
||||
// Requires both User-Agent (registered email) and Authorization-Key headers.
|
||||
func fetchUSAJobs(ctx context.Context, client *http.Client, info AdapterInfo) ([]Job, error) {
|
||||
if info.APIKey == "" {
|
||||
return nil, fmt.Errorf("usajobs: api key missing (set aggregator.api_key or USAJOBS_API_KEY env)")
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, info.URL, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
ua := info.UserAgent
|
||||
if ua == "" {
|
||||
ua = "apex-scanner/1.0"
|
||||
}
|
||||
if err := safeHeaderValue(ua); err != nil {
|
||||
return nil, fmt.Errorf("usajobs: user-agent: %w", err)
|
||||
}
|
||||
if err := safeHeaderValue(info.APIKey); err != nil {
|
||||
return nil, fmt.Errorf("usajobs: api-key: %w", err)
|
||||
}
|
||||
req.Header.Set("User-Agent", ua)
|
||||
req.Header.Set("Authorization-Key", info.APIKey)
|
||||
req.Header.Set("Host", "data.usajobs.gov")
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("HTTP %d", resp.StatusCode)
|
||||
}
|
||||
var payload struct {
|
||||
SearchResult struct {
|
||||
SearchResultItems []struct {
|
||||
MatchedObjectDescriptor struct {
|
||||
PositionTitle string `json:"PositionTitle"`
|
||||
PositionURI string `json:"PositionURI"`
|
||||
ApplyURI []string `json:"ApplyURI"`
|
||||
OrganizationName string `json:"OrganizationName"`
|
||||
PositionLocationDisplay string `json:"PositionLocationDisplay"`
|
||||
} `json:"MatchedObjectDescriptor"`
|
||||
} `json:"SearchResultItems"`
|
||||
} `json:"SearchResult"`
|
||||
}
|
||||
if err := json.NewDecoder(io.LimitReader(resp.Body, maxResponseBytes)).Decode(&payload); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
items := payload.SearchResult.SearchResultItems
|
||||
out := make([]Job, 0, len(items))
|
||||
for _, it := range items {
|
||||
md := it.MatchedObjectDescriptor
|
||||
jobURL := ""
|
||||
if len(md.ApplyURI) > 0 {
|
||||
jobURL = md.ApplyURI[0]
|
||||
}
|
||||
if jobURL == "" {
|
||||
jobURL = md.PositionURI
|
||||
}
|
||||
company := md.OrganizationName
|
||||
if company == "" {
|
||||
company = "US Federal"
|
||||
}
|
||||
out = append(out, Job{
|
||||
Title: md.PositionTitle,
|
||||
URL: jobURL,
|
||||
Company: company,
|
||||
Location: md.PositionLocationDisplay,
|
||||
Source: "usajobs",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Generic RSS aggregator: mirrors parseRSSItems but tags Source=rss and
|
||||
// splices defaultCompany for feeds that don't embed a company per item.
|
||||
func fetchRSS(ctx context.Context, client *http.Client, feedURL, defaultCompany string) ([]Job, error) {
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, feedURL, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Accept", "application/rss+xml, application/xml, text/xml")
|
||||
req.Header.Set("User-Agent", "apex-scanner/1.0")
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
return nil, fmt.Errorf("HTTP %d", resp.StatusCode)
|
||||
}
|
||||
body, err := io.ReadAll(io.LimitReader(resp.Body, maxResponseBytes))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
type item struct {
|
||||
Title string `xml:"title"`
|
||||
Link string `xml:"link"`
|
||||
GUID string `xml:"guid"`
|
||||
Description string `xml:"description"`
|
||||
}
|
||||
type rss struct {
|
||||
Channel struct {
|
||||
Items []item `xml:"item"`
|
||||
} `xml:"channel"`
|
||||
}
|
||||
dec := newSafeXMLDecoder(bytes.NewReader(body))
|
||||
var feed rss
|
||||
if err := dec.Decode(&feed); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]Job, 0, len(feed.Channel.Items))
|
||||
for _, it := range feed.Channel.Items {
|
||||
link := strings.TrimSpace(it.Link)
|
||||
if link == "" {
|
||||
link = strings.TrimSpace(it.GUID)
|
||||
}
|
||||
title := strings.TrimSpace(it.Title)
|
||||
if title == "" || link == "" {
|
||||
continue
|
||||
}
|
||||
desc := strings.TrimSpace(it.Description)
|
||||
if len(desc) > 80 {
|
||||
desc = desc[:80]
|
||||
}
|
||||
out = append(out, Job{
|
||||
Title: title,
|
||||
URL: link,
|
||||
Company: defaultCompany,
|
||||
Location: desc,
|
||||
Source: "rss",
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// HTTPClient returns a timeout-limited client with safe defaults.
|
||||
func HTTPClient(timeout time.Duration) *http.Client {
|
||||
return &http.Client{Timeout: timeout}
|
||||
}
|
||||
Reference in New Issue
Block a user