- get_scout_stats aggregator (app/engine/stats): assembles the REAL Summary metrics —
funnel + cohort engagement rank + active-window (activity timestamps by hour) from our DB,
accumulating salary band (avg of each deck's peak, ₹L), match/competition stats from the
feed, Momentum/QX + Q-Score trend (qscore-service), day streak (user-service). Honest:
unsourced cards return None so the UI omits/locks them, never faked. posted_date extractor.
- Activity tracking: viewed/saved flags + search_count → funnel (Matches→Viewed→Shortlisted→
Applied) + engagement percentile. matchesFound = all-time count.
- Per-board search cursors {board: page} (replaces the single cursor): only boards that truly
paginate (LinkedIn) get one; cursor = LAST page fetched (1st search of a new query → 1).
Resets on query change OR >24h (boards refresh ~daily). Dropped Naukri incremental/stateKey
(opaque, exhausting, cross-account dedup state) — dedup is the PER-USER seen-net only.
- tests: stats helpers (posting-age, histogram, active-window, engagement score).
227 lines
10 KiB
Python
227 lines
10 KiB
Python
"""
|
||
Per-board query builders — `ScoutPrefs` (dict) → each Apify actor's input.
|
||
See `docs/ENGINE_DESIGN.md` §1.2. Returns the `input` dict you POST to the actor's
|
||
`run-sync-get-dataset-items` endpoint.
|
||
|
||
System knobs (budget / freshness) are arguments with sane defaults, not user inputs.
|
||
"""
|
||
from __future__ import annotations
|
||
from typing import Any
|
||
|
||
from . import coerce as C
|
||
|
||
Prefs = dict[str, Any]
|
||
|
||
|
||
# ── Naukri (India workhorse — richest structured filters) ────────────────────
|
||
def build_naukri_input(prefs: Prefs, *, max_jobs: int = 80, freshness: str = "7", recall: bool = False) -> dict:
|
||
# recall=True (retrieve stage): keep keyword + city + work-mode only; drop the
|
||
# narrowing filters (experience/department/industry/salary/stage) — those become
|
||
# ENGINE ranking later, not hard board cuts. Wider freshness for recall.
|
||
if recall:
|
||
freshness = "30"
|
||
locs = C.parsed_locations(prefs)
|
||
cities: list[int] = []
|
||
for lp in locs:
|
||
if lp["city"]:
|
||
cities += C.CITY_NAUKRI.get(lp["city"], []) # unmapped city → dropped (kept in keyword)
|
||
|
||
out: dict[str, Any] = {
|
||
"keyword": C.title_of(prefs) or C.build_keyword(prefs),
|
||
"maxJobs": max_jobs, # REQUIRED — the budget
|
||
"freshness": freshness, # liveness gate
|
||
"sortBy": "date",
|
||
"fetchDetails": True, # → apply redirect / full details
|
||
}
|
||
if cities:
|
||
out["cities"] = [str(c) for c in sorted(set(cities))] # Naukri select codes are STRINGS
|
||
|
||
wm = [C.WORKMODE_NAUKRI[w] for w in (prefs.get("workMode") or []) if w in C.WORKMODE_NAUKRI]
|
||
if C.is_remote(prefs) and "remote" not in wm:
|
||
wm.append("remote")
|
||
if wm:
|
||
out["workMode"] = wm
|
||
|
||
if recall:
|
||
return out # retrieve stage stops here — keyword + cities + work-mode only
|
||
|
||
yrs = C.years_of(prefs)
|
||
if yrs is not None:
|
||
out["experience"] = str(yrs) # numeric-string enum "0".."30"
|
||
|
||
depts = [C.ROLE_NAUKRI_DEPT[r] for r in (prefs.get("role") or []) if r in C.ROLE_NAUKRI_DEPT]
|
||
if depts:
|
||
out["department"] = [str(d) for d in sorted(set(depts))] # codes → strings
|
||
|
||
inds = [C.INDUSTRY_NAUKRI[i] for i in (prefs.get("industry") or []) if i in C.INDUSTRY_NAUKRI]
|
||
if inds:
|
||
out["industry"] = [str(i) for i in sorted(set(inds))] # codes → strings
|
||
|
||
if prefs.get("targetComp") in C.COMP_NAUKRI_SALARY:
|
||
out["salaryRange"] = [C.COMP_NAUKRI_SALARY[prefs["targetComp"]]]
|
||
|
||
if prefs.get("companyStage") in C.COMPANY_TYPE_NAUKRI:
|
||
out["companyType"] = [C.COMPANY_TYPE_NAUKRI[prefs["companyStage"]]]
|
||
|
||
return out
|
||
|
||
|
||
# ── Naukri (blackfalcondata feed — city-precise, real offsite applyRedirectUrl) ──
|
||
def build_naukri_feed_input(prefs: Prefs, *, max_jobs: int = 50, fetch_details: bool = True) -> dict:
|
||
# City-filtered at the board (hyper-relevant + cheap). fetch_details=True surfaces the
|
||
# offsite `applyRedirectUrl`; skipReposts dedups WITHIN a run. Narrowing (dept/industry/exp)
|
||
# stays off here — that's engine ranking later.
|
||
# NOTE: no `incremental`/`stateKey` — that put opaque, exhausting, cross-account dedup state on
|
||
# Apify. Cross-deck dedup is the PER-USER seen-net in our DB. (The feed actor has no page/offset.)
|
||
city = next((lp["city"] for lp in C.parsed_locations(prefs) if lp["city"]), "")
|
||
return {
|
||
"keyword": C.title_of(prefs) or C.build_keyword(prefs),
|
||
"location": city, # precise city, e.g. "New Delhi"
|
||
"maxResults": max_jobs,
|
||
"fetchDetails": fetch_details,
|
||
"skipReposts": True,
|
||
"sortBy": "date",
|
||
}
|
||
|
||
|
||
# ── Foundit / Monster India (city-precise; apply_url is the offsite redirect) ──
|
||
# Foundit uses legacy city names — it returns 0 for "New Delhi"/"Bengaluru".
|
||
_FOUNDIT_CITY = {
|
||
"new delhi": "Delhi", "delhi ncr": "Delhi", "bengaluru": "Bangalore",
|
||
"gurugram": "Gurgaon", "mumbai": "Mumbai", "pune": "Pune", "hyderabad": "Hyderabad",
|
||
"chennai": "Chennai", "kolkata": "Kolkata", "noida": "Noida",
|
||
}
|
||
|
||
|
||
def build_foundit_input(prefs: Prefs, *, results_wanted: int = 50) -> dict:
|
||
city = next((lp["city"] for lp in C.parsed_locations(prefs) if lp["city"]), "")
|
||
city = _FOUNDIT_CITY.get(city.lower(), city) # alias to Foundit's canonical name
|
||
return {
|
||
"keyword": C.title_of(prefs) or C.build_keyword(prefs),
|
||
"location": city,
|
||
"results_wanted": results_wanted,
|
||
}
|
||
|
||
|
||
# ── LinkedIn (takes the band, not years; ₹ comp incompatible) ────────────────
|
||
def build_linkedin_input(prefs: Prefs, *, max_items: int = 80, posted: str = "week",
|
||
recall: bool = False, page: int = 1) -> dict:
|
||
# recall=True (retrieve stage): title + COUNTRY-level location only — drop the narrowing
|
||
# filters (easyApply / workplaceType / experienceLevel / industryIds / freshness), which
|
||
# become ENGINE ranking later. City-level location + a wrong industryId zero the actor out.
|
||
out: dict[str, Any] = {
|
||
"jobTitles": [C.title_of(prefs)],
|
||
"maxItems": max_items,
|
||
"sortBy": "date",
|
||
"page": page, # Start Page Number — advance per re-run for NEW jobs
|
||
}
|
||
|
||
cities: list[str] = [] # bare city names ("New Delhi") — harvestapi's location format
|
||
cities_q: list[str] = [] # city, country ("New Delhi, India") — used outside recall
|
||
countries: list[str] = []
|
||
for lp in C.parsed_locations(prefs):
|
||
if lp["country"] and lp["country"] not in countries:
|
||
countries.append(lp["country"])
|
||
if lp["city"]:
|
||
cities.append(lp["city"])
|
||
cities_q.append(f"{lp['city']}, {lp['country']}" if lp["country"] else lp["city"])
|
||
|
||
if recall:
|
||
# City-level for precision (engine ranks finer later). Fall back to country if no city.
|
||
out["locations"] = cities or countries
|
||
return out
|
||
|
||
out["postedLimit"] = posted # freshness
|
||
out["easyApply"] = False # favour offsite (auto-applyable) over login-gated easy-apply
|
||
if cities_q or countries:
|
||
out["locations"] = cities_q or countries
|
||
# TODO: geoIds = [lookup(loc)] is more reliable than free-text locations.
|
||
|
||
wm = [C.WORKMODE_LINKEDIN[w] for w in (prefs.get("workMode") or []) if w in C.WORKMODE_LINKEDIN]
|
||
if C.is_remote(prefs) and "remote" not in wm:
|
||
wm.append("remote")
|
||
if wm:
|
||
out["workplaceType"] = wm
|
||
|
||
band = C.seniority_band(prefs)
|
||
if band in C.SENIORITY_LINKEDIN:
|
||
out["experienceLevel"] = [C.SENIORITY_LINKEDIN[band]]
|
||
|
||
ind_ids = [C.INDUSTRY_LINKEDIN_ID[i] for i in (prefs.get("industry") or []) if i in C.INDUSTRY_LINKEDIN_ID]
|
||
if ind_ids:
|
||
out["industryIds"] = ind_ids # TODO verify IDs; unmapped industries already fall to keyword
|
||
|
||
return out
|
||
|
||
|
||
# ── Indeed (blunt: what + where + country only; everything else folds in) ────
|
||
def build_indeed_input(prefs: Prefs, *, max_items: int = 80) -> dict:
|
||
loc = next((l for l in C.parsed_locations(prefs) if l["city"] or l["country"]), {})
|
||
country_name = loc.get("country") or "India"
|
||
return {
|
||
# Indeed has no work-mode/seniority/industry params → fold them into the query.
|
||
"position": C.build_keyword(prefs, seniority=True, industry=True),
|
||
"location": loc.get("city") or "",
|
||
"country": C.COUNTRY_ISO2.get(country_name, "IN"),
|
||
"maxItemsPerSearch": max_items,
|
||
"followApplyRedirects": True, # → real offsite apply URL
|
||
"saveOnlyUniqueItems": True,
|
||
}
|
||
|
||
|
||
def build_indeed_searches(prefs: Prefs, *, max_items: int = 80) -> list[dict]:
|
||
"""Fan-out: Indeed `position`/`location` are single strings, so multi title×city
|
||
becomes N runs to merge+dedup downstream."""
|
||
titles = (prefs.get("targetTitles") or [])[:3] or [C.title_of(prefs)]
|
||
cities = [l["city"] for l in C.parsed_locations(prefs) if l["city"]] or [None]
|
||
base = build_indeed_input(prefs, max_items=max_items)
|
||
runs = []
|
||
for t in titles:
|
||
for city in cities:
|
||
r = dict(base)
|
||
r["position"] = " ".join(x for x in [t, *(["remote"] if C.is_remote(prefs) else [])] if x)
|
||
if city:
|
||
r["location"] = city
|
||
runs.append(r)
|
||
return runs
|
||
|
||
|
||
# ── Direct-ATS (curated-company lane — `companies` is passed in, NOT from prefs) ──
|
||
def build_ats_input(prefs: Prefs, *, companies: list[dict], recent_days: int = 30,
|
||
max_per_company: int = 50, output_profile: str = "compact",
|
||
recall: bool = False) -> dict:
|
||
loc = next((l for l in C.parsed_locations(prefs) if l["city"]), {})
|
||
inds = [i for i in (prefs.get("industry") or []) if i and i != "Any"]
|
||
out: dict[str, Any] = {
|
||
"companies": companies, # curated [{company}] — the lane's identity
|
||
"titleKeyword": C.title_of(prefs), # filters within those companies (role relevance)
|
||
"remoteOnly": C.is_remote(prefs),
|
||
"keywords": inds, # loose industry fold
|
||
"recentWindowDays": recent_days,
|
||
"maxJobsPerCompany": max_per_company,
|
||
"outputProfile": output_profile, # "compact" = cheaper (enough for display); "full" adds descriptions
|
||
}
|
||
# recall=True: skip the city filter — these curated companies are global/remote, so a
|
||
# city lock ("Bengaluru") empties the deck. Location becomes engine ranking later.
|
||
if not recall:
|
||
out["locationKeyword"] = loc.get("city", "")
|
||
return out
|
||
|
||
|
||
if __name__ == "__main__":
|
||
import json
|
||
demo: Prefs = {
|
||
"targetTitles": ["Payments PM"], "title": "Associate Product Manager",
|
||
"role": ["Product"], "industry": ["Fintech"],
|
||
"location": ["Bengaluru · India", "Delhi NCR · India"],
|
||
"workMode": ["Hybrid"], "experience": ["Junior"], "years": 2,
|
||
"targetComp": "₹16–22L", "companyStage": "Growth-stage",
|
||
# engine-side only (ignored by adapters): priorities/stretch/intent/availability/dealBreakers/sort
|
||
}
|
||
print("MVQ ok:", C.has_mvq(demo), "\n")
|
||
for name, fn in [("NAUKRI", build_naukri_input), ("LINKEDIN", build_linkedin_input),
|
||
("INDEED", build_indeed_input)]:
|
||
print(f"── {name} ──"); print(json.dumps(fn(demo), indent=2, ensure_ascii=False), "\n")
|
||
print("── DIRECT-ATS ──")
|
||
print(json.dumps(build_ats_input(demo, companies=[{"ats": "ashby", "company": "razorpay"}]), indent=2))
|