"""Keep your HubSpot contacts' job title, company and location as their LinkedIn profiles show them today, and flag who changed jobs.
https://datacircle.dev/blog/enrich-hubspot-contacts-linkedin

HUBSPOT_TOKEN=... DATACIRCLE_API_KEY=... python enrich-hubspot-contacts-linkedin.py
HUBSPOT_TOKEN=... DATACIRCLE_API_KEY=... python enrich-hubspot-contacts-linkedin.py --harvestapi   (past Up2Data's daily limit, HarvestAPI)
Run it every day: it checks each contact with a LinkedIn URL once every 7 days, never-checked first, until Up2Data's daily limit (800 a day
per account), through Datacircle at $1.25 per 1,000 profiles found. HUBSPOT_TOKEN is a HubSpot private app's access token with the scopes
crm.objects.contacts.read, crm.objects.contacts.write, crm.schemas.contacts.read and crm.schemas.contacts.write.
"""
import datetime
import os
import re
import sys
import time

import requests

API = "https://api.datacircle.dev"
KEY = os.environ["DATACIRCLE_API_KEY"]
HUBSPOT = "https://api.hubapi.com"
HUBSPOT_HEADERS = {"Authorization": f"Bearer {os.environ['HUBSPOT_TOKEN']}"}
VERSION = "2026-09"  # HubSpot's API version
EVERY_DAYS = 7  # how often each contact is checked
OVERWRITE = ["jobtitle", "city", "state", "country"]  # written from LinkedIn; remove one to keep yours (the company is always written)
MAX_PER_RUN = 10000  # HubSpot's search gives at most 10,000 results
NEW_PROPERTIES = [  # added to your contacts on the first run
    {"name": "linkedin_checked_on", "label": "LinkedIn checked on", "type": "date", "fieldType": "date"},
    {"name": "linkedin_job_change_on", "label": "LinkedIn job change on", "type": "date", "fieldType": "date"},
    {"name": "linkedin_previous_company", "label": "LinkedIn previous company", "type": "string", "fieldType": "text"},
]
READ = ["hs_linkedin_url", "jobtitle", "company", "linkedin_checked_on"]
SUFFIXES = {"inc", "llc", "ltd", "limited", "corp", "corporation", "co", "company", "gmbh", "plc", "sa", "ag", "bv", "lp", "llp"}
TODAY = datetime.datetime.now(datetime.timezone.utc).date()  # Up2Data's daily limit resets at 00:00 UTC


def hubspot(method, path, body=None):
    """One HubSpot call. Its rate limits (429) and its own failures (5xx) are tried again after 10 s, then 30 s."""
    for wait in (0, 10, 30):
        time.sleep(wait)
        answer = requests.request(method, f"{HUBSPOT}{path}", headers=HUBSPOT_HEADERS, json=body, timeout=60)
        if answer.status_code != 429 and answer.status_code < 500:
            break
    try:
        result = answer.json()
    except ValueError:
        result = {}
    if answer.status_code in (401, 403):
        sys.exit(f"HubSpot {answer.status_code}: {result.get('message')} Check HUBSPOT_TOKEN and its scopes (top of this file).")
    return answer.status_code, result


def add_properties():
    for prop in NEW_PROPERTIES:
        status, _ = hubspot("GET", f"/crm/properties/{VERSION}/contacts/{prop['name']}")
        if status == 404:
            status, body = hubspot("POST", f"/crm/properties/{VERSION}/contacts", {**prop, "groupName": "contactinformation"})
            if status not in (200, 201):
                sys.exit(f"HubSpot {status} adding the property {prop['name']}: {body.get('message')}")


def due_contacts():
    """The contacts with a LinkedIn URL never checked, then those last checked EVERY_DAYS ago or more, the oldest first."""
    cutoff = datetime.datetime.combine(TODAY - datetime.timedelta(days=EVERY_DAYS - 1), datetime.time(), datetime.timezone.utc)
    has_url = {"propertyName": "hs_linkedin_url", "operator": "HAS_PROPERTY"}
    searches = [[has_url, {"propertyName": "linkedin_checked_on", "operator": "NOT_HAS_PROPERTY"}],
                [has_url, {"propertyName": "linkedin_checked_on", "operator": "LT", "value": str(int(cutoff.timestamp() * 1000))}]]
    contacts = []
    for filters in searches:
        after = None
        while len(contacts) < MAX_PER_RUN:
            body = {"filterGroups": [{"filters": filters}], "properties": READ, "limit": 200,
                    "sorts": [{"propertyName": "linkedin_checked_on", "direction": "ASCENDING"}]}
            if after:
                body["after"] = after
            status, page = hubspot("POST", f"/crm/objects/{VERSION}/contacts/search", body)
            if status != 200:
                sys.exit(f"HubSpot {status} searching contacts: {page.get('message')}")
            contacts += page["results"]
            after = page.get("paging", {}).get("next", {}).get("after")
            if not after:
                break
            time.sleep(0.25)  # HubSpot's search takes 5 requests a second
    return contacts[:MAX_PER_RUN]


def text(value):
    return (value or "").strip()  # HarvestAPI's names and titles can end with a space


def place(city, state, country):
    return {"city": text(city), "state": text(state), "country": text(country)}


def call(provider, url):
    """One lookup. A provider failing (502, 503, 504) or Up2Data's own rate limit isn't charged: wait, then try twice more."""
    for wait in (0, 5, 15):
        time.sleep(wait)
        headers = {"Authorization": f"Token {KEY}", "X-Data-Provider": provider}
        try:
            if provider == "up2data":
                answer = requests.post(f"{API}/v1/profiles/enrich", headers=headers, json={"url": url}, timeout=90)
            else:
                answer = requests.get(f"{API}/linkedin/profile", headers=headers, params={"url": url}, timeout=90)
        except requests.RequestException as error:
            return 0, {"error": str(error)}  # no answer: the next run tries again
        try:
            body = answer.json()
        except ValueError:
            body = {}  # a gateway's error page: the status says enough
        rate_limited = answer.status_code == 429 and isinstance(body.get("error"), dict)  # Datacircle's daily limit is a string
        if not rate_limited and answer.status_code not in (502, 503, 504):
            break
    return answer.status_code, body


def check(url, provider):
    """One profile: (current jobs [(company, title)], location {city, state, country}, cost), "limit", "no profile", or None to try next run."""
    status, body = call(provider, url)
    if status == 401:
        sys.exit("401: the Datacircle API key is wrong. It's on your dashboard.")
    if status == 402:
        sys.exit("402: your balance can't cover the call. Add funds on your dashboard, then run this again.")
    if status == 429 and isinstance(body.get("error"), str):
        return "limit"
    if status == 400 or (provider == "up2data" and status == 422):
        return "no profile"  # not a profile URL, or a private or deleted profile: free
    if provider == "up2data" and status == 200:
        profile, cost = body["data"], body["datacircle_meta"]["cost_usd"]
        main, where = profile.get("current_company") or {}, profile.get("location") or {}
        jobs = [(text(main.get("name")), text(main.get("title")))] if main else []
        jobs += [(text(job.get("company")), text(job.get("title"))) for job in profile.get("positions") or [] if not job.get("ended_at")]
        return jobs, place(where.get("city"), where.get("region"), where.get("country")), cost
    if provider == "harvestapi" and status == 200:
        person, cost = body.get("element"), body["datacircle_meta"]["cost_usd"]  # null when HarvestAPI can't find the profile, still billed
        if not person:
            return [], None, cost
        where = (person.get("location") or {}).get("parsed") or {}
        held = [job for job in person.get("experience") or [] if (job.get("endDate") or {}).get("text") == "Present"]
        jobs = [(text(job.get("companyName")), text(job.get("position"))) for job in (person.get("currentPosition") or []) + held]
        return jobs, place(where.get("city"), where.get("state"), where.get("country")), cost
    reason = body.get("error", {}).get("message") if isinstance(body.get("error"), dict) else body.get("error", "")
    print(f"{url}: {status or 'no answer'} {reason}, not charged; the next run tries again", file=sys.stderr)
    return None


def plain(company):
    """A company's name to compare: "Acme, Inc." and "ACME" are the same company."""
    words = re.sub(r"[^a-z0-9]+", " ", company.lower().replace("&", " and ")).split()
    while words and words[-1] in SUFFIXES:
        words.pop()
    return " ".join(words)


def changes(contact, jobs, location):
    """The properties to write: LinkedIn's company, title and location, and the job change when the company in HubSpot is gone."""
    old = contact["properties"]
    company = (old.get("company") or "").strip()
    props = {"linkedin_checked_on": TODAY.isoformat()}
    if location is None:  # no profile found: only the date of the check
        return props, False
    kept = next((job for job in jobs if company and plain(job[0]) == plain(company)), None)  # the job HubSpot already knows, if it's still held
    props["company"], props["jobtitle"] = kept or (jobs[0] if jobs else ("", ""))
    props.update({name: value for name, value in location.items() if value})  # a place LinkedIn doesn't give leaves yours
    for name in {"jobtitle", "city", "state", "country"} - set(OVERWRITE):
        props.pop(name, None)
    moved = bool(company and not kept) or bool(old.get("linkedin_checked_on") and not company and jobs)  # a new company, gone, or a job after none
    if moved:
        props["linkedin_job_change_on"], props["linkedin_previous_company"] = TODAY.isoformat(), company
    return props, moved


def save(updates):
    for start in range(0, len(updates), 100):  # HubSpot updates 100 contacts a call
        status, body = hubspot("POST", f"/crm/objects/{VERSION}/contacts/batch/update", {"inputs": updates[start:start + 100]})
        if status == 207:
            for error in body.get("errors", []):
                print(f"HubSpot: {error.get('message')}", file=sys.stderr)
        elif status != 200:
            print(f"HubSpot {status}: {body.get('message')}; these contacts are checked again next run", file=sys.stderr)
    updates.clear()


def main(harvestapi):
    add_properties()
    contacts = due_contacts()
    provider, found, checked, moved, spent, pending = "up2data", {}, 0, 0, 0.0, []
    try:
        for contact in contacts:
            url = contact["properties"]["hs_linkedin_url"].strip()
            if url not in found:  # the same profile on two contacts is looked up once
                result = check(url, provider)
                if result == "limit" and harvestapi:
                    provider = "harvestapi"
                    result = check(url, provider)
                if result == "limit":
                    print("Up2Data's daily limit: the next run, after 00:00 UTC, goes on (or add --harvestapi to go on now through HarvestAPI).")
                    break
                if result is None:
                    continue
                found[url] = ([], None, 0) if result == "no profile" else result
                spent += found[url][2]
            jobs, location, _ = found[url]
            props, change = changes(contact, jobs, location)
            pending.append({"id": contact["id"], "properties": props})
            checked += 1
            moved += change
            if len(pending) == 100:
                save(pending)
    finally:
        save(pending)  # a stop keeps every check made
    print(f"{checked} of {len(contacts)} due contacts checked, {moved} job changes (LinkedIn job change on: {TODAY}), ${spent:.5f} spent")


if __name__ == "__main__":
    main("--harvestapi" in sys.argv[1:])
