"""Track your contacts' job changes: each LinkedIn profile checked live once a week through Datacircle, Up2Data at $1.25 per 1,000 found.
https://datacircle.dev/blog/track-linkedin-job-changes

DATACIRCLE_API_KEY=... python track-linkedin-job-changes.py contacts.csv                (contacts.csv has a linkedin_url column)
DATACIRCLE_API_KEY=... python track-linkedin-job-changes.py contacts.csv --harvestapi   (past Up2Data's daily limit, HarvestAPI)
Run it every day: it checks the contacts not checked in the last 7 days, until Up2Data's daily limit (800 a day per account).
checks.csv keeps every check, job_changes.csv every change found.
"""
import csv
import datetime
import os
import sys
import time

import requests

API = "https://api.datacircle.dev"
KEY = os.environ["DATACIRCLE_API_KEY"]
EVERY_DAYS = 7  # how often each contact is checked
NEW_JOB_DAYS = 90  # on a contact's first check, a job started this recently is a change already
CHECKS = ["linkedin_url", "checked_on", "status", "full_name", "company_id", "company", "title", "started_at", "cost_usd"]
CHANGES = ["checked_on", "linkedin_url", "full_name", "change", "old_company", "old_title", "new_company", "new_title", "new_company_url", "started_at"]
MONTHS = ["Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec"]
NO_JOB = {"company_id": "", "company": "", "title": "", "started_at": "", "company_url": ""}


def role(company_id, company, title, started_at, company_url):
    return {"company_id": company_id or "", "company": (company or "").strip(), "title": (title or "").strip(),
            "started_at": started_at or "", "company_url": company_url or ""}


def up2data_roles(profile):
    """The member's current jobs, the main one first."""
    main = profile.get("current_company") or {}
    jobs = [role(main.get("linkedin_id"), main.get("name"), main.get("title"), main.get("started_at"), main.get("url"))] if main else []
    return jobs + [role(job.get("linkedin_id"), job.get("company"), job.get("title"), job.get("started_at"), job.get("company_url"))
                   for job in profile.get("positions") or [] if not job.get("ended_at")]


def harvestapi_roles(person):
    def started(date):  # {"year": 2024, "month": "Jun"} as YYYY-MM, Up2Data's way
        date = date or {}
        return f"{date['year']}-{MONTHS.index(date['month']) + 1:02d}" if date.get("month") in MONTHS else str(date.get("year") or "")
    jobs = (person.get("currentPosition") or []) + [job for job in person.get("experience") or [] if (job.get("endDate") or {}).get("text") == "Present"]
    return [role(job.get("companyId"), job.get("companyName"), job.get("position"), started(job.get("startDate")), job.get("companyLinkedinUrl")) for job in jobs]


def call(provider, url):
    """One lookup. A provider failing (502, 503, 504) or Up2Data's own rate limit isn't charged: wait, then try twice more."""
    for wait in (0, 5, 15):
        time.sleep(wait)
        headers = {"Authorization": f"Token {KEY}", "X-Data-Provider": provider}
        try:
            if provider == "up2data":
                answer = requests.post(f"{API}/v1/profiles/enrich", headers=headers, json={"url": url}, timeout=90)
            else:
                answer = requests.get(f"{API}/linkedin/profile", headers=headers, params={"url": url}, timeout=90)
        except requests.RequestException as error:
            return 0, {"error": str(error)}  # no answer: the next run tries again
        try:
            body = answer.json()
        except ValueError:
            body = {}  # a gateway's error page: the status says enough
        rate_limited = answer.status_code == 429 and isinstance(body.get("error"), dict)  # Datacircle's daily limit is a string
        if not rate_limited and answer.status_code not in (502, 503, 504):
            break
    return answer.status_code, body


def check(url, provider):
    """One contact: (status, full name, current jobs, cost), "limit" past Up2Data's daily limit, or None to try again next run."""
    status, body = call(provider, url)
    if status == 401:
        sys.exit("401: the API key is wrong. It's on your dashboard.")
    if status == 402:
        sys.exit("402: your balance can't cover the call. Add funds on your dashboard, then run this again.")
    if status == 429 and isinstance(body.get("error"), str):
        return "limit"
    if status == 400:
        return "not a profile URL", "", [], 0  # free
    if provider == "up2data" and status == 422:
        return "not found", "", [], 0  # private or deleted: free
    if provider == "up2data" and status == 200:
        profile = body["data"]
        return "found", profile.get("full_name") or "", up2data_roles(profile), body["datacircle_meta"]["cost_usd"]
    if provider == "harvestapi" and status == 200:
        person = body.get("element")  # null when HarvestAPI can't find the profile, and the lookup is still billed
        if not person:
            return "not found (HarvestAPI)", "", [], body["datacircle_meta"]["cost_usd"]
        name = f"{person.get('firstName') or ''} {person.get('lastName') or ''}".strip()
        return "found (HarvestAPI)", name, harvestapi_roles(person), body["datacircle_meta"]["cost_usd"]
    reason = body.get("error", {}).get("message") if isinstance(body.get("error"), dict) else body.get("error", "")
    print(f"{url}: {status or 'no answer'} {reason}, not charged; the next run tries again", file=sys.stderr)
    return None


def compare(last, jobs, today):
    """(the change or None, the job to remember). `last` is the last check that found the profile."""
    main = jobs[0] if jobs else NO_JOB
    if last is None:  # first check
        since = (today - datetime.timedelta(days=NEW_JOB_DAYS)).strftime("%Y-%m")
        return ("new job" if len(main["started_at"]) == 7 and main["started_at"] >= since else None), main
    if not last["company_id"] and not last["company"]:  # no current job last time
        return ("new job" if jobs else None), main
    def same(job):
        return job["company_id"] == last["company_id"] if job["company_id"] and last["company_id"] else job["company"].lower() == last["company"].lower()
    kept = next((job for job in jobs if same(job)), None)
    if kept is None:
        return ("new company" if jobs else "left, no current job"), main
    return ("new title" if kept["title"] != last["title"] else None), kept


def open_csv(name, columns):
    file = open(name, "a", newline="", encoding="utf-8")
    writer = csv.DictWriter(file, columns)
    if file.tell() == 0:
        writer.writeheader()
    return file, writer


def main(source, harvestapi):
    today = datetime.datetime.now(datetime.timezone.utc).date()  # Up2Data's daily limit resets at 00:00 UTC
    with open(source, newline="", encoding="utf-8-sig") as file:  # utf-8-sig: Excel's byte order mark
        contacts = [{name.strip().lower(): (value or "").strip() for name, value in row.items() if name is not None} for row in csv.DictReader(file)]
    if contacts and "linkedin_url" not in contacts[0]:
        sys.exit(f"{source} has no linkedin_url column")
    urls = list(dict.fromkeys(contact["linkedin_url"] for contact in contacts if contact["linkedin_url"]))
    last_check, last_found = {}, {}
    if os.path.exists("checks.csv"):
        with open("checks.csv", newline="", encoding="utf-8") as file:
            for row in csv.DictReader(file):
                last_check[row["linkedin_url"]] = row
                if row["status"].startswith("found"):
                    last_found[row["linkedin_url"]] = row
    due = [url for url in urls if url not in last_check or (today - datetime.date.fromisoformat(last_check[url]["checked_on"])).days >= EVERY_DAYS]
    due.sort(key=lambda url: last_check[url]["checked_on"] if url in last_check else "")  # never checked first, then the oldest
    checks_file, checks = open_csv("checks.csv", CHECKS)
    changes_file, changes = open_csv("job_changes.csv", CHANGES)
    provider, checked, changed, spent = "up2data", 0, 0, 0.0
    for url in due:
        result = check(url, provider)
        if result == "limit" and harvestapi:
            provider = "harvestapi"
            result = check(url, provider)
        if result == "limit":
            print("Up2Data's daily limit: the next run, after 00:00 UTC, goes on (or add --harvestapi to go on now through HarvestAPI).")
            break
        if result is None:
            continue
        status, name, jobs, cost = result
        job, change = NO_JOB, None
        if status.startswith("found"):
            last = last_found.get(url)
            change, job = compare(last, jobs, today)
            if change:
                old = last or NO_JOB
                changes.writerow({"checked_on": today, "linkedin_url": url, "full_name": name, "change": change, "old_company": old["company"],
                                  "old_title": old["title"], "new_company": job["company"], "new_title": job["title"],
                                  "new_company_url": job["company_url"], "started_at": job["started_at"]})
                changes_file.flush()
                changed += 1
        checks.writerow({"linkedin_url": url, "checked_on": today, "status": status, "full_name": name, **{column: job[column] for column in CHECKS[4:8]},
                         "cost_usd": cost})
        checks_file.flush()  # a stop keeps every check written
        checked += 1
        spent += cost
    print(f"{checked} of {len(due)} due contacts checked ({len(urls)} in all), {changed} job changes in job_changes.csv, ${spent:.5f} spent")


if __name__ == "__main__":
    main(sys.argv[1], "--harvestapi" in sys.argv[2:])
