Anthropic OAuth LLM, entitlements (feature 1), renewals, + 7 enhancements

- LLM: Anthropic Sonnet 5 (effort=low) via OAuth Bearer (Claude Code keychain
  token, expiry-aware); NVIDIA/LM Studio endpoints kept commented for switch-back.
- Entitlements: scripts/import_entitlements.py joins ACL export to accounts by
  normalized parent name (324/342 matched; ID column Excel-corrupted), maps
  products->capabilities via data/product_capability_map.json. Raw ACL csv and
  entitlements.json are gitignored (confidential); accounts.json carries derived
  currentProducts + ownedCapabilityIds. Owned capabilities tag briefs EXPANSION.
- Renewals: local renewal-proximity score boost (window 365d, max +0.10) +
  'renewals' report; briefs/digests show RENEWAL WINDOW context.
- Batch runs: loop --limit/--rep with persistent cursor (strategic + stale-first).
- Digests: per-rep HTML export; timestamped digest_<date>_<time>.md kept forever.
- Account narrative memory fed into relevance/synthesis; synthesis maintains it.
- Signal quality gate drops PR fluff pre-LLM; SEC EDGAR filings connector.
- Citation verification: talking points must cite [S#] signals or are dropped.
- Brief filenames use account-name slug; whitespace report + installed-base view.
- USAGE.md: full command/flag/env reference.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-11 19:12:36 -04:00
parent 9131e83c08
commit c2854ba163
23 changed files with 22050 additions and 144 deletions

View File

@@ -0,0 +1,163 @@
#!/usr/bin/env python3
"""Load product entitlements (ACL export) into the account list — 100% locally.
CONFIDENTIAL-DATA HANDLING: the ACL CSV and the generated entitlements.json are
gitignored and are never sent to any API. All matching below is deterministic
local string work (no LLM).
NOTE: the FY26 Q3 export's 'Parent Account ID' column is Excel-corrupted
('3E+14' scientific notation), so accounts are joined on NORMALIZED PARENT NAME
(ACL 'Rpt Parent Name'/'Parent Account Name' vs accounts.json reference.parentName
/ accountName). Re-exporting without Excel would restore the clean ID join.
Outputs:
- data/entitlements.json (gitignored): per-account product detail + renewals
- data/accounts.json updated in place: currentProducts (readable strings)
+ ownedCapabilityIds (for local anti-signal checks)
- output/state/entitlement_unmapped.json: product families that matched no
capability pattern (extend data/product_capability_map.json from this)
Usage: python3 scripts/import_entitlements.py data/SE_ACL_FY26_Q3.csv
"""
from __future__ import annotations
import csv
import json
import re
import sys
from collections import defaultdict
from datetime import datetime
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
ACCOUNTS = ROOT / "data" / "accounts.json"
CAP_MAP = ROOT / "data" / "product_capability_map.json"
OUT_ENTITLEMENTS = ROOT / "data" / "entitlements.json"
OUT_UNMAPPED = ROOT / "output" / "state" / "entitlement_unmapped.json"
def norm_name(name: str) -> str:
"""Normalize a company/parent name for joining across systems."""
n = name.upper()
n = re.sub(r"\(GP\)", "", n)
n = re.sub(r"[^A-Z0-9 ]", "", n) # drops hyphens: FED-EX -> FEDEX
n = re.sub(
r"\b(INC|CORP|CORPORATION|CO|COMPANY|COMPANIES|HOLDINGS?|GROUP|PLC|LTD|LLC|THE)\b",
"", n)
return re.sub(r"\s+", " ", n).strip()
def load_cap_patterns() -> list[tuple[re.Pattern, str]]:
doc = json.loads(CAP_MAP.read_text())
return [(re.compile(p["match"], re.IGNORECASE), p["capability"]) for p in doc["patterns"]]
def parse_date(us_date: str) -> str:
"""'7/28/26' -> '2026-07-28' (blank-safe)."""
try:
return datetime.strptime(us_date.strip(), "%m/%d/%y").strftime("%Y-%m-%d")
except (ValueError, AttributeError):
return ""
def main(csv_path: str) -> None:
patterns = load_cap_patterns()
accounts_doc = json.loads(ACCOUNTS.read_text())
accounts = accounts_doc["accounts"]
# Index accounts by normalized parent name AND normalized account name.
by_norm: dict[str, dict] = {}
for a in accounts:
for candidate in (a.get("reference", {}).get("parentName", ""), a["accountName"]):
key = norm_name(candidate)
if key and key not in by_norm:
by_norm[key] = a
# Aggregate active entitlement lines per normalized parent name.
per_parent: dict[str, dict] = defaultdict(lambda: {
"parentNames": set(), "products": {}, "lines": 0,
})
total_lines = active_lines = 0
for row in csv.DictReader(open(csv_path, newline="", encoding="utf-8-sig")):
total_lines += 1
if (row.get("ACL Status") or "").strip().lower() != "active":
continue
active_lines += 1
parent = (row.get("Rpt Parent Name") or row.get("Parent Account Name") or "").strip()
key = norm_name(parent)
if not key:
continue
agg = per_parent[key]
agg["parentNames"].add(parent)
agg["lines"] += 1
prod_key = " / ".join(x for x in (
(row.get("Product BU") or "").strip(),
(row.get("Product Family") or "").strip(),
(row.get("Product Subfamily") or "").strip(),
) if x)
p = agg["products"].setdefault(prod_key, {
"bu": (row.get("Product BU") or "").strip(),
"family": (row.get("Product Family") or "").strip(),
"subfamily": (row.get("Product Subfamily") or "").strip(),
"lines": 0, "licenseTypes": set(), "nextRenewal": "",
})
p["lines"] += 1
p["licenseTypes"].add((row.get("License Type Calculated") or "").strip())
rd = parse_date(row.get("Renewal Date", ""))
if rd and (not p["nextRenewal"] or rd < p["nextRenewal"]):
p["nextRenewal"] = rd
# Map products -> capabilities and join to accounts.
entitlements: dict[str, dict] = {}
unmapped: dict[str, int] = defaultdict(int)
matched_accounts = 0
for key, agg in per_parent.items():
acct = by_norm.get(key)
if not acct:
continue
matched_accounts += 1
owned_caps: set[str] = set()
products = []
for prod_key, p in sorted(agg["products"].items()):
probe = f"{p['bu']} {p['family']} {p['subfamily']}"
caps = sorted({cap for rx, cap in patterns if rx.search(probe)})
if not caps:
unmapped[prod_key] += p["lines"]
owned_caps.update(caps)
products.append({
"product": prod_key, "lines": p["lines"],
"licenseTypes": sorted(t for t in p["licenseTypes"] if t),
"nextRenewal": p["nextRenewal"], "capabilities": caps,
})
entitlements[acct["accountId"]] = {
"accountName": acct["accountName"],
"matchedOn": sorted(agg["parentNames"]),
"activeLines": agg["lines"],
"products": products,
"ownedCapabilityIds": sorted(owned_caps),
}
# Update the account record: readable strings for LLM prompts,
# capability ids for local anti-signal checks.
acct["currentProducts"] = sorted({
f"{p['family']}" + (f" ({p['subfamily']})" if p["subfamily"] and p["subfamily"] != p["family"] else "")
for p in agg["products"].values()
})
acct["ownedCapabilityIds"] = sorted(owned_caps)
OUT_ENTITLEMENTS.write_text(json.dumps(entitlements, indent=2))
ACCOUNTS.write_text(json.dumps(accounts_doc, indent=2))
OUT_UNMAPPED.parent.mkdir(parents=True, exist_ok=True)
OUT_UNMAPPED.write_text(json.dumps(
dict(sorted(unmapped.items(), key=lambda kv: -kv[1])), indent=2))
print(f"ACL lines read: {total_lines}")
print(f" active: {active_lines}")
print(f" distinct parents in file: {len(per_parent)}")
print(f"Accounts matched: {matched_accounts}/{len(accounts)}")
print(f" with owned capabilities: {sum(1 for e in entitlements.values() if e['ownedCapabilityIds'])}")
print(f"Unmapped product families: {len(unmapped)} -> {OUT_UNMAPPED.relative_to(ROOT)}")
print(f"Wrote {OUT_ENTITLEMENTS.relative_to(ROOT)} (gitignored) and updated data/accounts.json")
if __name__ == "__main__":
main(sys.argv[1] if len(sys.argv) > 1 else str(ROOT / "data" / "SE_ACL_FY26_Q3.csv"))