From 98dca237f4d635b9cb91edccfa31d8a9e3acf546 Mon Sep 17 00:00:00 2001 From: Adnan Zahir Date: Tue, 8 Sep 2026 16:11:36 +0700 Subject: [PATCH] Add infra-ops-toolkit plugin: topology diagram, html-to-pdf, teleport onboarding skills Distills patterns from the Teleport access topology work: verified-data diagram building, as-displayed HTML-to-PDF export via headless Chromium, and container-scoped/host/database onboarding into an existing Teleport cluster. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01R3ZTfgQrkR3q8DSEoZAmvs --- .claude-plugin/marketplace.json | 6 + marketplace.json | 12 ++ .../.claude-plugin/plugin.json | 17 +++ .../.codex-plugin/plugin.json | 6 + plugins/infra-ops-toolkit/README.md | 11 ++ .../skills/html-to-pdf/SKILL.md | 75 +++++++++ .../skills/teleport-onboard/SKILL.md | 144 ++++++++++++++++++ .../skills/topology-diagram/SKILL.md | 56 +++++++ 8 files changed, 327 insertions(+) create mode 100644 plugins/infra-ops-toolkit/.claude-plugin/plugin.json create mode 100644 plugins/infra-ops-toolkit/.codex-plugin/plugin.json create mode 100644 plugins/infra-ops-toolkit/README.md create mode 100644 plugins/infra-ops-toolkit/skills/html-to-pdf/SKILL.md create mode 100644 plugins/infra-ops-toolkit/skills/teleport-onboard/SKILL.md create mode 100644 plugins/infra-ops-toolkit/skills/topology-diagram/SKILL.md diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 9f5bdf5..a33644c 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -19,6 +19,12 @@ "source": "./plugins/shared-vault-sync", "description": "Safely pulls the shared vault at session start.", "version": "0.1.2" + }, + { + "name": "infra-ops-toolkit", + "source": "./plugins/infra-ops-toolkit", + "description": "Topology diagram building, as-is HTML→PDF export, and Teleport resource onboarding skills.", + "version": "0.1.0" } ] } diff --git a/marketplace.json b/marketplace.json index 14bbde8..b14aa02 100644 --- a/marketplace.json +++ b/marketplace.json @@ -39,6 +39,18 @@ "authentication": "ON_INSTALL" }, "category": "Productivity" + }, + { + "name": "infra-ops-toolkit", + "source": { + "source": "local", + "path": "./plugins/infra-ops-toolkit" + }, + "policy": { + "installation": "AVAILABLE", + "authentication": "ON_INSTALL" + }, + "category": "DevOps" } ] } diff --git a/plugins/infra-ops-toolkit/.claude-plugin/plugin.json b/plugins/infra-ops-toolkit/.claude-plugin/plugin.json new file mode 100644 index 0000000..bd65ea0 --- /dev/null +++ b/plugins/infra-ops-toolkit/.claude-plugin/plugin.json @@ -0,0 +1,17 @@ +{ + "name": "infra-ops-toolkit", + "version": "0.1.0", + "description": "Skills for building the infra access topology diagram, exporting any HTML page to PDF as-displayed, and onboarding new resources into Teleport.", + "author": { "name": "Senkensha" }, + "interface": { + "displayName": "Infra Ops Toolkit", + "shortDescription": "Topology diagram, as-is HTML→PDF export, and Teleport onboarding skills.", + "longDescription": "Three skills distilled from real MBU Group infrastructure work: (1) topology-diagram — build/update the light-theme HTML+SVG access topology and its drawio mirror from live-verified doctl/aws/tsh data; (2) html-to-pdf — export any HTML page to a PDF that matches the live browser view exactly, using headless Chromium in screen mode with a content-sized page instead of the browser's fixed-paper print dialog; (3) teleport-onboard — onboard a new VPS, EC2 host, container-scoped app login, or database into a Teleport cluster following the team's established least-privilege patterns.", + "developerName": "Senkensha", + "category": "DevOps", + "capabilities": [ + "Skills" + ], + "defaultPrompt": "Help me update the infra topology diagram." + } +} diff --git a/plugins/infra-ops-toolkit/.codex-plugin/plugin.json b/plugins/infra-ops-toolkit/.codex-plugin/plugin.json new file mode 100644 index 0000000..97d3e8e --- /dev/null +++ b/plugins/infra-ops-toolkit/.codex-plugin/plugin.json @@ -0,0 +1,6 @@ +{ + "name": "infra-ops-toolkit", + "version": "0.1.0+codex.20260908000000", + "description": "Skills for building the infra access topology diagram, exporting any HTML page to PDF as-displayed, and onboarding new resources into Teleport.", + "author": { "name": "Senkensha" } +} diff --git a/plugins/infra-ops-toolkit/README.md b/plugins/infra-ops-toolkit/README.md new file mode 100644 index 0000000..20d7427 --- /dev/null +++ b/plugins/infra-ops-toolkit/README.md @@ -0,0 +1,11 @@ +# infra-ops-toolkit + +Three skills distilled from real infrastructure work, for anyone building infra documentation or onboarding resources into Teleport: + +- **topology-diagram** — build/update a light-theme HTML+SVG access topology diagram (mirrored into a `.drawio.xml`), always re-verifying every IP/spec/capacity against live sources (`doctl`, `aws`, `tsh`) instead of trusting the diagram's own prior state. +- **html-to-pdf** — export any HTML page to a PDF that matches its live browser rendering exactly, using headless Chromium in screen mode with a content-sized page, instead of fighting a browser's fixed-paper Print dialog. +- **teleport-onboard** — onboard a new VPS, EC2 host, container-scoped app login, or database into a Teleport cluster by discovering and mirroring the team's existing patterns (forced-login-shell wrapper scripts, scoped sudoers, `tctl`-created roles) rather than inventing a new access model. + +## Install + +Install `infra-ops-toolkit` from the `infra-plugins` marketplace. Skills activate automatically when a prompt matches their description, or invoke directly: `/topology-diagram`, `/html-to-pdf`, `/teleport-onboard`. diff --git a/plugins/infra-ops-toolkit/skills/html-to-pdf/SKILL.md b/plugins/infra-ops-toolkit/skills/html-to-pdf/SKILL.md new file mode 100644 index 0000000..f4df226 --- /dev/null +++ b/plugins/infra-ops-toolkit/skills/html-to-pdf/SKILL.md @@ -0,0 +1,75 @@ +--- +name: html-to-pdf +description: Export a local HTML file to a PDF that looks exactly like the live browser rendering — no shrink-to-fit distortion, no forced page splits, no vanished sticky/grid elements. Use when the user says a browser's own Print-to-PDF/Print dialog produced a broken, tiny, or wrongly-paginated result and they want the PDF "as-is / seperti yang tampil di browser" instead of a print-reflowed version. +--- + +# HTML → PDF, exactly as displayed + +## Why the browser's own Print dialog fails for this + +A system print dialog always targets a **fixed paper size** (Letter/A4). For any page wider than ~800px of real content — a multi-column CSS Grid layout, a wide diagram — the browser either shrinks the *entire* page to fit one sheet (readable content becomes a postage stamp) or splits it across pages in ways that don't match what's on screen. Two extra failure modes compound this: +- `position: sticky` elements often just vanish or mis-render under the print media type. +- If the page has its own `@media print` rules (e.g. for a *deliberately* paginated print layout), those override the normal on-screen styling — which is the opposite of what "as-is" means here. + +Neither "no print CSS at all" nor "add print CSS to force page-breaks" gives you the *live browser view* as a PDF. The actual fix is to render with a **page size equal to the content's own dimensions**, which no manual print dialog lets you set. + +## The fix: headless Chromium in screen mode, content-sized page + +```bash +mkdir -p /tmp/pdfgen && cd /tmp/pdfgen +npm init -y >/dev/null 2>&1 +npm install puppeteer # bundles a compatible Chromium — no separate browser install needed +``` + +```js +// render.js +const puppeteer = require('puppeteer'); +const path = require('path'); + +(async () => { + const srcPath = '/absolute/path/to/source.html'; + const outPath = '/absolute/path/to/output.pdf'; + + const browser = await puppeteer.launch(); + const page = await browser.newPage(); + + // Match the viewport width the page's CSS was actually designed around + // (check its max-width / .page container, not an arbitrary guess). + await page.setViewport({ width: 1940, height: 1200, deviceScaleFactor: 2 }); + await page.goto('file://' + srcPath, { waitUntil: 'networkidle0' }); + + // The critical line: force normal on-screen styles, ignoring any + // @media print rules the page might define for its own purposes. + await page.emulateMediaType('screen'); + + // Measure the real rendered size so the PDF page is exactly that size — + // one continuous page, nothing scaled, nothing split. + const { width, height } = await page.evaluate(() => { + const el = document.querySelector('.page') || document.body; + const rect = el.getBoundingClientRect(); + return { width: Math.ceil(rect.width), height: Math.ceil(document.documentElement.scrollHeight) }; + }); + + await page.pdf({ + path: outPath, + width: `${width}px`, + height: `${height}px`, + printBackground: true, // otherwise CSS background colors silently disappear + margin: { top: 0, right: 0, bottom: 0, left: 0 }, + }); + + await browser.close(); + console.log('Saved:', outPath, width, 'x', height); +})().catch(err => { console.error(err); process.exit(1); }); +``` + +```bash +node render.js +``` + +## Notes + +- `printBackground: true` is easy to forget and its absence is easy to misdiagnose as "the PDF looks washed out" rather than "backgrounds are just missing." +- If the page measures its own scroll height *before* web fonts finish loading, dimensions can be slightly off — `waitUntil: 'networkidle0'` on `goto` covers most cases; add an explicit `page.evaluateHandle('document.fonts.ready')` await first if fonts still look unloaded in the output. +- This same technique works for any "I want the PDF to look like the page, not like a printout" request — it's not specific to any one diagram or document. +- Verify the result before calling it done: read the generated PDF back (most tools can render/preview a PDF's first page) and compare it against the live browser view side by side, don't just trust that `page.pdf()` succeeded without error. diff --git a/plugins/infra-ops-toolkit/skills/teleport-onboard/SKILL.md b/plugins/infra-ops-toolkit/skills/teleport-onboard/SKILL.md new file mode 100644 index 0000000..83becc7 --- /dev/null +++ b/plugins/infra-ops-toolkit/skills/teleport-onboard/SKILL.md @@ -0,0 +1,144 @@ +--- +name: teleport-onboard +description: Onboard a new resource into a Teleport cluster — a VPS/droplet, an EC2 (or any fresh VM) host, a container-scoped least-privilege app login on a multi-app host, or a database (db_service) — following an existing team's established patterns rather than inventing a new access model. Use when asked to register, join, or onboard something into Teleport, add scoped/restricted access to a specific app or container, or when Teleport resources (nodes/databases/roles) don't match what's actually running. +--- + +# Teleport resource onboarding + +## First: discover the existing pattern, don't assume one + +Before creating anything, SSH into a host that's already onboarded the way you intend to onboard the new one, and read what's actually there: +```bash +cat /etc/passwd # look for non-standard login shells — a sign of scoped access +sudo cat /etc/sudoers.d/ # the exact commands that login is allowed to run +sudo cat /etc/teleport.yaml # or find it: /opt/teleport/config/, /etc/teleport/ — it varies +ps aux | grep teleport # more than one teleport process on a host is a real bug (see gotcha below) +``` +Do not invent a new access model if one already exists — mirror it exactly, including its exact `sudoers` syntax, its wrapper-script conventions, and its Teleport role shape. + +**Where `tctl` actually runs**: the cluster's auth server is often just one specific host's Docker container (look for an image like `teleport-distroless` in `docker ps`), not a separate admin machine. `sudo docker exec teleport tctl ...` gives full cluster admin without a separate tctl binary or identity file — check for this before assuming you need new credentials. + +## Pattern A — container-scoped forced-login-shell (multi-app host, least privilege per app) + +Use when one host runs several unrelated apps as containers and different people need access to *only their own* app's container, never the host shell or a sibling's container. + +1. **Wrapper script** `/usr/local/bin/enter-.sh` (root:root, `0755`): + ```bash + #!/bin/bash + set -euo pipefail + # Forced login shell — must never fall back to a real host shell. + if [ ! -t 0 ]; then echo "Interactive TTY required." >&2; exit 1; fi + echo "=== - container access ===" + echo "1) " + echo "2) " + read -rp "Pilih [1-2]: " choice + case "$choice" in + 1) exec sudo /usr/bin/docker exec -it sh ;; + 2) exec sudo /usr/bin/docker exec -it sh ;; + *) echo "Pilihan tidak valid."; exit 1 ;; + esac + ``` + One numbered menu entry per container this login may reach. Never a generic `docker exec -it "$1"` — the container name must be hard-coded per case, or `sudoers` command-matching below becomes meaningless. + +2. **OS user**: `useradd -m -s /usr/local/bin/enter-.sh ` — mind the **32-character Linux username limit** (`useradd: invalid user name` is exactly that error); shorten rather than truncate blindly so the name stays meaningful. + +3. **Sudoers**, one line per allowed container, no wildcards: + ``` + ALL=(root) NOPASSWD: /usr/bin/docker exec -it sh + ALL=(root) NOPASSWD: /usr/bin/docker exec -it sh + ``` + Write to `/etc/sudoers.d/`, `chmod 440`, and **always** `visudo -c -f ` before trusting it — a syntax error here can be silent until someone tries to log in, or can break sudo more broadly if written wrong. + +4. **Teleport role** (via `tctl create -f`, from wherever the auth server actually runs): + ```yaml + kind: role + version: v7 + metadata: + name: access- + spec: + allow: + logins: [""] + node_labels: {"*": "*"} # or {name: [""]} to scope to one specific host + ``` + Check how existing analogous roles are assigned (`tctl get users --format=json`) — new roles usually get created *unassigned*, with assignment to a real person happening later as a separate, explicit step. Don't assume you should assign it to anyone. + +5. **Verify** — don't just trust that the files look right: + ```bash + sudo -l -U # must show exactly the intended docker exec commands, nothing else + ``` + Also spot-check that `sh` actually exists in each target container image (`docker exec sh -c 'echo ok'`) — a container built on a shell-less base image will make the wrapper script fail at the worst time. + +## Pattern B — single-app host (simpler, no wrapper needed) + +When a host only runs one app's containers, there's nothing to scope *within* the host — a plain restricted login is enough: +```bash +useradd -m -s /bin/bash -G docker,adm # docker group for the app's containers, adm for logs +``` +No sudoers file, no wrapper script. Confirm this is really the pattern in use (Pattern A hosts and Pattern B hosts can coexist across a fleet) before picking one over the other. + +## Pattern C — onboarding a brand-new host (EC2 or any fresh VM) + +1. Bootstrap plain SSH access first — the host isn't in Teleport yet, so Teleport can't get you in. Try available keys, confirm passwordless sudo: + ```bash + ssh -i ~/.ssh/ @ "sudo -n true && echo ok" + ``` +2. Install the Teleport agent at the **same major.minor version as the cluster** (check with `tctl status` first): + ```bash + curl https://apt.releases.teleport.dev/gpg -o /tmp/teleport-pubkey.asc + sudo tee /etc/apt/keyrings/teleport-archive-keyring.asc < /tmp/teleport-pubkey.asc + echo "deb [signed-by=/etc/apt/keyrings/teleport-archive-keyring.asc] https://apt.releases.teleport.dev/ubuntu $(. /etc/os-release; echo $VERSION_CODENAME) stable/v" | sudo tee /etc/apt/sources.list.d/teleport.list + sudo apt-get update -qq && sudo apt-get install -y teleport + ``` +3. Generate a short-lived join token from the auth server: + ```bash + tctl tokens add --type=node,db --ttl=15m --format=json + ``` +4. Write `/etc/teleport.yaml` on the new host: + ```yaml + version: v3 + teleport: + nodename: + data_dir: /var/lib/teleport + proxy_server: :443 + join_params: { token_name: "", method: token } + auth_service: { enabled: "no" } + proxy_service: { enabled: "no" } + ssh_service: + enabled: "yes" + labels: { name: "", tier: infra, provider: } + ``` + Add a `db_service` block too if this host also fronts a database (see Pattern D). +5. `sudo systemctl enable teleport && sudo systemctl start teleport`, then confirm from your own client — `tsh ls` should show the new node within seconds. +6. Also create the broad host-level login (`devops`/whatever the fleet convention is, with passwordless sudo) on the new host so it inherits the same team-wide access as every other host — a new scoped role is **additive**, never a replacement for existing broad access. Double-check the login actually exists on the OS (a cloud image may only ship `ubuntu`, not `devops`). + +## Pattern D — registering a database (`db_service`) + +```yaml +db_service: + enabled: "yes" + databases: + - name: + protocol: postgres # or mysql + uri: : + tls: + mode: verify-full # has a real, verifiable TLS cert (managed/cloud DB) + ca_cert_file: /etc/teleport/db-ca/.crt + # OR, for a self-hosted DB with no real cert (a plain dev postgres container, etc): + # mode: insecure + static_labels: { env: prod|dev, tier: infra } +``` +For AWS RDS, download the provider's CA bundle rather than guessing at `insecure` mode: +```bash +curl -s https://truststore.pki.rds.amazonaws.com/global/global-bundle.pem -o /etc/teleport/db-ca/aws-rds-global-bundle.pem +``` +Restart the agent (`systemctl restart teleport`) after any `teleport.yaml` edit. **If you're SSHed in over Teleport itself, this drops your own session** — that's expected, not a failure; reconnect and check `systemctl is-active teleport` + `journalctl -u teleport -n 40 --no-pager` for real errors. + +## Known gotcha: orphaned teleport processes shadow new resources + +A host can end up running **two teleport processes** — the current systemd-managed one, and an older one left over from a migration (started manually or via an old init method, its config file possibly already deleted, invisible to `systemctl`). The old process keeps heartbeating a stale copy of a resource under the same name, and the auth server can end up not surfacing your freshly-registered same-named resource in `tsh ls` / `tsh db ls` even though your new agent's own log says "started successfully." + +Symptom: a database or node you just configured doesn't show up, no errors anywhere obvious. Before chasing TLS or RBAC theories, check for a second process: +```bash +ps aux | grep '[t]eleport start' +``` +If found and clearly orphaned (no matching systemd unit, config path that no longer exists), confirm with the user, then stop it (`kill -TERM `, escalate to `-KILL` only if it doesn't exit) and restart the real service — the resource should appear cleanly. diff --git a/plugins/infra-ops-toolkit/skills/topology-diagram/SKILL.md b/plugins/infra-ops-toolkit/skills/topology-diagram/SKILL.md new file mode 100644 index 0000000..7c83471 --- /dev/null +++ b/plugins/infra-ops-toolkit/skills/topology-diagram/SKILL.md @@ -0,0 +1,56 @@ +--- +name: topology-diagram +description: Build or update the MBU Group SSH/database access topology diagram (a light-theme HTML+SVG page mirrored into a .drawio.xml file). Use when asked to visualize, draw, update, or "perbarui" the infra access topology, network diagram, or Teleport topology — including adding a new provider/host/database, correcting IPs or specs, or fixing arrow/layout issues. Always re-verify every value against live infra before writing it; never hand-type an IP, spec, or resource name from memory. +--- + +# Topology Diagram + +Two files always ship together and must stay in sync: +- `topologi-ssh-teleport.html` — the primary, richly-styled artifact (what people actually read) +- `topologi-ssh-teleport.drawio.xml` — an editable mirror for the team, generated from the same coordinates/content, one visual generation behind on cosmetic-only details (see "drawio limitations" below) + +Ask the user where these live in their vault/Downloads if unknown — don't assume a path. + +## Golden rule: verify before you draw + +Never write an IP, spec, capacity number, or resource name from memory or from the diagram's own prior state. Before touching the file, pull fresh data from the actual source: +- Droplets/VMs: `doctl compute droplet list --format Name,PublicIPv4,PrivateIPv4,Memory,VCPUs,Disk,Region,Status` +- Managed databases: `doctl databases list -o json` (get `storage_size_mib`, `size` slug, `num_nodes` — the list view alone omits storage) +- AWS compute: `aws ec2 describe-instances` (loop every region if the primary region search comes up empty — check with `aws ec2 describe-regions`) +- AWS databases: `aws rds describe-db-instances` +- Teleport's own view: `tsh ls`, `tsh db ls -f json`, `tsh app ls` — this is what's *actually* registered, which can differ from what a droplet list implies +- Cluster roles/config: `tctl get roles --format=json` (see teleport-onboard skill for where tctl actually runs) + +If a resource shows up in the diagram's current text but you can't re-confirm it from a live source, don't silently trust it — flag it as unverified in the diagram itself (dashed border + a short "perlu verifikasi manual" note) rather than deleting it or leaving it looking equally confident as verified data. This has caught real bugs: a database name that looked like a duplicate turned out to be a distinct self-hosted DB reachable only via `tsh db ls`, not `doctl`. + +## Design system (HTML) + +Everything is driven by CSS custom properties on `:root` — change the palette in one place: +```css +--bg, --surface, --line, --line-soft, --text, --text-dim, --text-faint /* neutrals */ +--accent /* the single ingress chokepoint (e.g. the proxy/gateway) — spend this color deliberately, once */ +--user, --core, --agent, --do, --db, --wire, --legacy, --aws /* one hue per semantic category, reused everywhere: node icon, node border stripe, matching edges, zone eyebrow labels */ +``` +Pick a light or dark ground deliberately; light was chosen last for readability and print/PDF friendliness. Typography: a display sans (Archivo) for titles, a mono face (JetBrains Mono) for every IP/port/command/capacity number — the mono choice is not decorative, it's because the subject is literally terminal/IP data. + +**Node card pattern** (repeated ~30×): a `` (not ``) drawing a rectangle rounded *only on the right* (left corners stay square, matching the left accent stripe): +``` +M{x},{y} L{x2-r},{y} A{r},{r} 0 0 1 {x2},{y+r} L{x2},{y2-r} A{r},{r} 0 0 1 {x2-r},{y2} L{x},{y2} Z +``` +plus a thin `` (`width:3-4, rx:0`) at the same x/y as the left accent stripe in the category color, plus a `` holding real HTML (icon svg + `` title + `.n-meta` lines, one of which is usually `.faint` for the least-important line). foreignObject is what makes rich HTML-in-SVG possible at all — don't try to hand-roll the same layout with plain `` elements. + +**Zone headers**: a small mono eyebrow label + a thin `` rule spanning the full content width — not a big tinted background band (tried first, looked muddy). **Provider containers** (DigitalOcean / Biznet / AWS): white fill, 2px border in that provider's category color, not a pastel fill. + +**Edges**: orthogonal ``s in the category color of what they represent (not what they connect), each with an arrow `` in ``. Labels are centered pills: wrap the label text in a flex-centered `
` inside a `` sized/positioned at the edge's true midpoint, styled as `border-radius:999px; background:var(--surface); border:1px solid var(--line);` — this reads as a chip floating on the line, not text mashed into the background. + +**Layout discipline**: when adding a new column (a new provider, a new agent lane), recenter it under its own container and re-run the numbers for every sibling that shares a header divider or trunk edge — don't leave one column narrower/off-center while others got the treatment. Widen the `viewBox` and extend header `` rules to the new right margin; don't leave them stopping short. + +**Print/PDF**: add an `@media print` block (`@page{size:landscape}`, `.layout{display:block}`, `page-break-after` on the diagram frame) so a *browser's own* print dialog doesn't crush a wide multi-column layout into an unreadable thumbnail. But if the user wants a PDF that matches the *on-screen* view exactly (not a print-reflowed one), that's a different tool — see the `html-to-pdf` skill instead of fighting the print dialog further. + +## drawio mirror + +drawio's `mxCell` style language can't do per-corner rounding, arbitrary SVG icon paths, or CSS-variable-driven theming. Don't hand-edit 30+ `mxCell` blocks for a structural change — write a small Python generator with helper functions (`node()`, `container()`, `edge()`, `eyebrow()`) that takes the *same* coordinate/color/text data as the HTML and emits the XML. This is far less error-prone than manual edits and keeps the two files derivable from one source of truth even though they're not literally the same code. + +Known, accepted gaps versus the HTML (state these to the user rather than silently mismatching): symmetric `arcSize` rounding on all four corners (not right-only), no custom icons (title + category color + left stripe carries the scanability instead), edge labels approximated via a child "edgeLabel" vertex with `rounded=1;arcSize=100` for the pill look. + +After any edit to either file: validate XML with `python3 -c "import xml.etree.ElementTree as ET; ET.parse('file.xml')"`, check ``/`` tag balance in the HTML, then `open ` to eyeball it in a browser before calling it done.