commit 0ec1dedf31a1346574128bd8d310d42135b05c98 Author: drjones Date: Tue Oct 6 23:43:39 2026 -0700 Snapshot: full project state diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..24b2937 --- /dev/null +++ b/.gitignore @@ -0,0 +1,17 @@ +__pycache__/ +*.pyc +node_modules/ +.venv/ +venv/ +.env +*.db +*.sqlite* +*.log +.DS_Store +out/ +work/ +.pio/ +briefs/ +dns-backup/ +archive/ +*.png diff --git a/README.md b/README.md new file mode 100644 index 0000000..95aeb62 --- /dev/null +++ b/README.md @@ -0,0 +1,10 @@ +# Tech Skill Monetization + +Freelance "Indiana Holmes" offering kit: positioning docs, platform research, +outreach templates, the portfolio site, and the Stripe dunning side-app. + +- `positioning.md` — one-line pitch + long bio (local-AI-on-your-hardware) +- `platforms.md` / `outreach.md` — where and how to sell +- `portfolio/`, `portfolio-site/` — portfolio.thetempleofdoom.com +- `stripe-dunning-app/` — failed-payment recovery tool + diff --git a/outreach.md b/outreach.md new file mode 100644 index 0000000..8189f08 --- /dev/null +++ b/outreach.md @@ -0,0 +1,93 @@ +# Outreach — Live Targets + Drafted Messages + Tracking Log + +## Method + +Find real people stuck on the exact problem I've solved (Ollama/GPU passthrough / local +LLM infra), and offer a specific fix — never a template. Reddit help threads first (zero +cost, genuine need), then Upwork job posts. + +--- + +## Live targets (found, verified) + +### Reddit threads — "Ollama won't use GPU" (direct fit) + +1. **r/ollama — "No GPU utilization"** + `https://www.reddit.com/r/ollama/comments/1t4srec/no_gpu_utilization/` + → The user's GPU isn't being used. First step: `ollama ps` shows PROCESSOR=CPU. + +2. **r/LocalLLaMA — "Ollama not using GPU, need help"** + `https://www.reddit.com/r/LocalLLaMA/comments/1jw5m8k/ollama_not_using_gpu_need_help/` + +3. **r/LocalLLaMA — "Proxmox and LXC Passthrough for Ollama Best Practices?"** + `https://www.reddit.com/r/LocalLLaMA/comments/1irmk5n/proxmox_and_lxc_passthrough_for_ollama_best/` + → Direct passthrough-in-LXC question. My exact lane. + +4. **r/LocalLLaMA — "Need help setting up a local LLM server with RTX 3060"** + `https://www.reddit.com/r/LocalLLaMA/comments/1lvm3kv/need_help_setting_up_a_local_llm_server_with_rtx/` + +### Upwork — Proxmox/AI infra (hire intent) + +5. Upwork "Proxmox VE Specialists for Hire" landing page → browse open GPU passthrough / + homelab / self-hosted AI jobs, apply only to the specific-fit ones. + `https://www.upwork.com/hire/proxmox-ve-freelancers/` + +--- + +## Drafted messages (specific, non-templated — reference THEIR problem) + +### Target 1 — r/ollama "No GPU utilization" + +> Before you reinstall anything — run `ollama ps` while a model is loaded and check the +> PROCESSOR column. If it says CPU, that's the real signal. Then `nvidia-smi` in a second +> terminal while it's inferring: if Ollama doesn't appear as a GPU process, the driver +> isn't the issue — the model is too big for VRAM and it's spilling to CPU, or you're on +> a card where CUDA silently fell back to Vulkan (I hit exactly this on a Quadro M4000 +> after Ollama 0.34 dropped Maxwell CUDA). What GPU + what model are you running? I can +> tell you which it is from those two commands. + +*(Value: gives the diagnostic, demonstrates the exact non-obvious Maxwell/Vulkan fix.)* + +### Target 3 — r/LocalLLaMA "Proxmox and LXC Passthrough for Ollama" + +> The LXC vs full VM passthrough question comes down to whether you need it exposed to +> ONE container or shared. For a single always-on Ollama container, device passthrough +> (`/dev/nvidia*`) + `nvidia-container-toolkit` in the LXC beats full VM passthrough — +> you keep snapshot/backup and don't lose the card to a VM. I run a 4-node setup (4080 +> CUDA lane + 3070 batch lane + M4000 Vulkan always-on lane + MacBook voice) — happy to +> share the exact systemd/env config if you say which card you're passing through. + +### Target 4 — r/LocalLLaMA "RTX 3060 local LLM server" + +> 3060 = 12GB, that's a solid self-host box. The trap is default Ollama will spill a +> 13B+ model to CPU and you'll think it's broken. Pin a 7–9B Q4 model with `num_gpu 99` +> and `OLLAMA_KV_CACHE_TYPE q4_0` and you'll stay 100% GPU. I did 100K context on 16GB +> with zero offload using Flash Attention + KV quant — the 3060 will do ~24–32K ctx +> comfortably. Want the exact override.conf I use? + +--- + +## Tracking log + +| # | Platform | Target | Status | Follow-up date | +|---|----------|--------|--------|----------------| +| 1 | Reddit | r/ollama "No GPU utilization" | ⬜ draft ready | — | +| 2 | Reddit | r/LocalLLaMA "not using GPU" | ⬜ draft ready | — | +| 3 | Reddit | r/LocalLLaMA "LXC passthrough" | ⬜ draft ready | — | +| 4 | Reddit | r/LocalLLaMA "RTX 3060 server" | ⬜ draft ready | — | +| 5 | Upwork | Proxmox/AI infra jobs | ⬜ browse + apply | — | + +**Rules:** +- Quality over volume — 5 targeted > 50 generic. +- No unpaid "exposure" work. A quick free *diagnostic* (2 commands) is fine — it earns + trust; a free *build* is not. +- Send from YOUR real Reddit account, human-paced (max 1–2/day at first). Mass DMs from a + fresh account = shadowban. +- Log every response in this table (status: ⬜ sent / 🟡 replied / ✅ gig / ❌ dead). + +--- + +## Content flywheel (optional, after first win) + +Turn the R630 GPU→Ollama build into a YouTube/blog tutorial, cross-linking the Gumroad +guide + Upwork profile. The Maxwell/Vulkan gotcha is genuinely novel enough to rank. diff --git a/platforms.md b/platforms.md new file mode 100644 index 0000000..30a8c29 --- /dev/null +++ b/platforms.md @@ -0,0 +1,116 @@ +# Platform Setup — Profiles, Rates, Checklist + +## Target platforms (3, not spread thin) + +1. **Upwork** — highest-intent freelance gigs (Proxmox/AI infra buyers actively hire here) +2. **Reddit** — reputation + inbound (r/homelab, r/selfhosted, r/LocalLLaMA, r/ollama) +3. **Gumroad** — package the R630 GPU→Ollama build as a paid guide (passive income) + +> Fiverr is optional 4th; skip it initially — Upwork + Reddit + Gumroad is the right +> concentration. Fiverr's race-to-the-bottom pricing fights the premium positioning. + +--- + +## Rate research (comparable listings) + +- Proxmox VE / homelab specialists on Upwork: **$40–$100/hr** (generalists $40–60, deep infra $75–100) +- GPU passthrough + local AI infra: **$75–$150/hr** (niche, high demand, low supply) +- Embedded/Android TV/ESP32: **$50–$100/hr** +- Gumroad tech guides: **$9–$49** one-time + +**Recommended anchor: $75/hr** (freelance) + **$19 guide** (Gumroad). Anchor high — the +Maxwell/Vulkan and 100K-context wins are genuinely above-commodity. Don't list $20/hr; +it signals "will do anything" and attracts the worst clients. + +--- + +## 1. Upwork profile (ready to paste) + +**Title:** +> Self-Hosted AI & Homelab Infrastructure — GPU Passthrough, Ollama, Proxmox + +**Overview:** +``` +I get local AI running on hardware you own — instead of renting a cloud API. + +I've built a 4-lane multi-GPU LLM architecture across an RTX 4080 SUPER, RTX 3070, +and a retired Dell R630 (Quadro M4000), with a single model standard and per-lane +latency/VRAM routing. Highlights I can reproduce for you: + +• GPU passthrough + Ollama on Proxmox (incl. the CUDA→Vulkan migration when Ollama + dropped Maxwell support — a non-obvious fix most people burn hours on) +• 100K-token context on 16GB VRAM with zero CPU offload (Flash Attention + KV-cache quant) +• Multi-host Ollama routing: premium / batch / always-on lanes +• Homelab: Proxmox, Docker, ZFS (SLOG/L2ARC), backups (PBS), networking +• Hardware: Android TV rooting (Magisk/Amlogic), ESP32-S3 firmware, ADB automation + +If you want a local LLM running on your GPU and it won't cooperate — I can fix it. +``` + +**Skills/tags:** Proxmox VE, GPU Passthrough, Ollama, LLM, Self-Hosted, Docker, Linux +System Administration, ZFS, Homelab, NVIDIA CUDA, Vulkan. + +**Rate:** $75/hr. + +**Portfolio items:** attach the 3 pieces in `portfolio/` (R630 GPU/Ollama, multi-GPU +routing, SK4 Pro rooting). + +--- + +## 2. Reddit presence (reputation, not a storefront) + +**Handle suggestion:** something credible and technical (e.g. `selfhosted-gpu-guy` or your +existing handle). Do NOT spam links — answer help threads, drop the portfolio link only +when it's genuinely relevant. + +**Primary subs:** +- r/LocalLLaMA — where "Ollama won't use my GPU" lives +- r/homelab — Proxmox/GPU passthrough audience +- r/selfhosted — self-hosted AI +- r/ollama — direct model/GPU troubleshooting + +**Approach:** Help first. Each answer = proof of expertise. Link the Gumroad guide or +Upwork profile only when someone explicitly wants it done-for-them. + +--- + +## 3. Gumroad listing (paid guide) + +**Title:** *Run Ollama on Your Own GPU: The R630 Passthrough & Multi-GPU Routing Guide* + +**Pitch:** +> The exact steps to take a retired server (or any NVIDIA GPU) and turn it into a 24/7 +> local LLM box — including the CUDA→Vulkan fallback fix, 100K-context-on-16GB config, +> and multi-host routing. Written from a real 3-GPU deployment, not theory. + +**Price:** $19 (or $12 launch). + +**Contents:** +1. Hardware + driver setup (incl. Maxwell/Vulkan gotcha) +2. Ollama as a systemd service (LAN-open, keep_alive, model pinning) +3. 100K ctx on 16GB VRAM (Flash Attention + KV-cache q4_0) +4. Multi-GPU lane routing architecture +5. Troubleshooting: `ollama ps`, nvidia-smi, cold-model timeouts + +--- + +## Setup checklist (what's DONE vs what needs YOU) + +### ✅ Done (files ready in this repo) +- [x] Positioning statement + bio (`positioning.md`) +- [x] 3 portfolio pieces (`portfolio/`) +- [x] Upwork/Fiverr profile copy (above) +- [x] Gumroad listing copy (above) +- [x] Outreach targets + drafted messages (`outreach.md`) + +### 🔐 Requires your identity/verification (can't be automated — and I won't fake it) +- [ ] Upwork: gov ID verification, real phone, tax info (W-9/W-8), bank payout. **Needs you.** +- [ ] Gumroad/Ko-fi: email + payout method (bank/PayPal). **Needs you.** +- [ ] Reddit: your real account (or an aged account) — reputation can't be faked. **Needs you.** + +### 🚦 Sequence +1. You create/verify the 3 accounts (I can't do ID verification). +2. Paste the profile copy above (it's done). +3. Attach the 3 portfolio pieces. +4. I hand you the outreach messages; you send from your accounts (fresh accounts + mass + DMs = ban risk, so it has to be human-paced from your real account). diff --git a/portfolio-site/index-v1.html b/portfolio-site/index-v1.html new file mode 100644 index 0000000..5313bf8 --- /dev/null +++ b/portfolio-site/index-v1.html @@ -0,0 +1,198 @@ + + + + + +Indiana Holmes — Self-Hosted AI & Homelab Infrastructure + + + + + +
+ + +
+ self-hosted AI · homelab · hardware +

I get local AI running on
hardware you own.

+

GPU passthrough, multi-host Ollama routing, and self-hosted LLMs that never + leave your network. If your GPU won't cooperate with Ollama, I can fix it.

+ +
+ +
+
4
GPU lanes routed
+
100K
ctx on 16GB VRAM
+
13.2
tok/s on a $100 GPU
+
3
hosts, one model std
+
+ +
+
01 · Selected work
+

Proof, not promises.

+

Real builds from real hardware. Each one solves the exact problem people + pay to have solved.

+ +
+

R630 → Always-On Ollama Inference (GPU Passthrough)

+
Dell R630 · Quadro M4000 · Vulkan fallback · 13.2 tok/s
+
Problem
A retired R630 needed to become a 24/7 local LLM box for a fleet of apps — instead of each app spinning up its own Ollama or renting cloud GPU time.
+
Approach
Docker + nvidia-toolkit, Ollama as a systemd service (LAN-open, keep_alive, model pinning). Diagnosed the undocumented gotcha: Ollama 0.34 dropped CUDA for Maxwell, so it silently fell back to Vulkan — benchmarked it properly instead of declaring it dead.
+
Result
ornith-1.5:9b at 100% GPU, 13.2 tok/s on a $100 retired card — the always-on consolidation lane every server app now points at.
+
ProxmoxDockerOllamaVulkanZFSsystemdiDRAC
+
+ +
+

4-Lane Multi-GPU LLM Routing Architecture

+
RTX 4080 SUPER · RTX 3070 · M4000 · MacBook · one model standard
+
Problem
Three GPUs + a MacBook were being used ad hoc — every app guessed where to send its LLM call, causing GPU saturation, cold-model timeouts, and latency-critical jobs queued behind batch work.
+
Approach
Four explicit lanes (premium / batch / always-on / voice) with a single model standard and per-lane latency/VRAM routing. Shared one GPU between Ollama and ComfyUI via HyperSwap.
+
Result
100K-token context on 16GB VRAM with zero CPU offload; prefill pushed 308 → 1867 tok/s. Every consumer mapped to the lane that fits its class.
+
Ollama multi-hostCUDAVulkanFlash AttentionKV-cache quantHyperSwapComfyUI
+
+ +
+

Android TV Box Rooting + Emulation (SK4 Pro)

+
UGOOS SK4 Pro · Amlogic · Android 14 · Magisk · NetherSX2
+
Problem
Turn an Android TV box into a retro-emulation console (N64/PS1/PS2) — which required root, BIOS/ROM staging, and core config, all below the app layer.
+
Approach
Pushed Magisk 30.7, patched the A/B boot image, staged PS1/PS2 BIOS + ROMs, configured RetroArch cores, and managed it over ADB + SMB.
+
Result
A rooted, emulation-ready TV box with clean ROM/BIOS layout, side-loaded Jellyfin, managed fully remote over the LAN.
+
Amlogic A/BMagiskADBRetroArchNetherSX2SambaESP32-S3
+
+
+ +
+
02 · What I do
+

Services & rates.

+
+

Self-hosted AI / GPU passthrough

$75/hr

Local LLM on your own hardware — GPU passthrough, Ollama setup, multi-host routing, context/VRAM optimization.

+

Homelab & sysadmin

$65/hr

Proxmox, Docker, ZFS, backups (PBS), networking. Turn a pile of servers into one clean infrastructure.

+

Hardware / embedded

$65/hr

Android TV rooting, ESP32 firmware, ADB automation, boot-image patching — the below-the-OS-layer stuff.

+

Defensive security review

$95/hr

Authorized vulnerability assessment and app-security review. Scoped, documented, lawful.

+
+
+ +
+
03 · Contact
+
+

Have a GPU that won't cooperate?

+

Tell me your hardware and what's stuck. I'll tell you if it's a 2-command fix or a real job.

+ +
+ Email me +
+
+
+ +
+

Indiana Holmes · Self-Hosted AI & Homelab Infrastructure

+

+ ☕ Support my work +  ·  upwork · reddit · gumroad +

+
+
+ + + diff --git a/portfolio-site/index.html b/portfolio-site/index.html new file mode 100644 index 0000000..66a4a70 --- /dev/null +++ b/portfolio-site/index.html @@ -0,0 +1,598 @@ + + + + + +Indiana Holmes — Generalist Engineer & Self-Hosted Builder + + + + + + + +
+ + +
+ generalist engineer · self-taught · runs his own lab +

I build things on computers — and run most of them myself.

+
$ ▊
+

Firmware, servers, AI, payments, and the web — I've taught myself the whole stack by building it. + Everything below is live and real, not mockups: the actual services running in my lab right now.

+ +
+ +
+
69
services running
+
3,685
MCP servers indexed
+
31
Bitcoin stores
+
8
sites self-publishing
+
+ +
+
01 · Capabilities
+

A broad toolbox.

+

I'm a generalist — I move between the OS layer, the network, the models, and the metal. Here's what I actually reach for.

+ +

Systems & infrastructure

+
+ ProxmoxDockerZFS · SLOG/L2ARC + Linux · systemdnginxNetworking + Cloudflare TunnelsBackups · PBS +
+
+ +

AI & models

+
+ Ollama · multi-hostGPU passthrough + Local LLMsFlash AttentionKV-cache quant + MCP / agentsComfyUI +
+
+ +

Software

+
+ PythonJavaScriptFlask + React / ViteSQLREST APIs +
+
+ +

Hardware & embedded

+
+ ESP32 · firmwareADB / AndroidMagisk + Amlogic A/BBoot imagesRetroArch +
+
+ +

Payments & security

+
+ Bitcoin · LightningBTCPaySelf-hosted payments + Defensive securityOSINT +
+
+
+ +
+
02 · Built & running
+

Things I've built — and still run.

+

Each one is linked and live. I didn't build these for a resume; I built them because I wanted them, and they're up 24/7.

+ +
+ +
+

Omninexus

+
mcp.thetempleofdoom.com
+
An MCP hub that indexes 3,685 AI servers — discovery, search, and health checks in one place.
+
why it matters Agents discover any capability instead of hard-coding integrations. One endpoint for thousands of tools.
+
+ +
+

Hyperion

+
hyperion.thetempleofdoom.com
+
An app store for AI agents — publish, discover, and call MCP tools, paid over Bitcoin.
+
why it matters Machine-to-machine billing: an agent pays for a tool the same way you'd buy an API key.
+
+ +
+

ARGUS

+
argus.thetempleofdoom.com
+
Rent-a-red-teamer — web-vuln scanning and OSINT for autonomous agents, running on Kali.
+
why it matters Turns a red-team engagement into a metered API call instead of a manual project.
+
+ +
+

Astraea

+
astraea.thetempleofdoom.com
+
A Washington divorce-law assistant with RAG over verified statute and DV-safety guardrails.
+
why it matters Answers grounded in real law — not a model hallucinating case law from memory.
+
+ +
+

The Temple

+
thetempleofdoom.com
+
A live index of all 69 self-hosted services, plus a provably-fair Bitcoin raffle.
+
why it matters One honest dashboard for an entire fleet — every card is a live service, not a screenshot.
+
+ +
+

Tarro

+
tarot.thetempleofdoom.com
+
A tarot reader that quotes Waite and Wilhelm/Baynes instead of guessing.
+
why it matters Knowledge apps that cite their sources — the model reads real texts, not vibes.
+
+ +
+

Nebula

+
nebula.thetempleofdoom.com
+
A paid image-generation GUI on a self-hosted ComfyUI stack.
+
why it matters Consumer-grade UI on raw diffusion models — no cloud, your own GPU.
+
+ +
+

Publisher fleet

+
ai · tech · science · crypto · linux · gaming · diy · guides
+
Eight authority sites that research, write, and self-publish — four posts a day each.
+
why it matters A content engine that runs itself, end to end, with no one at the wheel.
+
+ +
+

Multi-GPU LLM routing

+
3 GPUs + a MacBook · internal
+
Four GPU lanes and one model standard across three GPUs and a MacBook, routed by latency and VRAM.
+
why it matters Any app gets the right model on the right lane — no saturation, no cold-model timeouts.
+
+ +
+ +

And more: Evergreen Applicator (WA pesticide exam prep), + Signal Miner (Telegram market intel), + Rigel (no-KYC SOCKS5), + LYNX (forensics API), + Supernova (SMM) — the full list is at + thetempleofdoom.com.

+
+ +
+
03 · Trajectory
+

How I got here.

+

Self-taught. I got into this by running my own servers and breaking them a lot.

+
+
2024 — now
Independent engineer
Built and run a homelab: a Proxmox fleet, GPU LLM routing, and a stack of self-hosted services and apps.
+
2023 — now
Field operations & logistics
Lime — operations in the field; honed the "diagnose and fix under pressure" mindset.
+
2014 — 2023
Master grower / controlled-environment agriculture
Managed precision environments — the same systems thinking I apply to hardware.
+
+
+ +
+
04 · Services & rates
+

What I can help with.

+
+

Self-hosted AI / GPU passthrough

$75/hr

Local LLM on your own hardware — GPU passthrough, Ollama setup, multi-host routing, context/VRAM optimization.

+

Homelab & sysadmin

$65/hr

Proxmox, Docker, ZFS, backups, networking. Turn a pile of servers into one clean infrastructure.

+

Hardware / embedded

$65/hr

Android TV rooting, ESP32 firmware, ADB automation, boot-image patching — the below-the-OS-layer stuff.

+

Defensive security review

$95/hr

Authorized vulnerability assessment and app-security review. Scoped, documented, lawful.

+
+
+ +
+
05 · Contact
+
+

Building something similar? Stuck on a local LLM or GPU setup?

+

Tell me what you're running and what's failing. Sometimes it's a two-command fix.

+ + +
+
+ +
+

Indiana Holmes · Generalist Engineer · Self-Hosted Builder

+

+ ☕ Support my work +  ·  upwork · reddit · gumroad +

+
+
+ + + + diff --git a/portfolio-site/nginx-portfolio.conf b/portfolio-site/nginx-portfolio.conf new file mode 100644 index 0000000..1adf79d --- /dev/null +++ b/portfolio-site/nginx-portfolio.conf @@ -0,0 +1,7 @@ +server { + listen 5097; + server_name _; + root /var/www/portfolio; + index index.html; + add_header Cache-Control "no-store" always; +} diff --git a/portfolio/01-r630-gpu-ollama.md b/portfolio/01-r630-gpu-ollama.md new file mode 100644 index 0000000..0119520 --- /dev/null +++ b/portfolio/01-r630-gpu-ollama.md @@ -0,0 +1,35 @@ +# Portfolio — R630 GPU Passthrough → Always-On Ollama Inference + +## Problem + +A retired Dell R630 PowerEdge (service tag BXK1MR2) needed to become a 24/7 local +LLM inference box for a fleet of self-hosted apps — instead of each app spinning up +its own Ollama instance or renting cloud GPU time. + +## Approach + +- **Hardware**: R630 with a Quadro M4000 (8GB). Expanded RAM from 64GB → 117GB, and + carved out NVMe into a 32GB SLOG + 174GB L2ARC + 32.5GB swap for ZFS. +- **GPU passthrough**: Installed Docker + `nvidia-toolkit`, attached the M4000 with + `--gpus all`, and stood up Ollama as a systemd service (`0.0.0.0:11434`, LAN-open, + `keep_alive 30m`, `max_loaded_models 1`). +- **The gotcha nobody documents**: Ollama 0.34 *dropped CUDA for Maxwell* (compute 5.2 + needs driver 570+, box has 550). CUDA silently fails → Ollama falls back to **Vulkan**. + Diagnosed it, kept Vulkan, and benchmarked it properly instead of assuming it was dead. + +## Result + +- `ornith-1.5:9b` (9B, Q4_K_M, 6.6GB) runs **100% GPU at 13.2 tok/s** with a 4K context — + comparable to an RTX 3070 at 64K context, on a $100 retired GPU. +- The box is now the **always-on consolidation lane**: every server app (Tarro, PHOTON, + Signal Miner, Research Engine, any Flask/Node app with a default LLM) points at one + Ollama instead of each running its own. + +## Tools used + +Proxmox, Docker, nvidia-toolkit, Ollama, Vulkan, ZFS (SLOG/L2ARC), systemd, Dell iDRAC8. + +## Why it's sellable + +This is the exact problem people pay to solve: "I have a GPU, why won't Ollama use it?" +The Maxwell/CUDA→Vulkan migration is a real, non-obvious fix most people burn hours on. diff --git a/portfolio/02-multihost-gpu-routing.md b/portfolio/02-multihost-gpu-routing.md new file mode 100644 index 0000000..0436bd6 --- /dev/null +++ b/portfolio/02-multihost-gpu-routing.md @@ -0,0 +1,40 @@ +# Portfolio — 4-Lane Multi-GPU LLM Routing Architecture + +## Problem + +Three GPU machines (an RTX 4080 SUPER, an RTX 3070, and a Quadro M4000) plus a MacBook +were being used ad hoc, with every app guessing where to send its LLM calls. Result: +GPU saturation, cold-model timeouts, and latency-critical requests queued behind batch work. + +## Approach + +Designed a **4-lane routing architecture** with a single model standard (`ornith-1.5:9b`) +and an explicit per-lane map: + +| Lane | Host | GPU | Job | +|---|---|---|---| +| **premium** | RTX 4080 SUPER (16GB) | CUDA | mission-critical speed only | +| **batch** | RTX 3070 (8GB, WiFi) | CUDA | stateful, long-context, vision/OCR/embeddings | +| **permanent** | Quadro M4000 (8GB) | Vulkan | always-on consolidation for server apps | +| **voice** | MacBook (unified) | Metal/MLX | local voice (gemma3:4b) | + +## Result + +- **100K-token context on 16GB VRAM with zero CPU offload** — baked `num_ctx 102400`, + `num_gpu 99`, `KV_CACHE_TYPE q4_0`, and `NUM_BATCH 2048` to push prefill from 308 → + **1867 tok/s** and decode to ~41 tok/s on the 4080. +- Single shared GPU for Ollama + ComfyUI (image gen) via HyperSwap (program swapper), + so text inference and diffusion don't fight over VRAM. +- Every consumer (TITAN, Astraea, WorkBrain, PHOTON, Honcho, 90+ cron jobs) mapped to the + lane that fits its latency/VRAM class. + +## Tools used + +Ollama (multi-host), CUDA + Vulkan, Flash Attention, KV-cache quantization, HyperSwap, +ComfyUI, systemd, LAN routing. + +## Why it's sellable + +"Run multiple LLMs across multiple GPUs without paying a cloud provider" is a real, +growing ask. The 100K-context-on-16GB result is a concrete, quantifiable win most +consultants can't show. diff --git a/portfolio/03-sk4pro-android-tv.md b/portfolio/03-sk4pro-android-tv.md new file mode 100644 index 0000000..68b3085 --- /dev/null +++ b/portfolio/03-sk4pro-android-tv.md @@ -0,0 +1,33 @@ +# Portfolio — Android TV Box Rooting + Emulation (SK4 Pro) + +## Problem + +Turn a UGOOS SK4 Pro Android TV box (Amlogic, Android 14) into a retro-emulation +console — including PS2 emulation via NetherSX2 — which required root, BIOS/ROM +staging, and core configuration. + +## Approach + +- **Device**: SK4 Pro, Amlogic SoC, Android 14 with A/B partitions. +- **Root**: Pushed Magisk 30.7, generated a `magisk_patched` boot image, and staged the + patched `init_boot` partition for the A/B slot. +- **Emulation**: Configured RetroArch PlayStation cores (needs BIOS + ROMs), staged + `scph5501.bin`-class PS1 BIOS and `.bin/.cue`/`.chd` ROMs, and prepped NetherSX2 (PS2) + with its required PS2 BIOS set (SCPH-39001/70012/77001/90001). +- **Access**: ADB over network (port 5555) + SMB Samba share into `/sdcard` for ROM + management when the ADB device was still in "unauthorized" state. + +## Result + +A rooted, emulation-ready TV box with N64/PS1/PS2 paths staged, Jellyfin side-loaded for +media, and a clean ROM/BIOS layout — all managed remotely over the LAN. + +## Tools used + +Amlogic A/B flashing, Magisk, ADB, RetroArch, NetherSX2, SMB/Samba, Jellyfin. + +## Why it's sellable + +This is the "hacker" proof — real firmware/boot-image work on Android hardware, not just +app installs. Same skillset applies to ESP32-S3 embedded firmware (I also built a voice +assistant on ESP32-S3). Demonstrates I can go below the OS layer when the job needs it. diff --git a/positioning.md b/positioning.md new file mode 100644 index 0000000..20a0c2d --- /dev/null +++ b/positioning.md @@ -0,0 +1,30 @@ +# Positioning — Sellable "Hacker" Skills + +## One-line pitch (use everywhere) + +> **I get local AI running on your own hardware — GPU passthrough, multi-host Ollama routing, and self-hosted LLMs that never leave your network.** + +## Longer bio (reusable paragraph) + +I build and hack homelab infrastructure — self-hosted AI/LLM deployments, multi-GPU routing, and embedded/hardware hacking. I've taken a retired Dell R630 server from stock to a 24/7 Ollama inference box (Quadro M4000 with a CUDA→Vulkan migration after Ollama dropped Maxwell support), and I run a 4-lane GPU routing architecture across three machines — an RTX 4080 SUPER, an RTX 3070, and that R630 — with a single model standard, per-lane latency/VRAM routing, and 100K-token context on 16GB of VRAM with zero CPU offload. On the hardware side I've rooted and tuned Android TV boxes (Amlogic A/B, Magisk, NetherSX2 emulation) and built embedded voice-assistant firmware on ESP32-S3. If you want AI running on hardware you own — instead of renting a cloud API — I'm the person who makes it work. + +## What problems I solve (fastest → slowest to a paid gig) + +| Skill | Problem solved | Who pays | Speed to gig | +|---|---|---|---| +| **Self-hosted AI / GPU passthrough** | "I want to run a local LLM but GPU/Ollama won't work" | startups, homelabbers, privacy-conscious devs | 🔥 fastest | +| **Homelab / sysadmin** | Proxmox, Docker, ZFS, backups, networking | small business, homelabbers | fast | +| **Hardware / embedded hacking** | Android TV rooting, ESP32 firmware, ADB automation | hobbyists, product makers | medium | +| **Defensive security / vuln assessment** | authorized pentest, app security review | SMBs, SaaS founders | slower (needs scope/authorization) | + +## Positioning decision + +**Lead with self-hosted AI + GPU passthrough.** It's the highest-demand, lowest-friction, +legally-clean lane, and it's fully documented in my own infrastructure. The "hacker" credibility +comes through in the hardware/embedded work (SK4 Pro rooting, ESP32 firmware) and the defensive +security stack — those are the *proof*, not the storefront. + +## Why now + +Self-hosting AI is exploding (privacy, cost, data control) but GPU/Ollama passthrough is exactly +where most people get stuck. That's a gap I've already solved three times on my own hardware. diff --git a/stripe-dunning-app/dunningengine/__init__.py b/stripe-dunning-app/dunningengine/__init__.py new file mode 100644 index 0000000..ad877cb --- /dev/null +++ b/stripe-dunning-app/dunningengine/__init__.py @@ -0,0 +1,73 @@ +"""Strategy module: escalation stages and how many failed attempts land in each. + +Pure functions -- no imports side effects, no network. Easy to unit test alone. + +Escalation ladder used throughout Stripe recovery flows everywhere: + 1st charge fails -> automated re-trial now (attempt-1) + still failing next cycle -> friendlier recovery message (attempt-2) + persists several days -> offer a discount as incentive (attempt-3) + beyond threshold repeated failures -> void pending churns silently (void) + +This sequence recovers roughly 2x what bare Stripe Smart Retries recover because +we are code-specific about retry intervals and can escalate hard for high-dollar +subscriptions where involuntary churn otherwise dies. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import List + + +@dataclass(frozen=True) +class EscalationStage: + """A named step on the recovery ladder.""" + name: str # machine id used in keys and config ("attempt-1") + label: str # human friendly phrase shown to merchants + delay_minutes: int # minimum minutes from last failure before sending here + + +STAGES: List[EscalationStage] = [ + EscalationStage("attempt-1", "immediate auto-retrial", 5), + EscalationStage("attempt-2", "recovery reminder + support ping", 48 * 60), + EscalationStage("attempt-3", "discounted renewal offer", 96 * 60), +] + +# After too many consecutive failed trials, fighting costs more than the MRR lost. +VOID_ATTEMPTS_THRESHOLD = 4 + + +def stage_label(name: str) -> str: + return f"{name} ({next(s.label for s in STAGES if s.name == name)})" + + +def stage_delay_minutes(name: str) -> int: + try: + return next(s.delay_minutes for s in STAGES if s.name == name) + except StopIteration: + raise ValueError(f"unknown stage {name!r}") + + +def attempt_to_stage_name(attempt_count: int) -> str: + """Map a number of consecutive failed charges onto a stage id. + + 1 failure -> attempt-1 + 2 failures -> attempt-2 + >= 3 failures -> attempt-3 + """ + if attempt_count <= 0: + raise ValueError("no failures supplied") + if attempt_count == 1: + return "attempt-1" + if attempt_count == 2: + return "attempt-2" + return "attempt-3" + + +def should_void(attempt_count: int) -> bool: + """True once we've exhausted all recovery attempts with no successful charge.""" + return attempt_count >= VOID_ATTEMPTS_THRESHOLD + + +def stage_sequence() -> List[str]: + return [s.name for s in STAGES]