#!/usr/bin/env python3 """LYRA — Muse & Director for model + photographer shoots. Single-file Flask app. Deployed on Proxmox CT 172.""" import base64, json, os, random, re, requests from flask import Flask, jsonify, request, render_template_string OLLAMA_URL = os.environ.get("LYRA_OLLAMA_URL", "http://10.30.20.222:11434") TEXT_MODEL = os.environ.get("LYRA_TEXT_MODEL", "ornith-1.5:9b-64k") VISION_MODEL = os.environ.get("LYRA_VISION_MODEL", "minicpm-v4.5:8b") app = Flask(__name__) # ---------------------------------------------------------------- offline libraries LINES = { "sweet": [ "You have no idea what you look like through this lens right now.", "The light found you first today — I'm just keeping up.", "Every frame of you could hang in a gallery and you know it.", "You make my job suspiciously easy.", "That little smile just paid for this whole shoot.", ], "witty": [ "Careful — if you get any prettier the camera files a complaint.", "I'd tell you to look natural, but 'devastatingly gorgeous' is your natural.", "The camera asked for your autograph. I said no, it's my turn.", "Warning: frame is now 40% more beautiful than when we started.", "You're upstaging the lighting. The lighting is upset.", ], "confident": [ "Chin down, eyes to me. Own the frame — it's yours.", "Slow everything down. You're not posing, you're arriving.", "Give me that look like you already know the shot is perfect.", "Shoulders back, weight on the back hip. Power, not effort.", "Don't smile for me — smirk for the world.", ], "playful": [ "Laugh like I just told you the worst joke in the world. Go.", "Pretend the tripod said something rude about your outfit.", "Move like the music in your head is fantastic.", "Give me 'caught mid-secret.' Now 'committed to the secret.'", "Whisper your favorite curse word and we'll call it candid.", ], "hype": [ "THAT is the shot. Frame of the day, locked.", "You just made a 50mm lens feel inadequate.", "If this doesn't break the internet I'll fix the internet.", "Series complete — and you didn't even break a sweat.", "We're not deleting a single one of those.", ], } POSES = [ # standing ("Standing — Contrapposto", "Weight on back leg, front knee soft, hips angled 30° off camera, shoulders square to lens.", "standing"), ("Standing — Over-the-Shoulder Look", "Body turned away, look back over shoulder toward lens, hand near collarbone.", "standing"), ("Standing — Wall Lean", "Lean shoulder blades to wall, one knee bent, foot flat on wall.", "standing"), ("Standing — Hands in Hair", "Fingers threaded through hair at temples, elbows lifted, chin slightly up.", "standing"), ("Standing — Crossed Arms Confidence", "Arms crossed loosely, weight back, chin level, direct gaze.", "standing"), ("Standing — Skirt/Dress Swish", "Mid-motion fabric swish, elbows out, eyes closed or down.", "standing"), ("Standing — Back to Camera", "Face away, look up and over shoulder, arms relaxed at sides.", "standing"), ("Standing — Arms Framing Face", "Both hands framing jaw, fingers relaxed, eyes soft-focus to lens.", "standing"), # seated ("Seated — Chair Forward Lean", "Sit astride or sideways, forearms on chair back, lean toward lens.", "seated"), ("Seated — Floor Knees", "Kneel sitting on heels, hands on thighs, spine long, chin up.", "seated"), ("Seated — Legs to the Side", "Legs folded to one side, one hand propped, other in lap.", "seated"), ("Seated — Stool Elbow on Knee", "Elbow to knee, chin resting on knuckles, dead-eye gaze.", "seated"), ("Seated — Reclined Support", "Recline back on one straight arm, legs crossed at ankle.", "seated"), # wall / leaning ("Wall — Forearm Brace", "Forearm flat on wall, body angled, far hip toward lens.", "wall"), ("Wall — Full Back Press", "Back to wall, chin slightly down, eyes to lens, hands in pockets.", "wall"), ("Wall — Peeking Around Edge", "Head around corner of wall, playful suspicion.", "wall"), ("Doorway — Frame in Frame", "Stand in doorway, shoulders relaxed, feet crossed at ankle.", "wall"), # floor ("Floor — Lying on Stomach", "Prop on forearms, ankles crossed in air, chin on hands.", "floor"), ("Floor — Side Recline", "Lying on side, head on hand, top leg bent over.", "floor"), ("Floor — Back with Hair Fan", "On back, hair fanned, camera directly overhead.", "floor"), ("Floor — Sitting Hug Knees", "Hug knees, chin resting on knee, eyes up to lens.", "floor"), # movement ("Movement — Hair Flip", "Mid hair-flip, eyes closed 70% of the flips, burst mode.", "movement"), ("Movement — Walk Toward Lens", "Walk naturally, gaze at lens, arms loose, slight smile.", "movement"), ("Movement — Twirl", "Slow twirl with skirt or dress, shutter 1/500+, burst.", "movement"), ("Movement — Jump Capture", "Small jump, arms free — take 6+ frames, pick the apex.", "movement"), ("Movement — Turn Back Mid-Step", "Walking away, snap head-turn back at the peak.", "movement"), # closeup ("Closeup — Hand Near Face", "Fingertips on cheekbone or lips, eyes to lens.", "closeup"), ("Closeup — Eyes Closed Serenity", "Face to light, eyes closed, lips relaxed.", "closeup"), ("Closeup — Profile Edge", "Hard profile against clean background, chin lifted 5°.", "closeup"), ("Closeup — Over-Hand Gaze", "One hand over forehead shading eyes, gaze under it.", "closeup"), ("Closeup — Laugh Macro", "Real laugh, slight head tilt back — catch the in-between.", "closeup"), ] SYSTEM_LINE = ( "You are Lyra, a witty photographer's assistant on set. Generate exactly %d short lines " "the photographer can say ALOUD to the model between shots. Category: %s. " "Spice level 1-3: %d (1=sweet, 3=flirty-bold but tasteful). One line each, max 14 words, " "no numbering, no quotes, no explanation. One per line." ) SYSTEM_VISION = ( "You are a professional photography director reviewing a RAW frame from a live shoot. " "Reply in EXACTLY this format:\nVERDICT: PASS or NEEDS WORK\nSCORE: n/10\n" "LIGHTING: one sentence\nPOSE: one sentence of concrete adjustment for the model " "(body, hands, chin, eyes)\nSAY THIS: one short line the photographer can say out loud " "right now. Be specific, be brief." ) def ollama_generate(prompt, model, images=None, timeout=60): payload = {"model": model, "prompt": prompt, "stream": False, "options": {"temperature": 0.9, "num_predict": 700}} if "qwen3" in model: payload["think"] = False if images: payload = {"model": model, "messages": [ {"role": "user", "content": prompt, "images": images}], "stream": False, "keep_alive": "30m", "options": {"temperature": 0.4, "num_predict": 500}} r = requests.post(f"{OLLAMA_URL}/api/chat", json=payload, timeout=timeout) r.raise_for_status() return (r.json().get("message") or {}).get("content", "").strip() r = requests.post(f"{OLLAMA_URL}/api/generate", json=payload, timeout=timeout) r.raise_for_status() return r.json().get("response", "").strip() @app.get("/health") def health(): return jsonify(ok=True, app="lyra") @app.get("/api/lines") def api_lines(): cat = request.args.get("category", "witty") spice = int(request.args.get("spice", 2)) count = min(int(request.args.get("count", 5)), 10) fallback = random.sample(LINES.get(cat, LINES["witty"]), min(count, len(LINES.get(cat, [])))) try: out = ollama_generate(SYSTEM_LINE % (count, cat, spice), TEXT_MODEL, timeout=90) out = re.sub(r".*?", "", out, flags=re.S) # strip CoT blocks lines = [l.strip().strip('"-•') for l in out.splitlines() if l.strip()] lines = [l for l in lines if 3 < len(l) < 140 and not re.match( r"^(okay|let me|i (will|'ll|'m)|sure|here|the user|we need|first|next|okay so)", l, re.I)] lines = lines[-count:] if len(lines) > count else lines # CoT models put answers last if not lines: raise ValueError("empty") return jsonify(source="ollama", lines=lines) except Exception: return jsonify(source="offline", lines=fallback) @app.post("/api/analyze") def api_analyze(): data = request.get_json(force=True) img = data.get("image", "") if "," in img: img = img.split(",", 1)[1] try: raw = base64.b64decode(img) if len(raw) > 8_000_000: return jsonify(error="image too large (max ~8MB)"), 413 b64 = base64.b64encode(raw).decode() out = ollama_generate(SYSTEM_VISION, VISION_MODEL, images=[b64], timeout=120) return jsonify(analysis=out, source="ollama") except Exception as e: return jsonify(error=f"analysis unavailable: {e}"), 502 PAGE = r""" Lyra ✦ Muse & Director
✦

LYRA

muse & director — for the photographer and his star
browser voice

Say something ✦ tap any line to speak it

Mood Spice Count
Tap Generate — Lyra writes fresh lines via the local model, or falls back to her set book.

Pose book ✦

Light rig

Color Brightness Pulse

Turn screen brightness to max and face it at her as a soft fill. Pulse adds a slow breathing warmth.

Frame analyzer ✦ capture or drop a shot

Shot list

Shoot timer

00:00
Burst everys

Chimes each interval — keep the energy moving.

Settings ✦ saved on this device

Voice engine
Voice
Speech rate 1.0
Pitch 1.0
Lyra text model
Vision model
Ollama host
Auto-director interval 45s
Theme warmth
""" @app.get("/") def index(): poses_js = json.dumps([[n, d, c] for n, d, c in POSES]) html = PAGE.replace("__POSES__", poses_js) return html @app.post("/api/tts") def api_tts(): """Try Piper on the configured host (OpenAI-compat /v1/audio/speech); 404 -> client falls back.""" text = request.json.get("text", "")[:500] host = request.json.get("host", "") base = host or OLLAMA_URL try: r = requests.post(f"{base.rsplit(':', 1)[0]}:8080/v1/audio/speech", json={"input": text, "voice": "piper"}, timeout=20) if r.status_code == 200: return app.response_class(r.content, mimetype="audio/wav") except Exception: pass return jsonify(error="piper unavailable"), 404 if __name__ == "__main__": app.run(host="0.0.0.0", port=5000, threaded=True)