#!/usr/bin/env -S uv run --script
#
# /// script
# requires-python = ">=3.14"
# dependencies = [
# "accelerate>=1.14.0",
# "diffusers>=0.39.0",
# "gguf>=0.19.0",
# "llama-cpp-python>=0.3.34",
# "numpy>=2.5.1",
# "pillow>=12.3.0",
# "requests>=2.34.2",
# "spandrel>=0.4.1",
# "suntimes>=1.1.2",
# "torch>=2.13.0",
# "torchvision>=0.28.0",
# "transformers>=5.14.1",
# ]
# [tool.uv.sources]
# torch = [
# { index = "pytorch-cpu" },
# ]
# torchvision = [
# { index = "pytorch-cpu" },
# ]
# [[tool.uv.index]]
# name = "pytorch-cpu"
# url = "https://download.pytorch.org/whl/cpu"
# explicit = true
# ///
import os
OFFLINE = os.environ.get("OFFLINE", "1") != "0"
os.environ["HF_HUB_OFFLINE"] = "1" if OFFLINE else "0"
os.environ["HF_HUB_DISABLE_TELEMETRY"] = "1"
from pathlib import Path
from typing import Any
import argparse
import datetime as dt
import logging
import random
import re
import subprocess
import sys
import gc
import numpy as np
import requests
import torch
from llama_cpp import Llama
from huggingface_hub import hf_hub_download, snapshot_download
from diffusers import (
Flux2KleinPipeline,
Flux2Transformer2DModel,
GGUFQuantizationConfig,
)
from PIL import Image
from spandrel import ModelLoader
from suntimes import SunTimes
logger = logging.getLogger("wallpaper")
CAP = torch.backends.cpu.get_cpu_capability()
COMPUTE_DTYPE = torch.bfloat16 if CAP.startswith("AVX512") else torch.float32
DEVICE = "cpu"
def step_context(
now: dt.datetime,
latitude: float | None,
longitude: float | None,
elevation: int = 0,
) -> str:
Y = 2000 # dummy leap year to allow input X-02-29 (leap day)
seasons = [
("winter", (dt.date(Y, 1, 1), dt.date(Y, 3, 20))),
("spring", (dt.date(Y, 3, 21), dt.date(Y, 6, 20))),
("summer", (dt.date(Y, 6, 21), dt.date(Y, 9, 22))),
("autumn", (dt.date(Y, 9, 23), dt.date(Y, 12, 20))),
("winter", (dt.date(Y, 12, 21), dt.date(Y, 12, 31))),
]
date = now.date()
date = date.replace(year=Y)
season = next(season for season, (start, end) in seasons if start <= date <= end)
today = now.strftime(f"Date and time: %A %B %d, %Y at %H:%M %Z. Season: {season}.")
if latitude is None or longitude is None:
return today
sun = SunTimes(longitude, latitude, elevation)
today_sr = sun.riselocal(now)
today_ss = sun.setlocal(now)
sun_times = (
f"sunrise {today_sr.strftime('%H:%M')}, sunset {today_ss.strftime('%H:%M')}"
)
detail = "unknown"
_WMO_SHORT = {
0: "clear sky",
1: "mostly clear",
2: "partly cloudy",
3: "overcast",
45: "foggy",
48: "rime fog",
51: "light drizzle",
53: "drizzle",
55: "heavy drizzle",
61: "light rain",
63: "rain",
65: "heavy rain",
71: "light snow",
73: "snow",
75: "heavy snow",
80: "showers",
81: "moderate showers",
82: "heavy showers",
95: "thunderstorm",
96: "thunderstorm + hail",
99: "severe thunderstorm",
}
try:
url = (
f"https://api.open-meteo.com/v1/forecast?"
f"latitude={latitude}&longitude={longitude}"
f"¤t=weather_code,temperature_2m&timezone=auto"
)
data = requests.get(url, timeout=10).json()["current"]
code = data["weather_code"]
code_desc = _WMO_SHORT.get(code, "")
temp = float(data["temperature_2m"])
detail = f"{code_desc} {temp:.0f}°C"
except Exception as exc:
logger.warning("Weather fetch failed (%s), continuing anyway.", exc)
return f"{today} Daylight: {sun_times}. Weather: {detail}" if detail else today
def step_llm(
repo_id: str,
filename: str,
system_prompt: str,
max_tokens: int,
taste_prompts: list[str],
perspectives: list[str],
temperature: float,
top_p: float,
min_p: float,
ctx: str,
) -> str:
logger.info("Loading llm model...")
model_path = hf_hub_download(
repo_id=repo_id,
filename=filename,
)
llm = Llama(
model_path=model_path,
n_ctx=2048,
verbose=False,
)
taste_prompts = list(taste_prompts)
random.shuffle(taste_prompts)
taste_prompt_str = ", ".join(taste_prompts)
perspective = random.choice(perspectives)
logger.info(f"Perspective: {perspective}")
prompt = (
f"Taste reference: {taste_prompt_str}. "
f"Required perspective: {perspective}. "
f"Context: {ctx}. "
f"Reply with one text-to-image prompt only."
)
logger.info("Running llm...")
output = llm.create_chat_completion(
messages=[
{"role": "system", "content": system_prompt},
{
"role": "user",
"content": prompt,
},
],
max_tokens=max_tokens,
temperature=temperature,
top_p=top_p,
min_p=min_p,
)
del llm
gc.collect()
# might as well
text = output["choices"][0]["message"]["content"]
text = re.sub(r".*?", "", text, flags=re.DOTALL)
text = re.sub(r".*", "", text, flags=re.DOTALL)
return text.strip()
def step_diffuse(
repo_id: str,
filename: str,
gen_width: int,
gen_height: int,
steps: int,
prompt: str,
) -> list[Any]:
ckpt_path = f"https://huggingface.co/{repo_id}/blob/main/{filename}"
# populate cache
if not OFFLINE:
snapshot_download("black-forest-labs/FLUX.2-klein-4B")
# Phase 1: text encoding only
logger.info("Encoding prompt...")
pipe = Flux2KleinPipeline.from_pretrained(
"black-forest-labs/FLUX.2-klein-4B",
transformer=None,
vae=None,
torch_dtype=torch.bfloat16, # always, for memory purposes
local_files_only=OFFLINE,
).to(DEVICE)
with torch.inference_mode():
prompt_embeds, _text_ids = pipe.encode_prompt(prompt=prompt)
del pipe
gc.collect()
prompt_embeds = prompt_embeds.to(COMPUTE_DTYPE)
# Phase 2: denoise + decode, without added memory load
logger.info("Denoising...")
transformer = Flux2Transformer2DModel.from_single_file(
ckpt_path,
quantization_config=GGUFQuantizationConfig(compute_dtype=COMPUTE_DTYPE),
torch_dtype=COMPUTE_DTYPE,
config="black-forest-labs/FLUX.2-klein-4B",
subfolder="transformer",
device_map="cpu",
local_files_only=OFFLINE,
).to(DEVICE)
pipe = Flux2KleinPipeline.from_pretrained(
"black-forest-labs/FLUX.2-klein-4B",
text_encoder=None,
tokenizer=None,
transformer=transformer,
torch_dtype=COMPUTE_DTYPE,
local_files_only=OFFLINE,
).to(DEVICE)
out = pipe(
prompt_embeds=prompt_embeds,
width=gen_width,
height=gen_height,
num_inference_steps=steps,
)
del pipe, transformer, prompt_embeds
gc.collect()
return out.images
def upscale_tiled(
model: Any,
x: torch.Tensor,
scale: int,
tile: int = 256,
overlap: int = 16,
) -> torch.Tensor:
# x: [1, C, H, W] in [0, 1]. Peak memory is bounded by the tile size,
# not the image size. Each tile is padded with `overlap` px of context,
# which is cropped away on write-back, so tiles blend seamlessly.
_, c, h, w = x.shape
out = torch.zeros(1, c, h * scale, w * scale)
stride = tile - 2 * overlap
for y0 in range(0, h, stride):
for x0 in range(0, w, stride):
iy0, ix0 = max(y0 - overlap, 0), max(x0 - overlap, 0)
iy1 = min(y0 + stride + overlap, h)
ix1 = min(x0 + stride + overlap, w)
with torch.inference_mode():
up = model(x[:, :, iy0:iy1, ix0:ix1])
oy0, ox0 = y0 * scale, x0 * scale
oy1 = min(y0 + stride, h) * scale
ox1 = min(x0 + stride, w) * scale
py0, px0 = (y0 - iy0) * scale, (x0 - ix0) * scale
out[:, :, oy0:oy1, ox0:ox1] = up[
:, :, py0 : py0 + (oy1 - oy0), px0 : px0 + (ox1 - ox0)
]
del up
return out
def step_upscale(
images: list[Any],
target_width: int,
target_height: int,
repo_id: str,
filename: str,
tile: int,
) -> list[Any]:
model_path = hf_hub_download(repo_id=repo_id, filename=filename)
model = ModelLoader().load_from_file(model_path)
model.to(DEVICE).eval()
results: list[Any] = []
for im in images:
arr = np.array(im.convert("RGB"), dtype=np.uint8)
x = (
torch.from_numpy(arr)
.permute(2, 0, 1)
.unsqueeze(0)
.float()
.div(255)
.to(DEVICE)
)
if tile > 0:
up = upscale_tiled(model, x, model.scale, tile=tile)
else:
with torch.inference_mode():
up = model(x)
# Tensor is [1, C, H', W'] -> Permute to [H', W', C], scale from
# [0, 1] to uint8 and clamp
img_array = up[0].permute(1, 2, 0).mul(255).clamp(0, 255).byte().numpy()
del x, up
gc.collect()
final_img = Image.fromarray(img_array)
# by default 4x lands above the target (1280x720 -> 5120x2880), then
# Lanczos downscales to the target, averaging away upscaler noise.
target_size = (target_width, target_height)
if (final_img.width, final_img.height) != target_size:
final_img = final_img.resize(target_size, Image.LANCZOS)
results.append(final_img)
del model
gc.collect()
return results
def step_save(
images: list[Any], output_dir: Path, now: dt.datetime, prompt_text: str
) -> list[Path]:
output_dir.mkdir(parents=True, exist_ok=True)
saved = []
tag = now.strftime("%Y%m%d-%H%M%S")
for i, img in enumerate(images):
fname = f"wallpaper-{tag}" + (f"-{i}" if len(images) > 1 else "") + ".jpg"
path = output_dir / fname
img.save(path, "JPEG", quality=95)
saved.append(path)
txt_path = output_dir / f"prompt-{tag}.txt"
txt_path.write_text(prompt_text)
return saved
def step_wallpaper(command: list[str], saved: list[Path], store: Path) -> None:
idx = random.randrange(len(saved))
try:
perma = store / "generated_wallpaper.jpg"
_ = perma.open("wb").write(Path(saved[idx]).read_bytes())
logger.info(f"Setting wallpaper to {perma}")
_ = subprocess.run(command + [perma], check=True)
except Exception as e:
logger.warning(f"Could not set wallpaper: {e}")
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Generate wallpaper. Use OFFLINE=0 env var for first run.",
formatter_class=argparse.ArgumentDefaultsHelpFormatter,
)
wall = parser.add_argument_group("wallpaper")
wall.add_argument(
"--taste-prompt",
nargs="+",
default=[
"natural scenes",
"serene",
"trees",
"plants",
"landscape",
"brutalist concrete",
"hypermodern concrete architecture",
"moss",
"rocks",
"abstract concrete sculptures",
"precise water channels",
"megastructure of strange shape",
"distinct scales of objects, some very large for contrast",
"distant structures",
],
help="Taste keywords for the LLM art-director prompt.",
)
wall.add_argument(
"--perspectives",
nargs="+",
default=[
"low ground-level view looking up",
"aerial view from great height",
"interior view looking outward through an opening",
"interior view looking inward to a courtyard",
"distant view across open landscape, structures on the horizon",
"in-situ view from within the structures, human eye level",
"view from across a body of water",
"view from a field",
"framed view through a narrow gap or corridor",
],
help="Camera perspectives; one is chosen at random per run and "
"injected as a hard directive.",
)
wall.add_argument(
"--ephemeral-dir", default="/tmp/wallpaper/", help="Output directory."
)
wall.add_argument(
"--single-file-persist-path",
default="~/media/pictures/",
help="Single target for persistance.",
)
wall.add_argument(
"--target-width", type=int, default=3840, help="Final wallpaper width."
)
wall.add_argument(
"--target-height", type=int, default=2160, help="Final wallpaper height."
)
wall.add_argument(
"--set-command",
nargs="+",
default=["awww", "img"],
help="Command to run when setting wallpaper.",
)
llm = parser.add_argument_group("llm")
llm.add_argument(
"--llm-repo-id",
default="unsloth/gemma-4-E4B-it-GGUF",
help="HuggingFace repo for LLM to use to prompt image model.",
)
llm.add_argument(
"--llm-filename",
default="gemma-4-E4B-it-Q4_K_M.gguf",
help="File name of LLM in the repo to use.",
)
llm.add_argument(
"--llm-max-tokens",
type=int,
default=200,
help="Max tokens of prompt to generate.",
)
llm.add_argument(
"--llm-temperature",
type=float,
default=0.4,
help="Sampling temperature; llama-cpp-python's default is 0.2",
)
llm.add_argument(
"--llm-top-p",
type=float,
default=0.95,
help="Nucleus sampling threshold.",
)
llm.add_argument(
"--llm-min-p",
type=float,
default=0.05,
help="Minimum token probability relative to the top token",
)
llm.add_argument(
"--llm-system-prompt",
default=(
"You are a wallpaper art director. "
"Given the user's taste and the context, write one detailed text-to-image prompt for a diffusion model. "
"You are repeatedly called upon to do this, and must introduce variation into your output. "
"The taste reference lists the themes to build on; improvise around them, or drop some. "
"The required perspective MUST be used as the camera viewpoint of the image. "
"Times after sunset and before sunrise should generate dark images for night time. "
"Times during the day should track the brightness of the hour. "
"The weather should be reflected in your prompt. "
"The season should set the overall mood of the prompt. "
"Do not include the time, date, or location in your prompt. "
"Output ONLY the image prompt — no explanation or preamble."
),
help="System prompt for LLM.",
)
flux2 = parser.add_argument_group("Flux Klein 4B settings")
flux2.add_argument(
"--flux2-klein-repo-id",
default="unsloth/FLUX.2-klein-4B-GGUF",
help="HuggingFace repo for quantised model.",
)
flux2.add_argument(
"--flux2-klein-filename",
default="flux-2-klein-4b-Q8_0.gguf",
help="File name of quantised mode in the repo.",
)
flux2.add_argument(
"--flux2-klein-steps",
type=int,
default=4,
help="Steps to use when generating image.",
)
flux2.add_argument(
"--flux2-klein-gen-width", type=int, default=1280, help="Generated image width."
)
flux2.add_argument(
"--flux2-klein-gen-height",
type=int,
default=720,
help="Generated image height.",
)
sr = parser.add_argument_group("super-resolution")
sr.add_argument(
"--sr-repo-id",
default="Comfy-Org/Real-ESRGAN_repackaged",
help="HuggingFace repo hosting the ESRGAN-family checkpoint.",
)
sr.add_argument(
"--sr-filename",
default="RealESRGAN_x4plus.safetensors",
help="Checkpoint filename in the repo (any spandrel-supported "
".pth/.safetensors works).",
)
sr.add_argument(
"--sr-tile",
type=int,
default=256,
help="Tile size for tiled upscaling; bounds peak memory on "
"constrained machines. 0 = full-frame (best quality).",
)
wx = parser.add_argument_group("weather")
wx.add_argument(
"--latitude", type=float, default=51.75, help="Latitude for weather."
)
wx.add_argument(
"--longitude", type=float, default=-1.25, help="Longitude for weather."
)
wx.add_argument(
"--elevation",
type=int,
default=61,
help="Altitude (m) for sunrise/sunset times.",
)
return parser.parse_args()
def run() -> None:
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(message)s",
datefmt="%H:%M:%S",
stream=sys.stderr,
)
args = parse_args()
# crazy
now = dt.datetime.now(dt.datetime.now().astimezone().tzinfo)
logger.info(f"Capability {CAP}, dtype {COMPUTE_DTYPE}")
# 1. context
ctx = step_context(now, args.latitude, args.longitude, args.elevation)
logger.info(f"Context: {ctx}")
# 2. prompt rewrite via local LLM
prompt_text = step_llm(
args.llm_repo_id,
args.llm_filename,
args.llm_system_prompt,
args.llm_max_tokens,
list(args.taste_prompt),
list(args.perspectives),
args.llm_temperature,
args.llm_top_p,
args.llm_min_p,
ctx,
)
logger.info(f"Prompt: {prompt_text}")
# 3. diffusion at base resolution
logger.info(
"Diffuse: %dx%d @ %d steps",
args.flux2_klein_gen_width,
args.flux2_klein_gen_height,
args.flux2_klein_steps,
)
raw_images = step_diffuse(
args.flux2_klein_repo_id,
args.flux2_klein_filename,
args.flux2_klein_gen_width,
args.flux2_klein_gen_height,
args.flux2_klein_steps,
prompt_text,
)
# 4. super-resolution then scale
logger.info(
"Upscale then rescale: %dx%d -> %dx%d -> %dx%d via %s with tiling %d",
args.flux2_klein_gen_width,
args.flux2_klein_gen_height,
args.flux2_klein_gen_width * 4,
args.flux2_klein_gen_height * 4,
args.target_width,
args.target_height,
args.sr_filename,
args.sr_tile,
)
upscaled = step_upscale(
raw_images,
args.target_width,
args.target_height,
args.sr_repo_id,
args.sr_filename,
args.sr_tile,
)
# 5. save to configured output directory
saved = step_save(upscaled, Path(args.ephemeral_dir).expanduser(), now, prompt_text)
for p in saved:
logger.info("Saved: %s", p.resolve())
# 6. set wallpaper
step_wallpaper(
args.set_command,
saved,
Path(args.single_file_persist_path).expanduser(),
)
if __name__ == "__main__":
run()