From e5e923dd345001ffb826f04d7d86f35e9ed15dbb Mon Sep 17 00:00:00 2001 From: Hong Jiarong Date: Sat, 11 Jul 2026 14:30:50 +0800 Subject: [PATCH] feat: add repeatable alpha silo setup --- hub/deploy/NEW_SILO_RUNBOOK.md | 111 ++++++++ hub/deploy/README.md | 11 + hub/deploy/new_silo.sh | 360 ++++++++++++++++++++++++++ hub/deploy/render_new_silo_bundle.mjs | 211 +++++++++++++++ 4 files changed, 693 insertions(+) create mode 100644 hub/deploy/NEW_SILO_RUNBOOK.md create mode 100755 hub/deploy/new_silo.sh create mode 100755 hub/deploy/render_new_silo_bundle.mjs diff --git a/hub/deploy/NEW_SILO_RUNBOOK.md b/hub/deploy/NEW_SILO_RUNBOOK.md new file mode 100644 index 0000000..fb53c47 --- /dev/null +++ b/hub/deploy/NEW_SILO_RUNBOOK.md @@ -0,0 +1,111 @@ +# New Alpha Silo runbook + +Use this runbook for a new Organization. A Silo is not merely another row in +the existing database: it has an independent PostgreSQL role/database, Linux +service identity, systemd unit, secret directory/keyring, workspace, skill +store, loopback port, domain, Feishu app and provider credential. + +The repeatable entry point is: + +```sh +bash hub/deploy/new_silo.sh +``` + +It gathers values and writes a private deployment bundle below +`~/.cph-silo-plans//`. The directory and all generated files are +mode `0700`/`0600`. Never commit, paste into chat, or copy that directory into +an immutable release. Run the wizard once per Organization; do not edit a copy +from another Organization. + +## Inputs to collect + +The platform operator chooses the host, release, unique instance id, short +workspace path, unique loopback port, database name/role, resource ceilings and +public domain. The Organization administrator supplies: + +- Organization display name and slug; +- Feishu App ID, App Secret and bot Open ID; +- the first OWNER's Open ID and display name; +- an Organization-exclusive provider token and provider base URL. + +The Feishu app is scoped to this Silo. OAuth users authenticated by that app are +automatically admitted to this Organization; OWNER remains the initial +privileged membership used for controlled administration and bootstrap. An +empty initial team list does not block the Alpha. + +Obtain a person's Open ID from the Feishu user-get documentation page by +clicking the `user_id` value picker and selecting the person. Configure the +redirect URL shown by the generated `OPERATE.md`; it is required for first-time +OAuth admission. + +## Host prerequisites + +Before the first Silo on a host, install Node.js 24+, npm, rsync, PostgreSQL +server/client, `pg_isready`, systemd, bubblewrap, socat, `runuser`, `setpriv`, +`pg_dump`, tar, sha256sum, Nginx, Certbot, Typst and a compatible `cph` binary. +Configure outbound proxying independently at host/service level and verify both +GitHub and the selected model provider through it. The wizard does not install +or select proxy nodes. + +Use a deployment account with only the required passwordless sudo operations. +The application itself always runs as the installer-created non-root +`cph-` user. PostgreSQL must have a separate login role and logical +database per Silo even when all Silo databases share one PostgreSQL server. + +## Execute a generated bundle + +Open the bundle's `OPERATE.md` and perform its numbered gates in order: + +1. DNS, Feishu redirect URL, permissions and event subscription. +2. Dedicated PostgreSQL role and database. +3. Immutable release publication. Exit 78 is expected only when the first + installer call seeds this instance's keyring and environment template. +4. Root-owned secret installation and off-host keyring recovery copy. +5. Prisma migration, stopped service installation and idempotent bootstrap. +6. Explicit runtime role/skill installation. +7. Nginx/TLS, service start, internal/external health and Feishu acceptance. +8. First off-host backup. + +Every command must fail fast. Do not add `|| true` around install, migration, +bootstrap, Nginx validation, health, or backup checks. If a check fails, retain +the unit logs and the exact failed stage before changing configuration. + +## Default runtime role and skills + +Roles and skills are dynamic Silo state, not release contents. The release +contains only the management CLI. Stage approved skill directories on the host +and install them with `agent_config.sh`; PostgreSQL records role definitions and +skill selections while the versioned skill content lives in the Silo state +directory and is included in backups. + +For the current Alpha, upsert the default role with the agreed education +assistant system prompt, model selection, tools policy and approved skill list. +Keep the prompt in a root-controlled staging file, pass it via +`--system-prompt-file`, then verify with `agent_config.sh list`. Do not sync an +operator's entire personal skill directory: each enabled skill must be reviewed +and named explicitly. Typst being installed on the host and the Typst skill +being enabled are separate gates. + +## Acceptance gate + +A Silo is ready only when all of the following pass: + +- its systemd service is active and both loopback and TLS health endpoints pass; +- startup preflight sees exactly the configured Organization plus active Feishu + and provider connections; +- OWNER completes OAuth and can interact with the bot; +- a non-OWNER completes OAuth and can interact with the same app/Organization; +- two Feishu groups bind distinct projects/sessions; +- a restart preserves persisted session cursor behavior; +- the configured provider/model succeeds through the host proxy; +- every enabled skill is listed, and a Typst task succeeds if Typst is enabled; +- the first business backup and separate recovery backup are stored off-host. + +## Rollback boundary + +For a failed code release, point only this instance back to the previous +immutable release and rerun its installer with the same instance parameters. +Do not roll back a database after migrations unless that release's documented +database compatibility permits it. Preserve the environment, keyring, database, +workspace and skill store. For destructive recovery, stop traffic and use the +separate restore procedure; never substitute another Silo's state. diff --git a/hub/deploy/README.md b/hub/deploy/README.md index a4dd145..4e766e9 100644 --- a/hub/deploy/README.md +++ b/hub/deploy/README.md @@ -1,5 +1,16 @@ # Alpha Silo service installation +For a brand-new Organization, start with the repeatable wizard and end-to-end +operator runbook: + +```sh +bash hub/deploy/new_silo.sh +``` + +See [NEW_SILO_RUNBOOK.md](NEW_SILO_RUNBOOK.md). The remainder of this document +describes the individual installer and maintenance primitives used by the +generated bundle. + The supervised alpha runs one Organization per named Silo. The supported host has systemd, PostgreSQL, Node.js 24+, `pg_isready`, `runuser`, `setpriv`, bubblewrap, `socat`, `pg_dump`, `tar`, `sha256sum`, and a compatible `cph`. diff --git a/hub/deploy/new_silo.sh b/hub/deploy/new_silo.sh new file mode 100755 index 0000000..2266a22 --- /dev/null +++ b/hub/deploy/new_silo.sh @@ -0,0 +1,360 @@ +#!/usr/bin/env bash +# +# A wizard — walks a human through a manual procedure step by step. +# Generated by the /wizard skill. +# +# Everything above the "STAGES" marker is the wizard library: do not hand-edit +# it. Author the per-step stages below the marker. + +set -euo pipefail + +# ────────────────────────────────────────────────────────────────────────── +# Wizard library — delightful, consistent UX. Identical across every wizard. +# ────────────────────────────────────────────────────────────────────────── + +if [[ -t 1 ]] && command -v tput >/dev/null 2>&1 && [[ "$(tput colors 2>/dev/null || echo 0)" -ge 8 ]]; then + BOLD=$(tput bold); DIM=$(tput dim); RESET=$(tput sgr0) + BLUE=$(tput setaf 4); GREEN=$(tput setaf 2); YELLOW=$(tput setaf 3); RED=$(tput setaf 1) +else + BOLD=""; DIM=""; RESET=""; BLUE=""; GREEN=""; YELLOW=""; RED="" +fi + +# Author sets these two at the top of the stages section. +TOTAL_STAGES=0 +TOTAL_MINUTES=0 + +_STAGE_INDEX=0 +_MINUTES_ELAPSED=0 +ENV_FILE="${ENV_FILE:-.env}" +WRITTEN_ENV=() # KEYs written to ENV_FILE this run +WRITTEN_SECRET=() # secret NAMEs set this run +SKIPPED=() # things we couldn't do (e.g. gh missing) + +# _clear — wipe the terminal so only the current step is on screen. No-op when +# output isn't a terminal, so piped logs stay readable. +_clear() { + [[ -t 1 ]] || return 0 + if command -v tput >/dev/null 2>&1; then tput clear; else printf '\033[2J\033[3J\033[H'; fi +} + +# banner "Title" — opening frame: what this wizard does and how long it takes. +banner() { + _clear + printf '\n%s%s %s%s\n' "$BOLD" "$BLUE" "$1" "$RESET" + printf '%s %s stages · about %s minutes%s\n\n' \ + "$DIM" "$TOTAL_STAGES" "$TOTAL_MINUTES" "$RESET" + printf '%s You drive the browser; this wizard tells you exactly what to do and\n' "$DIM" + printf ' captures the values you copy back. Stop any time with Ctrl-C and re-run\n' + printf ' later — it remembers values already saved.%s\n' "$RESET" + pause "Ready to start?" +} + +# stage "Name" — clear the screen, then announce a stage and show +# progress + time remaining. Clearing keeps only the current step on screen. +stage() { + _clear + _STAGE_INDEX=$((_STAGE_INDEX + 1)) + local remaining=$((TOTAL_MINUTES - _MINUTES_ELAPSED)) + (( remaining < 0 )) && remaining=0 + _MINUTES_ELAPSED=$((_MINUTES_ELAPSED + ${2:-0})) + printf '\n%s%s▸ Stage %s/%s · %s%s %s(~%s min left)%s\n' \ + "$BOLD" "$BLUE" "$_STAGE_INDEX" "$TOTAL_STAGES" "$1" "$RESET" "$DIM" "$remaining" "$RESET" +} + +# say "..." — a plain instruction line. +say() { printf ' %s\n' "$1"; } +# step "..." — a numbered-feeling action the human takes in the browser. +step() { printf ' %s•%s %s\n' "$BLUE" "$RESET" "$1"; } +note() { printf ' %s%s%s\n' "$DIM" "$1" "$RESET"; } +warn() { printf ' %s⚠ %s%s\n' "$YELLOW" "$1" "$RESET"; } + +# open_url URL — open in the human's browser, cross-platform incl. WSL. +open_url() { + local url="$1" + printf ' %s↗ opening%s %s\n' "$GREEN" "$RESET" "$url" + { if command -v wslview >/dev/null 2>&1; then wslview "$url" + elif command -v explorer.exe >/dev/null 2>&1; then explorer.exe "$url" + elif command -v xdg-open >/dev/null 2>&1; then xdg-open "$url" + elif command -v open >/dev/null 2>&1; then open "$url" + else warn "couldn't open a browser — visit it manually: $url"; fi + } >/dev/null 2>&1 || warn "couldn't open a browser — visit it manually: $url" +} + +# pause "msg" — wait for the human to confirm they've done the manual part. +pause() { + printf ' %s%s%s ' "$DIM" "${1:-Press Enter to continue}" "$RESET" + read -r _ || true +} + +# confirm "question" — y/N gate; returns success on yes. +confirm() { + local reply="" + printf ' %s? %s [y/N] ' "$YELLOW" "$1" + read -r reply || true + [[ "$reply" =~ ^[Yy] ]] +} + +# _existing KEY — current value of KEY in ENV_FILE, if any. +_existing() { + [[ -f "$ENV_FILE" ]] || return 1 + local line; line=$(grep -E "^${1}=" "$ENV_FILE" | tail -n1) || return 1 + printf '%s' "${line#*=}" +} + +# ask KEY "Prompt" — read a value into $KEY. Offers the existing .env value as +# a default on re-runs (Enter keeps it). Visible input (non-secret). +ask() { + local key="$1" prompt="$2" current input + current=$(_existing "$key" || true) + if [[ -n "$current" ]]; then + printf ' %s%s%s %s[Enter keeps current]%s ' "$BOLD" "$prompt" "$RESET" "$DIM" "$RESET" + else + printf ' %s%s%s ' "$BOLD" "$prompt" "$RESET" + fi + read -r input || true + [[ -z "$input" && -n "$current" ]] && input="$current" + printf -v "$key" '%s' "$input" +} + +# ask_secret KEY "Prompt" — like ask, but input is hidden. +ask_secret() { + local key="$1" prompt="$2" current input + current=$(_existing "$key" || true) + if [[ -n "$current" ]]; then + printf ' %s%s%s %s[Enter keeps current]%s ' "$BOLD" "$prompt" "$RESET" "$DIM" "$RESET" + else + printf ' %s%s%s ' "$BOLD" "$prompt" "$RESET" + fi + read -rs input || true + printf '\n' + [[ -z "$input" && -n "$current" ]] && input="$current" + printf -v "$key" '%s' "$input" +} + +# write_env KEY VALUE — upsert KEY=VALUE into ENV_FILE (creates it; replaces +# any existing line). Idempotent. +write_env() { + local key="$1" value="$2" tmp + touch "$ENV_FILE" + tmp=$(mktemp) + grep -vE "^${key}=" "$ENV_FILE" > "$tmp" || true + printf '%s=%s\n' "$key" "$value" >> "$tmp" + mv "$tmp" "$ENV_FILE" + WRITTEN_ENV+=("$key") + printf ' %s✓ wrote%s %s → %s\n' "$GREEN" "$RESET" "$key" "$ENV_FILE" +} + +# set_secret NAME VALUE — set a GitHub Actions repo secret via gh. Falls back +# to a warning (and records it) if gh is unavailable or unauthenticated. +set_secret() { + local name="$1" value="$2" + if command -v gh >/dev/null 2>&1 && gh auth status >/dev/null 2>&1; then + if printf '%s' "$value" | gh secret set "$name" >/dev/null 2>&1; then + WRITTEN_SECRET+=("$name") + printf ' %s✓ set%s GitHub secret %s\n' "$GREEN" "$RESET" "$name" + return + fi + fi + SKIPPED+=("GitHub secret $name (set it manually: gh secret set $name)") + warn "skipped GitHub secret $name — gh not ready; set it later" +} + +# set_var NAME VALUE — set a GitHub Actions repo variable (non-secret). +set_var() { + local name="$1" value="$2" + if command -v gh >/dev/null 2>&1 && gh auth status >/dev/null 2>&1; then + if gh variable set "$name" --body "$value" >/dev/null 2>&1; then + printf ' %s✓ set%s GitHub variable %s\n' "$GREEN" "$RESET" "$name" + return + fi + fi + SKIPPED+=("GitHub variable $name") + warn "skipped GitHub variable $name — gh not ready; set it later" +} + +# finish — clear, then a closing summary of everything configured. +finish() { + _clear + printf '\n%s%s ✓ Setup complete%s\n' "$BOLD" "$GREEN" "$RESET" + (( ${#WRITTEN_ENV[@]} )) && note "wrote ${#WRITTEN_ENV[@]} value(s) to $ENV_FILE: ${WRITTEN_ENV[*]}" + (( ${#WRITTEN_SECRET[@]} )) && note "set ${#WRITTEN_SECRET[@]} GitHub secret(s): ${WRITTEN_SECRET[*]}" + if (( ${#SKIPPED[@]} )); then + printf '\n'; warn "still to do by hand:" + for s in "${SKIPPED[@]}"; do note " - $s"; done + fi + printf '\n' +} + +# ────────────────────────────────────────────────────────────────────────── +# STAGES — author this section. One stage() per step the human takes. +# Replace the example below. Set the two totals to match the stages you write. +# ────────────────────────────────────────────────────────────────────────── + +TOTAL_STAGES=7 +TOTAL_MINUTES=35 + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" +PLAN_ROOT="${CPH_SILO_PLAN_ROOT:-$HOME/.cph-silo-plans}" +umask 077 + +require_value() { + local key="$1" value="$2" + [[ -n "$value" ]] || { warn "$key is required"; exit 1; } + [[ "$value" != *$'\n'* && "$value" != *$'\r'* ]] || { + warn "$key must be a single line" + exit 1 + } +} + +seed_default() { + local key="$1" value="$2" + _existing "$key" >/dev/null 2>&1 || write_env "$key" "$value" +} + +capture() { + local key="$1" prompt="$2" + ask "$key" "$prompt" + require_value "$key" "${!key}" + write_env "$key" "${!key}" +} + +capture_secret() { + local key="$1" prompt="$2" + ask_secret "$key" "$prompt" + require_value "$key" "${!key}" + write_env "$key" "${!key}" +} + +banner "New Alpha Silo" + +stage "Silo identity and private plan" 3 +say "One run creates one Organization's private deployment bundle." +ask INSTANCE_ID "Unique instance id (lowercase, max 24 chars; e.g. school-a):" +require_value INSTANCE_ID "$INSTANCE_ID" +[[ "$INSTANCE_ID" =~ ^[a-z0-9]([a-z0-9-]{0,22}[a-z0-9])?$ ]] || { + warn "invalid instance id" + exit 1 +} +OUTPUT_DIR="$PLAN_ROOT/$INSTANCE_ID" +mkdir -p "$OUTPUT_DIR" +chmod 0700 "$OUTPUT_DIR" +ENV_FILE="$OUTPUT_DIR/answers.env" +touch "$ENV_FILE" +chmod 0600 "$ENV_FILE" +write_env INSTANCE_ID "$INSTANCE_ID" +seed_default ORGANIZATION_ID "$INSTANCE_ID" +seed_default ORGANIZATION_SLUG "$INSTANCE_ID" +capture ORGANIZATION_ID "Organization id:" +capture ORGANIZATION_SLUG "Organization slug:" +capture ORGANIZATION_NAME "Organization display name:" + +stage "Host, release, and isolation" 5 +say "Choose values that are unique on the shared host. The service binds loopback only." +seed_default DEPLOY_USER "root" +seed_default DEPLOY_SSH_PORT "22" +seed_default DEPLOY_BASE "/srv/curriculum-project-hub" +seed_default RELEASE_ID "$(git -C "$REPO_ROOT" describe --tags --exact-match 2>/dev/null || git -C "$REPO_ROOT" rev-parse --short HEAD)" +seed_default MEMORY_MAX "16G" +seed_default CPU_QUOTA "400%" +seed_default TASKS_MAX "512" +seed_default CPH_BIN "/usr/local/bin/cph" +capture DEPLOY_HOST "Server IP or SSH host:" +capture DEPLOY_USER "SSH deploy user:" +capture DEPLOY_SSH_PORT "SSH port:" +capture DEPLOY_SSH_KEY "Absolute path to SSH private key:" +capture DEPLOY_BASE "Remote release base:" +capture RELEASE_ID "Immutable release id/tag:" +capture HUB_PORT "Unique loopback Hub port:" +capture WORKSPACE_ROOT "Unique short workspace path (at most 16 bytes; e.g. /w/102):" +capture MEMORY_MAX "systemd MemoryMax:" +capture CPU_QUOTA "systemd CPUQuota:" +capture TASKS_MAX "systemd TasksMax:" +capture CPH_BIN "Remote cph binary path:" + +stage "Dedicated PostgreSQL database" 4 +say "A PostgreSQL server may be shared, but this Silo gets a distinct login role and database." +seed_default DATABASE_HOST "127.0.0.1" +seed_default DATABASE_PORT "5432" +seed_default DATABASE_NAME "cph_${INSTANCE_ID//-/_}" +seed_default DATABASE_USER "cph_${INSTANCE_ID//-/_}" +capture DATABASE_HOST "Database host as seen by the Hub service:" +capture DATABASE_PORT "Database port:" +capture DATABASE_NAME "Dedicated database name:" +capture DATABASE_USER "Dedicated database login role:" +capture_secret DATABASE_PASSWORD "New database password:" +note "The generated OPERATE.md uses an interactive/protected SQL path; the password is never put in a command argument." + +stage "Public URL and Feishu app" 9 +open_url "https://open.feishu.cn/app" +say "Create or open the Organization's own app. Copy credentials from Credentials & Basic Info." +capture PUBLIC_BASE_URL "Public base URL including https:// (e.g. https://school-a.example.com):" +capture FEISHU_APP_ID "Feishu App ID:" +capture_secret FEISHU_APP_SECRET "Feishu App Secret:" +capture FEISHU_BOT_OPEN_ID "Bot Open ID:" +open_url "https://open.feishu.cn/document/server-docs/contact-v3/user/get" +step "In the user/get page, click the user_id value picker, select the first OWNER, and copy the returned open_id." +capture OWNER_OPEN_ID "OWNER Open ID (ou_...):" +capture OWNER_DISPLAY_NAME "OWNER display name:" +ask OWNER_UNION_ID "OWNER Union ID (optional; Enter to skip):" +write_env OWNER_UNION_ID "$OWNER_UNION_ID" +say "The exact redirect URL and acceptance steps will be written to OPERATE.md." + +stage "Provider and Alpha limits" 5 +say "Use a provider credential exclusive to this Organization. Host proxy setup is a separate prerequisite." +seed_default PROVIDER_ID "openrouter" +seed_default PROVIDER_BASE_URL "https://openrouter.ai/api" +seed_default DEFAULT_MODEL "anthropic/claude-sonnet-5" +seed_default DEFAULT_ROLE_ID "draft" +seed_default DEFAULT_ROLE_LABEL "草稿" +seed_default MAX_TURNS "25" +seed_default MAX_CONCURRENT_RUNS "4" +seed_default MAX_RUN_SECONDS "900" +seed_default HTTP_BODY_LIMIT_BYTES "1048576" +seed_default MAX_FILES_PER_MESSAGE "8" +seed_default MAX_FILE_BYTES "26214400" +seed_default HTTP_REQUESTS_PER_MINUTE "120" +seed_default FEISHU_EVENTS_PER_MINUTE "120" +capture PROVIDER_ID "Provider id:" +capture PROVIDER_BASE_URL "Provider base URL:" +capture_secret PROVIDER_AUTH_TOKEN "Provider auth token:" +capture DEFAULT_MODEL "Default model id exposed by this provider:" +capture DEFAULT_ROLE_ID "Default role id:" +capture DEFAULT_ROLE_LABEL "Default role label:" +ask APPROVED_SKILLS "Approved installed skill names, comma-separated (optional):" +write_env APPROVED_SKILLS "$APPROVED_SKILLS" +capture MAX_TURNS "Maximum turns per run:" +capture MAX_CONCURRENT_RUNS "Organization concurrent runs:" +capture MAX_RUN_SECONDS "Maximum run seconds:" +capture HTTP_BODY_LIMIT_BYTES "HTTP body limit bytes:" +capture MAX_FILES_PER_MESSAGE "Maximum files per message:" +capture MAX_FILE_BYTES "Maximum bytes per file:" +capture HTTP_REQUESTS_PER_MINUTE "HTTP requests per minute:" +capture FEISHU_EVENTS_PER_MINUTE "Feishu events per minute:" +if ! _existing HUB_SESSION_SECRET >/dev/null 2>&1; then + command -v openssl >/dev/null 2>&1 || { warn "openssl is required"; exit 1; } + HUB_SESSION_SECRET="$(openssl rand -hex 32)" + write_env HUB_SESSION_SECRET "$HUB_SESSION_SECRET" +fi + +stage "Render the private deployment bundle" 2 +say "This renders platform.env, bootstrap.json, deploy.env, nginx.conf and OPERATE.md." +command -v node >/dev/null 2>&1 || { warn "Node.js is required to render safely"; exit 1; } +node "$SCRIPT_DIR/render_new_silo_bundle.mjs" "$ENV_FILE" "$OUTPUT_DIR" +chmod 0700 "$OUTPUT_DIR" +chmod 0600 "$OUTPUT_DIR"/* +say "Bundle: $OUTPUT_DIR" +warn "It contains database, Feishu, provider and session secrets. Never commit or paste it." + +stage "Operator handoff and gates" 7 +say "Open OPERATE.md and execute its stages in order. Nothing has changed on the server yet." +step "Verify DNS, Feishu redirect/permissions/events and the host proxy." +step "Create the dedicated database role/database and publish the release." +step "Install root-only secrets, back up the keyring, migrate and bootstrap." +step "Install only reviewed runtime skills and the default role configuration." +step "Validate Nginx, start the service, run health/Feishu/session/Typst acceptance, then back up." +if confirm "Print the bundle filenames now?"; then + find "$OUTPUT_DIR" -maxdepth 1 -type f -exec basename {} \; | sort +fi + +finish diff --git a/hub/deploy/render_new_silo_bundle.mjs b/hub/deploy/render_new_silo_bundle.mjs new file mode 100755 index 0000000..a58750a --- /dev/null +++ b/hub/deploy/render_new_silo_bundle.mjs @@ -0,0 +1,211 @@ +#!/usr/bin/env node + +import { chmod, mkdir, readFile, writeFile } from "node:fs/promises"; +import { resolve } from "node:path"; + +const [answersPath, outputPath] = process.argv.slice(2); +if (!answersPath || !outputPath) { + throw new Error("usage: render_new_silo_bundle.mjs ANSWERS_ENV OUTPUT_DIR"); +} + +function parseAnswers(source) { + const values = new Map(); + for (const [index, line] of source.split("\n").entries()) { + if (!line || line.startsWith("#")) continue; + const separator = line.indexOf("="); + if (separator < 1) throw new Error(`invalid answers line ${index + 1}`); + const key = line.slice(0, separator); + const value = line.slice(separator + 1); + if (!/^[A-Z][A-Z0-9_]*$/.test(key)) { + throw new Error(`invalid answers key on line ${index + 1}: ${key}`); + } + if (value.includes("\r") || value.includes("\n")) { + throw new Error(`multiline value is not supported: ${key}`); + } + values.set(key, value); + } + return values; +} + +function required(values, key) { + const value = values.get(key); + if (!value) throw new Error(`missing required answer: ${key}`); + return value; +} + +function optional(values, key, fallback = "") { + return values.get(key) || fallback; +} + +function assertMatch(label, value, pattern) { + if (!pattern.test(value)) throw new Error(`invalid ${label}: ${value}`); +} + +function envLine(key, value) { + if (/[\r\n]/.test(value)) throw new Error(`unsafe newline in ${key}`); + return `${key}=${value}`; +} + +const answers = parseAnswers(await readFile(resolve(answersPath), "utf8")); +const instanceId = required(answers, "INSTANCE_ID"); +const orgId = required(answers, "ORGANIZATION_ID"); +const orgSlug = required(answers, "ORGANIZATION_SLUG"); +const port = required(answers, "HUB_PORT"); +const workspaceRoot = required(answers, "WORKSPACE_ROOT"); +const publicBaseUrl = required(answers, "PUBLIC_BASE_URL").replace(/\/$/, ""); +const domain = new URL(publicBaseUrl).hostname; +const databasePassword = required(answers, "DATABASE_PASSWORD"); +const databaseUrl = `postgresql://${encodeURIComponent(required(answers, "DATABASE_USER"))}:${encodeURIComponent(databasePassword)}@${required(answers, "DATABASE_HOST")}:${required(answers, "DATABASE_PORT")}/${encodeURIComponent(required(answers, "DATABASE_NAME"))}`; + +assertMatch("INSTANCE_ID", instanceId, /^[a-z0-9](?:[a-z0-9-]{0,22}[a-z0-9])?$/); +assertMatch("organization slug", orgSlug, /^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?$/); +assertMatch("port", port, /^[1-9][0-9]{1,4}$/); +assertMatch("database name", required(answers, "DATABASE_NAME"), /^[a-z_][a-z0-9_]*$/); +assertMatch("database user", required(answers, "DATABASE_USER"), /^[a-z_][a-z0-9_]*$/); +assertMatch("default role id", required(answers, "DEFAULT_ROLE_ID"), /^[a-z][a-z0-9-]*$/); +assertMatch("release id", required(answers, "RELEASE_ID"), /^[A-Za-z0-9._-]+$/); +if (!publicBaseUrl.startsWith("https://")) throw new Error("PUBLIC_BASE_URL must use https"); +for (const [label, value] of [ + ["WORKSPACE_ROOT", workspaceRoot], + ["DEPLOY_BASE", required(answers, "DEPLOY_BASE")], + ["DEPLOY_SSH_KEY", required(answers, "DEPLOY_SSH_KEY")], + ["CPH_BIN", required(answers, "CPH_BIN")], +]) { + if (!value.startsWith("/") || /\s/.test(value)) { + throw new Error(`${label} must be an absolute path without whitespace: ${value}`); + } +} +if (Number(port) > 65535) throw new Error(`invalid port: ${port}`); +if (Buffer.byteLength(workspaceRoot) > 16) { + throw new Error(`WORKSPACE_ROOT exceeds the 16-byte sandbox socket limit: ${workspaceRoot}`); +} + +const outputDir = resolve(outputPath); +await mkdir(outputDir, { recursive: true, mode: 0o700 }); +await chmod(outputDir, 0o700); + +const base = required(answers, "DEPLOY_BASE"); +const release = required(answers, "RELEASE_ID"); +const releaseHub = `${base}/releases/${release}/hub`; +const secretDir = `${base}/.secrets/${instanceId}`; +const envPath = `${secretDir}/platform.env`; +const keyringPath = `${secretDir}/secret-keyring.json`; +const unit = `cph-hub-${instanceId}.service`; + +const platformEnv = [ + "# Generated by hub/deploy/new_silo.sh. Install root:root mode 0600.", + envLine("NODE_ENV", "production"), + envLine("DATABASE_URL", databaseUrl), + envLine("HUB_SILO_ORGANIZATION_ID", orgId), + envLine("HUB_SYSTEMD_UNIT", unit), + envLine("CPH_BIN", required(answers, "CPH_BIN")), + envLine("HOST", "127.0.0.1"), + envLine("PORT", port), + envLine("HUB_PROJECT_WORKSPACE_ROOT", workspaceRoot), + envLine("HUB_PUBLIC_BASE_URL", publicBaseUrl), + envLine("HUB_SESSION_SECRET", required(answers, "HUB_SESSION_SECRET")), + envLine("HUB_AGENT_MAX_TURNS", required(answers, "MAX_TURNS")), + envLine("HUB_AGENT_MAX_CONCURRENT_RUNS", required(answers, "MAX_CONCURRENT_RUNS")), + envLine("HUB_AGENT_MAX_RUN_SECONDS", required(answers, "MAX_RUN_SECONDS")), + envLine("HUB_HTTP_BODY_LIMIT_BYTES", required(answers, "HTTP_BODY_LIMIT_BYTES")), + envLine("HUB_MAX_FILES_PER_MESSAGE", required(answers, "MAX_FILES_PER_MESSAGE")), + envLine("HUB_MAX_FILE_BYTES", required(answers, "MAX_FILE_BYTES")), + envLine("HUB_HTTP_REQUESTS_PER_MINUTE", required(answers, "HTTP_REQUESTS_PER_MINUTE")), + envLine("HUB_FEISHU_EVENTS_PER_MINUTE", required(answers, "FEISHU_EVENTS_PER_MINUTE")), + envLine("HUB_FEISHU_LISTENER_ENABLED", "true"), + envLine("CPH_SANDBOX_EXTRA_DENY_READ", `${envPath}:${keyringPath}`), + "", +].join("\n"); + +const bootstrap = { + organization: { + id: orgId, + slug: orgSlug, + name: required(answers, "ORGANIZATION_NAME"), + }, + owner: { + openId: required(answers, "OWNER_OPEN_ID"), + displayName: required(answers, "OWNER_DISPLAY_NAME"), + }, + feishu: { + appId: required(answers, "FEISHU_APP_ID"), + appSecret: required(answers, "FEISHU_APP_SECRET"), + botOpenId: required(answers, "FEISHU_BOT_OPEN_ID"), + }, + provider: { + providerId: required(answers, "PROVIDER_ID"), + baseUrl: required(answers, "PROVIDER_BASE_URL"), + authToken: required(answers, "PROVIDER_AUTH_TOKEN"), + }, + teams: [], +}; +const ownerUnionId = optional(answers, "OWNER_UNION_ID"); +if (ownerUnionId) bootstrap.owner.unionId = ownerUnionId; + +const deployEnv = [ + "# Source this file locally before deploy_platform.sh (contains no app/provider secrets).", + envLine("PLATFORM_DEPLOY_HOST", required(answers, "DEPLOY_HOST")), + envLine("PLATFORM_DEPLOY_SSH_KEY", required(answers, "DEPLOY_SSH_KEY")), + envLine("PLATFORM_DEPLOY_USER", required(answers, "DEPLOY_USER")), + envLine("PLATFORM_DEPLOY_PORT", required(answers, "DEPLOY_SSH_PORT")), + envLine("PLATFORM_DEPLOY_HUB_PORT", port), + envLine("PLATFORM_DEPLOY_BASE", base), + envLine("PLATFORM_DEPLOY_RELEASE", release), + envLine("PLATFORM_DEPLOY_INSTANCE", instanceId), + envLine("PLATFORM_DEPLOY_WORKSPACE_ROOT", workspaceRoot), + envLine("PLATFORM_DEPLOY_MEMORY_MAX", required(answers, "MEMORY_MAX")), + envLine("PLATFORM_DEPLOY_CPU_QUOTA", required(answers, "CPU_QUOTA")), + envLine("PLATFORM_DEPLOY_TASKS_MAX", required(answers, "TASKS_MAX")), + envLine("PLATFORM_DEPLOY_HEALTH_URL", `http://127.0.0.1:${port}/api/healthz`), + "", +].join("\n"); + +const nginx = `# Install as /etc/nginx/conf.d/${instanceId}.conf after obtaining TLS certificates.\nserver {\n listen 80;\n server_name ${domain};\n return 301 https://$host$request_uri;\n}\n\nserver {\n listen 443 ssl http2;\n server_name ${domain};\n\n ssl_certificate /etc/letsencrypt/live/${domain}/fullchain.pem;\n ssl_certificate_key /etc/letsencrypt/live/${domain}/privkey.pem;\n\n location / {\n proxy_pass http://127.0.0.1:${port};\n proxy_http_version 1.1;\n proxy_set_header Host $host;\n proxy_set_header X-Real-IP $remote_addr;\n proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;\n proxy_set_header X-Forwarded-Proto https;\n }\n}\n`; + +const defaultRolePrompt = `你是一位教育机构智能助手,服务于学校和教培机构的日常管理与协作。 +## 核心能力 +**教务管理**:熟悉排课调课、班级管理、课时统计、师资调度等常见管理流程。 +**教研支持**:理解课程设计、教案编写、例题/变式题、定理证明、随堂练习、阶段性测验、知识点拆解、大纲对标等教学业务概念。 +**数字化操作**:熟练使用shell命令行进行文件管理、批量处理、脚本编写、文本处理(grep/sed/awk等)、自动化任务;能协助处理Markdown、Typst等文档、表格数据、格式转换等。 +## 行为准则 +- **实事求是**:不确定时明确说明,能力边界外的任务如实告知,不为完成目标而编造信息或勉强输出 +- **简洁优先**:直接给出答案或方案,省略铺垫和客套 +- **按需展开**:仅在问题复杂或用户明确要求时提供详细说明 +- **结构清晰**:多步骤内容使用编号,便于飞书阅读 +- **可操作性**:涉及操作时给出具体命令或步骤 +保持专业、高效,像一位熟悉业务的同事。 +`; + +const operate = `# ${instanceId} deployment commands\n\nThis file contains no application/provider secrets. Run commands deliberately; do not source answers.env.\n\n## 1. DNS and Feishu\n\n- Point \`${domain}\` to \`${required(answers, "DEPLOY_HOST")}\`.\n- In the Feishu app, configure redirect URL: \`${publicBaseUrl}/auth/feishu/callback\`.\n- Enable the required bot/message/contact permissions and event subscription described in the customer setup document.\n\n## 2. Database (run as the PostgreSQL administrator)\n\nCreate one role and one database for this Silo. The password is in \`answers.env\`; use an interactive client or a protected SQL file, never a command-line argument.\n\n\`\`\`sql\nCREATE ROLE ${required(answers, "DATABASE_USER")} LOGIN PASSWORD '';\nCREATE DATABASE ${required(answers, "DATABASE_NAME")} OWNER ${required(answers, "DATABASE_USER")};\n\`\`\`\n\n## 3. Publish the immutable release\n\nThe normal deploy script seeds the first-instance secrets and exits 78. That exit is expected only on this first pass.\n\n\`\`\`sh\nset -a; . ./deploy.env; set +a\nbash hub/deploy/deploy_platform.sh\n\`\`\`\n\n## 4. Install the generated secrets\n\nCopy \`platform.env\` and \`bootstrap.json\` to the server through a protected channel. On the server:\n\n\`\`\`sh\ninstall -d -o root -g root -m 0700 '${secretDir}'\ninstall -o root -g root -m 0600 platform.env '${envPath}'\ninstall -o root -g root -m 0600 bootstrap.json '/root/${instanceId}-bootstrap.json'\n# Copy ${keyringPath} to separate off-host recovery storage before continuing.\n\`\`\`\n\n## 5. Migrate, install the stopped service, and bootstrap\n\n\`\`\`sh\nset -a; . '${envPath}'; set +a\nnode '${releaseHub}/node_modules/prisma/build/index.js' migrate deploy --schema '${releaseHub}/prisma/schema.prisma'\nBASE='${base}' HUB_DIR='${releaseHub}' INSTANCE_ID='${instanceId}' WORKSPACE_ROOT='${workspaceRoot}' PORT='${port}' MEMORY_MAX='${required(answers, "MEMORY_MAX")}' CPU_QUOTA='${required(answers, "CPU_QUOTA")}' TASKS_MAX='${required(answers, "TASKS_MAX")}' bash '${releaseHub}/deploy/install_service.sh'\nnode '${releaseHub}/dist/deployment/bootstrap-silo-cli.js' --config-file '/root/${instanceId}-bootstrap.json' --keyring-file '${keyringPath}'\nrm -f '/root/${instanceId}-bootstrap.json'\n\`\`\`\n\n## 6. Runtime role and skills\n\nRuntime role/skill configuration is intentionally separate from the release. Follow NEW_SILO_RUNBOOK.md, staging each approved skill directory outside the immutable release, then use \`agent_config.sh\`. Do not silently copy a local personal skill collection.\n\n## 7. Nginx, start, and acceptance\n\nInstall \`nginx.conf\`, run \`nginx -t\`, reload Nginx, then:\n\n\`\`\`sh\nsystemctl start '${unit}'\nsystemctl is-active '${unit}'\ncurl --fail --silent --show-error 'http://127.0.0.1:${port}/api/healthz'\ncurl --fail --silent --show-error '${publicBaseUrl}/api/healthz'\njournalctl -u '${unit}' --since '-10 min' --no-pager\n\`\`\`\n\nAcceptance requires: Feishu OAuth completes, OWNER and a non-OWNER can @bot, a second group creates a separate project/session, Typst works when that skill is enabled, and restart preserves session cursor. Then take the first off-host backup.\n\n## 8. Later releases\n\nAfter the instance exists, source \`deploy.env\` with the new release id and run \`deploy_platform.sh\`. Never reuse another org's database, secret directory, workspace root, port, service identity, domain, Feishu app, or provider credential.\n`; + +const selectedSkills = optional(answers, "APPROVED_SKILLS"); +const roleInstructions = ` +## Appendix: default runtime role + +Copy \`default-role-prompt.md\` to \`/root/${instanceId}-default-role-prompt.md\`. +After installing each reviewed skill with \`agent_config.sh install-skill\`, run: + +\`\`\`sh +INSTANCE_ID=${JSON.stringify(instanceId)} ENV_FILE=${JSON.stringify(envPath)} HUB_DIR=${JSON.stringify(releaseHub)} bash ${JSON.stringify(`${releaseHub}/deploy/agent_config.sh`)} upsert-role --organization ${JSON.stringify(orgId)} --role ${JSON.stringify(required(answers, "DEFAULT_ROLE_ID"))} --label ${JSON.stringify(required(answers, "DEFAULT_ROLE_LABEL"))} --model ${JSON.stringify(required(answers, "DEFAULT_MODEL"))} --system-prompt-file ${JSON.stringify(`/root/${instanceId}-default-role-prompt.md`)} --tools-json null +${selectedSkills ? `INSTANCE_ID=${JSON.stringify(instanceId)} ENV_FILE=${JSON.stringify(envPath)} HUB_DIR=${JSON.stringify(releaseHub)} bash ${JSON.stringify(`${releaseHub}/deploy/agent_config.sh`)} set-role-skills --organization ${JSON.stringify(orgId)} --role ${JSON.stringify(required(answers, "DEFAULT_ROLE_ID"))} --skills ${JSON.stringify(selectedSkills)} +` : "# No skills selected in the wizard; set-role-skills remains an explicit operator step.\n"}INSTANCE_ID=${JSON.stringify(instanceId)} ENV_FILE=${JSON.stringify(envPath)} HUB_DIR=${JSON.stringify(releaseHub)} bash ${JSON.stringify(`${releaseHub}/deploy/agent_config.sh`)} list --organization ${JSON.stringify(orgId)} +\`\`\` + +Changing a role's model, system prompt, tools or skill selection archives that +role's existing sessions by design. Finish this setup before inviting Alpha users. +`; + +async function privateWrite(name, contents) { + const path = resolve(outputDir, name); + await writeFile(path, contents, { encoding: "utf8", mode: 0o600 }); + await chmod(path, 0o600); +} + +await privateWrite("platform.env", platformEnv); +await privateWrite("bootstrap.json", `${JSON.stringify(bootstrap, null, 2)}\n`); +await privateWrite("deploy.env", deployEnv); +await privateWrite("nginx.conf", nginx); +await privateWrite("default-role-prompt.md", defaultRolePrompt); +await privateWrite("OPERATE.md", `${operate}${roleInstructions}`); + +console.log(`Rendered private Silo bundle: ${outputDir}`);