fix(landing): make the site indexable — blog shells, sitemap, robots, share card (#51)

Co-authored-by: Alice <alice@prismshadow.com>
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-07-23 22:23:06 +08:00
committed by GitHub
parent d163cf058a
commit aa6b3e710e
8 changed files with 332 additions and 9 deletions
+17 -2
View File
@@ -7,15 +7,30 @@
<meta name="theme-color" content="#030712" media="(prefers-color-scheme: dark)" />
<meta
name="description"
content="PenguinHarness — Efficient Self-Improving Harness for Everyone. Open-source infrastructure that builds AI agents for you, from automatic agent construction to recursive self-improvement."
content="A zero-code CLI and Web UI for building AI agents, connected to 1,000+ models. Open source, local-first, and recursively self-improving."
/>
<!-- Self-referencing URLs for the home page: scripts/postbuild.mjs rewrites both per
route as it stamps out the blog shells, and drops the canonical from 404.html.
Absolute because rel=canonical and Open Graph both ignore relative URLs, so
these name the production origin whatever BASE_PATH a build uses. -->
<link rel="canonical" href="https://penguin.ooo/" />
<meta property="og:url" content="https://penguin.ooo/" />
<meta property="og:type" content="website" />
<meta property="og:site_name" content="PenguinHarness" />
<meta property="og:title" content="PenguinHarness" />
<meta
property="og:description"
content="Efficient Self-Improving Harness for Everyone — open-source, local-first infrastructure that builds AI agents for you."
/>
<meta name="twitter:card" content="summary" />
<meta property="og:image" content="https://penguin.ooo/og-cover.png" />
<meta property="og:image:width" content="1200" />
<meta property="og:image:height" content="630" />
<meta
property="og:image:alt"
content="PenguinHarness — Efficient Self-Improving Harness for Everyone"
/>
<meta name="twitter:card" content="summary_large_image" />
<meta name="twitter:image" content="https://penguin.ooo/og-cover.png" />
<link rel="icon" type="image/svg+xml" href="./penguin-logo.svg" />
<title>PenguinHarness — Efficient Self-Improving Harness for Everyone</title>
<!-- Apply the stored/system theme before first paint so the page doesn't flash the
Binary file not shown.

After

Width:  |  Height:  |  Size: 104 KiB

+8
View File
@@ -0,0 +1,8 @@
# PenguinHarness — https://penguin.ooo/ (landing at the root, docs under /docs/).
# Everything here is public; nothing needs to be kept out of an index.
User-agent: *
Allow: /
# Absolute by spec, and per-origin anyway, so the production domain is hardcoded even
# though builds accept a SITE_ORIGIN override. scripts/build-site.mjs writes the file.
Sitemap: https://penguin.ooo/sitemap.xml
@@ -0,0 +1,32 @@
/**
* Render the social share card referenced by index.html's og:image / twitter:image.
*
* Same pattern as capture-game-mockup.mjs: og-cover.html is a static, dependency-free
* mockup, captured at exactly 1200x630 into packages/landing/public/og-cover.png —
* public/ rather than src/assets/ because the URL has to be stable and absolute for
* the platforms that fetch it, not a hashed bundle asset. PNG, not WebP: WeChat and a
* few older unfurlers still ignore WebP previews.
*
* The output is committed, so this only needs re-running when og-cover.html changes.
* Prereqs: Playwright's chromium only. Run: `node scripts/capture-og-cover.mjs`.
*/
import { writeFileSync } from "node:fs";
import path from "node:path";
import { fileURLToPath, pathToFileURL } from "node:url";
import { chromium } from "@playwright/test";
const HERE = path.dirname(fileURLToPath(import.meta.url));
const PAGE = pathToFileURL(path.join(HERE, "og-cover.html")).href;
const OUT = path.resolve(HERE, "../public/og-cover.png");
const browser = await chromium.launch();
const page = await browser.newPage({ viewport: { width: 1200, height: 630 } });
await page.goto(PAGE);
// The logo is an <img>; screenshotting before it decodes yields a card with a hole in it.
await page.waitForFunction(() => {
const img = document.querySelector("img.logo");
return img !== null && img.complete && img.naturalWidth > 0;
});
writeFileSync(OUT, await page.screenshot());
await browser.close();
console.log(`[og-cover] wrote ${OUT}`);
+125
View File
@@ -0,0 +1,125 @@
<!doctype html>
<!--
Source for the social share card (public/og-cover.png), rendered by
capture-og-cover.mjs. Static and dependency-free, like the other mockups here:
edit this file, re-run the script, commit the PNG.
Sized 1200x630 — the ratio X, Facebook, LinkedIn, Discord and WeChat all crop
from. Colors are the landing page's dark theme: gray-950 behind the brand-500 and
sky-500 glows the NeonBackground drifts across the real site.
-->
<html lang="en">
<head>
<meta charset="UTF-8" />
<style>
html,
body {
margin: 0;
width: 1200px;
height: 630px;
font-family:
ui-sans-serif,
system-ui,
-apple-system,
"Segoe UI",
Roboto,
"PingFang SC",
"Microsoft YaHei",
sans-serif;
}
.card {
position: relative;
width: 1200px;
height: 630px;
overflow: hidden;
background: #030712;
color: #f9fafb;
}
/* Same two glows as .neon-blob-a / .neon-blob-b, without the drift animation. */
.glow {
position: absolute;
border-radius: 9999px;
filter: blur(90px);
opacity: 0.22;
}
.glow-a {
top: -180px;
left: -140px;
width: 700px;
height: 700px;
background: radial-gradient(circle, #4285f4 0%, transparent 66%);
}
.glow-b {
right: -200px;
bottom: -260px;
width: 760px;
height: 760px;
background: radial-gradient(circle, #0ea5e9 0%, transparent 66%);
}
.inner {
position: relative;
display: flex;
height: 100%;
flex-direction: column;
justify-content: space-between;
box-sizing: border-box;
padding: 72px 76px;
}
.logo {
width: 108px;
height: 108px;
}
h1 {
margin: 0;
font-size: 82px;
font-weight: 700;
letter-spacing: -0.025em;
line-height: 1;
}
.tagline {
margin: 22px 0 0;
font-size: 34px;
font-weight: 500;
line-height: 1.25;
color: #8ab4f8;
}
.foot {
display: flex;
align-items: center;
justify-content: space-between;
border-top: 1px solid rgba(148, 163, 184, 0.22);
padding-top: 28px;
font-size: 25px;
color: #cbd5e1;
}
.facts span + span::before {
content: "·";
margin: 0 14px;
color: #64748b;
}
.domain {
font-weight: 600;
color: #8ab4f8;
}
</style>
</head>
<body>
<div class="card">
<div class="glow glow-a"></div>
<div class="glow glow-b"></div>
<div class="inner">
<img class="logo" src="../public/penguin-logo.svg" alt="" />
<div>
<h1>PenguinHarness</h1>
<p class="tagline">Efficient Self-Improving Harness for Everyone</p>
</div>
<div class="foot">
<div class="facts">
<span>Zero-code CLI &amp; Web UI</span><span>1,000+ models</span><span>Apache-2.0</span>
</div>
<div class="domain">penguin.ooo</div>
</div>
</div>
</div>
</body>
</html>
+50 -6
View File
@@ -1,13 +1,57 @@
/**
* Post-build step for GitHub Pages: copy index.html to 404.html so deep links
* (/blog/xxx) served by Pages' 404 fallback still boot the SPA router, and add
* .nojekyll so Pages serves the dist verbatim without Jekyll processing.
* Post-build step for GitHub Pages, which serves static files only:
*
* - A shell per blog route (dist/blog/index.html, dist/blog/<slug>/index.html), the
* same trick the docs package uses. Without them Pages answers every blog URL from
* 404.html: the SPA still boots and the page looks right, but the response carries
* HTTP 404, so crawlers drop it and the posts never get indexed. Each shell also
* carries its own canonical/og:url, which is what a crawler that does not run JS
* sees — the SPA cannot rewrite them in time to matter.
* - dist/404.html for paths that really do not exist (and for /docs/* misses, which the
* landing router hands to the docs index). Its canonical is stripped: claiming the
* home page as the canonical of an unknown URL would ask search engines to fold
* every typo into it.
* - .nojekyll so Pages serves the dist verbatim without Jekyll processing.
*/
import { copyFileSync, writeFileSync } from "node:fs";
import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
import { dirname, join } from "node:path";
import { fileURLToPath } from "node:url";
import { absoluteUrl, blogRoutes } from "./site-routes.mjs";
const dist = join(dirname(fileURLToPath(import.meta.url)), "..", "dist");
copyFileSync(join(dist, "index.html"), join(dist, "404.html"));
const shell = readFileSync(join(dist, "index.html"), "utf8");
// Loose enough to survive Vite's HTML rewriting, strict enough to hit only these two
// tags. Both must exist in index.html — if a future edit drops them, fail the build
// rather than silently shipping every route with the home page's canonical.
const CANONICAL = /<link\s+rel="canonical"[^>]*>/;
const OG_URL = /<meta\s+property="og:url"[^>]*>/;
for (const [name, pattern] of [
["canonical", CANONICAL],
["og:url", OG_URL],
]) {
if (!pattern.test(shell)) {
throw new Error(`[postbuild] no ${name} tag in dist/index.html — index.html must declare one`);
}
}
/** The shell with its self-referencing URLs pointed at `route`. */
function shellFor(route) {
const url = absoluteUrl(route);
return shell
.replace(CANONICAL, `<link rel="canonical" href="${url}" />`)
.replace(OG_URL, `<meta property="og:url" content="${url}" />`);
}
const routes = blogRoutes();
for (const { route } of routes) {
const dir = join(dist, route);
mkdirSync(dir, { recursive: true });
writeFileSync(join(dir, "index.html"), shellFor(route));
}
writeFileSync(join(dist, "404.html"), shell.replace(CANONICAL, ""));
writeFileSync(join(dist, ".nojekyll"), "");
console.log("[postbuild] wrote dist/404.html and dist/.nojekyll");
console.log(
`[postbuild] wrote ${routes.length} blog route shells, dist/404.html and dist/.nojekyll`,
);
+75
View File
@@ -0,0 +1,75 @@
/**
* The deployed site's public routes, derived from the same Markdown filenames the two
* routers read (landing: content/blog/<slug>.<lang>.md, docs: content/<slug>.<lang>.md).
*
* Two consumers share this list, and they have to agree: the landing postbuild writes
* one static shell per route (so GitHub Pages answers 200 instead of falling back to
* 404.html), and build-site.mjs writes sitemap.xml for the assembled tree. A route that
* gets a shell belongs in the sitemap, and a sitemap URL that has no shell is a 404.
*/
import { readdirSync, readFileSync } from "node:fs";
import { dirname, join } from "node:path";
import { fileURLToPath } from "node:url";
/** Production origin, no trailing slash. Override for a staging deploy. */
export const SITE_ORIGIN = (process.env.SITE_ORIGIN ?? "https://penguin.ooo").replace(/\/+$/, "");
/** Deploy subpath, no trailing slash ("" at the domain root, "/repo" under a Pages subpath). */
export const BASE_PATH = (process.env.BASE_PATH ?? "/").replace(/\/+$/, "");
/** Absolute URL for a site-root-relative route path. */
export function absoluteUrl(route) {
return `${SITE_ORIGIN}${BASE_PATH}${route}`;
}
const HERE = dirname(fileURLToPath(import.meta.url));
const BLOG_DIR = join(HERE, "..", "content", "blog");
const DOCS_DIR = join(HERE, "..", "..", "docs", "content");
/** `<slug>.<lang>.md` -> slug, deduplicated across the two languages, sorted. */
function slugsIn(dir) {
const slugs = new Set();
for (const file of readdirSync(dir)) {
const slug = /^(.+)\.(zh|en)\.md$/.exec(file)?.[1];
if (slug !== undefined) slugs.add(slug);
}
return [...slugs].sort();
}
/** Newest `date:` in a post's frontmatter across its language variants, if any. */
function newestDate(dir, slug) {
let newest;
for (const lang of ["en", "zh"]) {
let raw;
try {
raw = readFileSync(join(dir, `${slug}.${lang}.md`), "utf8");
} catch {
continue; // A post exists in one language only; the other is a legitimate miss.
}
const frontmatter = /^---\r?\n([\s\S]*?)\r?\n---/.exec(raw)?.[1] ?? "";
const date = /^date:\s*"?(\d{4}-\d{2}-\d{2})"?\s*$/m.exec(frontmatter)?.[1];
if (date !== undefined && (newest === undefined || date > newest)) newest = date;
}
return newest;
}
/**
* Routes served by the landing SPA. Trailing slashes are deliberate: Pages serves
* `<route>/index.html` and 301s the slashless form to it, so the slash form is the URL
* that actually answers 200 — the one canonical tags and the sitemap should name.
* The home page is excluded; Vite's own index.html already sits at the dist root.
*/
export function blogRoutes() {
return [
{ route: "/blog/" },
...slugsIn(BLOG_DIR).map((slug) => ({
route: `/blog/${slug}/`,
lastmod: newestDate(BLOG_DIR, slug),
})),
];
}
/** Routes served by the docs SPA, whose own postbuild already writes their shells. */
export function docsRoutes() {
return [{ route: "/docs/" }, ...slugsIn(DOCS_DIR).map((slug) => ({ route: `/docs/${slug}/` }))];
}
+25 -1
View File
@@ -11,9 +11,10 @@
* single artifact — one GitHub Pages deployment hosts both sites.
*/
import { execSync } from "node:child_process";
import { cpSync, rmSync } from "node:fs";
import { cpSync, rmSync, writeFileSync } from "node:fs";
import { dirname, join } from "node:path";
import { fileURLToPath } from "node:url";
import { absoluteUrl, blogRoutes, docsRoutes } from "../packages/landing/scripts/site-routes.mjs";
const root = join(dirname(fileURLToPath(import.meta.url)), "..");
const base = process.env.BASE_PATH ?? "/";
@@ -34,4 +35,27 @@ const landingDist = join(root, "packages", "landing", "dist");
const docsTarget = join(landingDist, "docs");
rmSync(docsTarget, { recursive: true, force: true });
cpSync(join(root, "packages", "docs", "dist"), docsTarget, { recursive: true });
// sitemap.xml goes in last, because only here do both dists exist: it is the one file
// that has to name routes from both sites. It is also the only way a crawler learns
// those routes — both pages ship an empty <div id="root"> and build their navigation
// in JS, so following links is not an option.
const routes = [{ route: "/" }, ...blogRoutes(), ...docsRoutes()];
const sitemap = [
'<?xml version="1.0" encoding="UTF-8"?>',
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
...routes.map(({ route, lastmod }) =>
[
" <url>",
` <loc>${absoluteUrl(route)}</loc>`,
...(lastmod === undefined ? [] : [` <lastmod>${lastmod}</lastmod>`]),
" </url>",
].join("\n"),
),
"</urlset>",
"",
].join("\n");
writeFileSync(join(landingDist, "sitemap.xml"), sitemap);
console.log(`\n[build-site] assembled site at ${landingDist} (docs under /docs/)`);
console.log(`[build-site] wrote sitemap.xml (${routes.length} URLs)`);