From 1815d91cdf2f8bcbb9bd2f1abd1341c5c4e538d4 Mon Sep 17 00:00:00 2001 From: naudachu Date: Mon, 10 Aug 2026 16:29:55 +0500 Subject: [PATCH] feat: discussion artifacts as wiki pages, in two new layers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A discussion leaves behind a directory of markdown somewhere outside this repo, and the only durable home for it is the Gitea wiki. Getting it there by hand means re-deriving the same three things every time: what each file should be called, where it goes, and whether the page already exists. Two skills wrap that, along the split the repo already uses. `skills/page` is domain, offline, stdlib-only, and knows nothing about Gitea. It imports a directory into a space under `tmp/wiki/`, titles every file, records the result in `.pages.json`, and writes the index. `skills/wiki` is the bridge — `wikimap.py` translates, and the transport is `_gitea.py`, the same one the issue side uses. There is no second transport, and `tea` has no wiki subcommand to offer one. The Gitea wiki is flat, and that fact shapes everything There are no directories. A title of `a/b` is stored as one file named `a%2Fb.md`, and Gitea escapes it by rules of its own: space becomes `-`, `/` becomes `%2F`, and a literal `-` forces a trailing `.-` marker so the two stay distinct. `Chain decisions — DC` under two levels of prefix comes back as `Simple-Chains%2FParked%2FChain-decisions-%E2%80%94-DC`. So `sub_url` is the identity, it is read back from whatever the API returned, and it is never constructed. One built by hand that is almost right does not fail — it creates a second page and abandons the first. And a real subdirectory committed into a wiki's git repository is a ghost: the file exists, the API and the web UI do not see it. `folder/page.md` in this repo's own wiki is one. Nothing here clones a wiki repo. A title is a decision, not a derivation Titles come from the first heading, because there is no mechanical route from `03-q-01-do-we-know-the-chain-participant-by-name.md` to `Q-01. Do We Know the Chain Participant by Name`. But they are derived exactly once. A re-import replaces bodies and keeps titles, so editing a heading cannot rename a published page — which would not rename it, it would publish a second one. `--retitle` opts in. It finds the prior entry by `source` rather than by path, because the path is derived from the title and a retitle moves it; looked up by path the page would read as new and the next push would duplicate it. The old file goes, `sub_url` comes along, and `pushed` is cleared — a rename can leave the body byte-identical, and push decides by body hash alone, so a stale hash would skip the rename forever. Ordering is a `NN-` file-name prefix and never reaches the title. `00-` means "this is the directory's own page", and that page is named for the directory, not for its own heading: a child's title has to extend its parent's exactly, and `ideas/00-intro.md` opens with "Ideas for chain business requirements". Path collisions are reported and never resolved. Picking a winner is how a discussion loses a document. The index is navigation, not decoration Nothing draws a tree from flat titles. `page_index.py` writes one as an ordinary page, nested by title depth rather than by manifest path order — those disagree, since on disk `Top/System.md` sorts before `Top/Ideas/Scale.md` while in the hierarchy System is a child and Scale a grandchild. A parent with no page of its own still gets a node, so its children are not hidden. Links use `sub_url` when there is one and Gitea's `[[Title|label]]` syntax when there is not, so the order is push, rebuild, push. The same stances as the issue store, for the same reasons Pull overwrites, push is additive and never deletes, change detection is one hash and there is no drift model. A page with no `sub_url` has never been published, and that is a durable state. Issues gain a `wiki:` field holding page titles — titles, not URLs, so the reference stays in the domain. It already round-trips as a foreign key; this documents it. Verified against a live Gitea 1.26.1: create, update with a message, unchanged-skip, prefix-filtered pull, byte-identical round trip, and the per-page revision history carrying the operator's own words. The probe pages were deleted afterwards. Co-Authored-By: Claude Opus 5 (1M context) --- .claude-plugin/plugin.json | 6 +- AGENTS.md | 74 +++- agents/tea-runner.md | 23 +- skills/issue/references/format.md | 2 + skills/page/SKILL.md | 88 +++++ skills/page/references/pages.md | 173 ++++++++++ skills/page/scripts/page.py | 523 +++++++++++++++++++++++++++++ skills/page/scripts/page_import.py | 159 +++++++++ skills/page/scripts/page_index.py | 87 +++++ skills/page/scripts/page_ls.py | 88 +++++ skills/wiki/SKILL.md | 119 +++++++ skills/wiki/scripts/wiki_ls.py | 77 +++++ skills/wiki/scripts/wiki_pull.py | 127 +++++++ skills/wiki/scripts/wiki_push.py | 152 +++++++++ skills/wiki/scripts/wikimap.py | 124 +++++++ tests/test_page_tree.py | 523 +++++++++++++++++++++++++++++ 16 files changed, 2319 insertions(+), 26 deletions(-) create mode 100644 skills/page/SKILL.md create mode 100644 skills/page/references/pages.md create mode 100644 skills/page/scripts/page.py create mode 100644 skills/page/scripts/page_import.py create mode 100644 skills/page/scripts/page_index.py create mode 100644 skills/page/scripts/page_ls.py create mode 100644 skills/wiki/SKILL.md create mode 100644 skills/wiki/scripts/wiki_ls.py create mode 100644 skills/wiki/scripts/wiki_pull.py create mode 100644 skills/wiki/scripts/wiki_push.py create mode 100644 skills/wiki/scripts/wikimap.py create mode 100644 tests/test_page_tree.py diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 7961eb6..d5b21fd 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,10 +1,10 @@ { "name": "tea", - "description": "Gitea issues as local markdown, cleanly layered: /tea:issue works on issues offline (format, validation, dependency graph), /tea:sync moves them to and from Gitea, /tea:use is the CLI reference, the tea-runner subagent executes the scripts on a cheap model, and a PreToolUse hook blocks any command that would touch Gitea without the operator-pinned login.", - "version": "2.1.0", + "description": "Gitea issues and wiki pages as local markdown, cleanly layered: /tea:issue works on issues offline (format, validation, dependency graph) and /tea:page turns a discussion's artifacts into a titled, ordered page tree; /tea:sync and /tea:wiki move each to and from Gitea; /tea:use is the CLI reference, the tea-runner subagent executes the scripts on a cheap model, and a PreToolUse hook blocks any command that would touch Gitea without the operator-pinned login.", + "version": "2.2.0", "author": { "name": "naudachu" }, "license": "MIT", - "keywords": ["gitea", "cli", "git", "issues", "login-guard"] + "keywords": ["gitea", "cli", "git", "issues", "wiki", "login-guard"] } diff --git a/AGENTS.md b/AGENTS.md index 81909fb..8ded64f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -15,14 +15,18 @@ ## Layers -The hard rule of this repo. Knowledge flows one way only: +The hard rule of this repo. Two domains, two bridges, one transport, and +knowledge flows one way only: ``` skills/issue DOMAIN what an issue is: format, validation, dependency graph +skills/page DOMAIN what a page tree is: title <-> path, order, the index ▲ offline — no tracker, no network, stdlib imports only │ imports -skills/sync BRIDGE map.py md <-> Gitea JSON, pure functions, no I/O - _gitea.py login pin, tea api, pagination, filters +skills/sync BRIDGE map.py md <-> Gitea issue JSON, pure, no I/O + _gitea.py login pin, tea api, pagination, filters +skills/wiki BRIDGE wikimap.py md <-> Gitea wiki JSON, pure, no I/O + transport is _gitea.py — there is no second one skills/use REFERENCE tea CLI docs for everything that is not an issue skills/auth IDENTITY pin the login the whole tracker side runs under ▲ @@ -30,16 +34,19 @@ skills/auth IDENTITY pin the login the whole tracker side runs under agents/ EXECUTION tea-runner: runs the scripts, reports a receipt ``` -`skills/issue` never imports from `skills/sync`. Delete `skills/sync` and the -domain layer keeps working. The check is mechanical — every import under -`skills/issue/scripts/` is stdlib, and `subprocess` is not among them: +A domain never imports its bridge, and the two domains do not import each +other: delete `skills/sync` and issues still work, delete `skills/wiki` and page +trees still work, delete either domain and the other is untouched. The check is +mechanical — every import under a domain's `scripts/` is stdlib, and +`subprocess` is not among them: ```bash grep -rh '^import \|^from ' skills/issue/scripts/ | sort -u +grep -rh '^import \|^from ' skills/page/scripts/ | sort -u ``` -If a tracker concept (issue number, login, HTTP call, label color) shows up in -the domain layer, it is in the wrong place. +If a tracker concept (issue number, login, HTTP call, label color, `sub_url`, +`content_base64`) shows up in a domain layer, it is in the wrong place. ## Repo layout @@ -59,6 +66,20 @@ the domain layer, it is in the wrong place. - `scripts/_gitea.py` — transport: login pin, `tea api`, pagination, filters, label ids, the remote-id map - `scripts/pull.py`, `push.py`, `remote.py`, `comment.py` +- `skills/page` — a discussion's artifacts as a page tree (`/tea:page`), + entirely offline + - `references/pages.md` — canonical page-tree format; single source of truth + - `scripts/page.py` — domain module: title ↔ path, ordering, the manifest, + importing a directory of markdown, the index + - `scripts/page_import.py` — copy a directory of markdown into a space, + titling every file + - `scripts/page_index.py` — write the table-of-contents page + - `scripts/page_ls.py` — the tree, the titles, one sync-state tag per page +- `skills/wiki` — move page trees between a local space and a Gitea wiki + (`/tea:wiki`) + - `scripts/wikimap.py` — md ↔ Gitea wiki JSON, pure, no I/O + - `scripts/wiki_ls.py`, `wiki_pull.py`, `wiki_push.py` — transport is + `skills/sync/scripts/_gitea.py` - `skills/use` — `tea` CLI reference for everything that is not an issue (`/tea:use`); `references/tea/` holds the command docs - `agents/tea-runner.md` — subagent on Haiku that executes the scripts and @@ -83,12 +104,12 @@ stdlib-only and the tests hold the same line. `skills/*/scripts/` are not packages, so a test that needs the domain module imports it with `sys.path.insert`. -**A test never touches `tmp/issues/`.** Anything that needs a store builds a -throwaway repository in a `tempfile.TemporaryDirectory()` — a `.git` marker, a -copy of the script layers, fixture issues — and runs the real scripts inside it -as subprocesses. That is the only way to test behavior that depends on where a -script is run from, and it keeps the developer's own store out of the blast -radius. +**A test never touches `tmp/issues/` or `tmp/wiki/`.** Anything that needs a +store builds a throwaway repository in a `tempfile.TemporaryDirectory()` — a +`.git` marker, a copy of the script layers, fixture issues or artifacts — and +runs the real scripts inside it as subprocesses. That is the only way to test +behavior that depends on where a script is run from, and it keeps the +developer's own store out of the blast radius. ## Local issue store @@ -113,3 +134,28 @@ plain grep works without a parser. - Pulling overwrites the body — a fetch, not a merge. - No drift tracking. `synced:` tells you how old your copy is; re-pull when it matters. + +## Local wiki cache + +`tmp/wiki//` (gitignored) holds page trees — a discussion's artifacts, +organized. Same stance as the issue store, resolved the same way from +`page.py`'s own location, with the same `--out` rule. + +- Identity is the **title**, and `/` inside it is the only hierarchy there is. + The Gitea wiki is flat: it escapes a title into one filename by rules of its + own (`space -> -`, `/ -> %2F`, a literal `-` forces a trailing `.-`). +- **`sub_url` is Gitea's address for a page and is never constructed.** It is + read back from the API and stored in `.pages.json`. One built by hand that is + almost right creates a second page instead of editing the first. +- **Never commit a subdirectory into a wiki's git repository.** Gitea does not + see it — the page exists on disk and nowhere in the API or the UI. Do not + clone the wiki repo to work in; use the scripts. +- A title is a decision, not a derivation. A re-import replaces bodies and + keeps titles, so editing a heading cannot silently rename a published page. + `--retitle` opts in, and the rename reaches the wiki on the next push. +- A page with no `sub_url` has never been published — a durable state, exactly + as `origin: local` is for an issue. +- Change detection is one hash (`pushed`). Pulling overwrites; pushing is + additive and never deletes. +- The `tea` CLI has no wiki subcommand. `tea api` is the only route, through + `_gitea.py`. diff --git a/agents/tea-runner.md b/agents/tea-runner.md index 46dd7ce..cc72fc6 100644 --- a/agents/tea-runner.md +++ b/agents/tea-runner.md @@ -32,9 +32,11 @@ Load the skill, do not remember the flags: - `/tea:sync` — `pull.py`, `push.py`, `comment.py`, `remote.py`, `labels.py` - `/tea:issue` — `issue_check.py`, `issue_tree.py`, `issue_index.py`, `issue_new.py`, `issue_ac.py` +- `/tea:wiki` — `wiki_ls.py`, `wiki_pull.py`, `wiki_push.py` +- `/tea:page` — `page_import.py`, `page_index.py`, `page_ls.py` -Invoke `Skill` with `tea:sync` or `tea:issue` at the start of the task, and use -the command table it gives you verbatim. The skill is the single source of +Invoke `Skill` with the one that owns the task at the start, and use the command +table it gives you verbatim. The skill is the single source of truth for the script surface; a flag you recall from another session is a guess. If the skill does not document a flag, it does not exist — report that instead of trying it. @@ -42,7 +44,8 @@ instead of trying it. ## Hard rules 1. **No raw `tea`.** Every tracker call goes through a script in - `skills/sync/scripts/`. The one exception is a diagnostic the skill itself + `skills/sync/scripts/` or `skills/wiki/scripts/`. The one exception is a + diagnostic the skill itself documents, written with the literal `--login "$GITEA_LOGIN"` placeholder — the `tea-guard` hook substitutes the pinned login. Never name a login. 2. **No writing to issue files.** You have no `Edit` and no `Write`. Scripts @@ -52,12 +55,14 @@ instead of trying it. only the items the caller named, by the number or the substring the caller gave. Whether a criterion is actually met is a judgement about content, and content is never yours. -3. **Push only what you were told to push.** `push.py` publishes to a tracker - other people read. Run it with the ids the caller named, or with the filter - the caller named. Never widen the set, never run a bare `push.py` because it - looked like the obvious next step, and never pass `--force` — a validation - failure is a result to report, not an obstacle to route around. -4. **Do not close, delete, or retitle anything** on either side. +3. **Push only what you were told to push.** `push.py` and `wiki_push.py` + publish to a tracker other people read. Run them with the ids, titles, or + filter the caller named. Never widen the set, never run a bare `push.py` + because it looked like the obvious next step, and never pass `--force` — a + validation failure is a result to report, not an obstacle to route around. + `wiki_push.py` needs `-m`; use the caller's words, never your own summary. +4. **Do not close, delete, or retitle anything** on either side. On the wiki + that means no `--retitle`: renaming a published page abandons the old one. 5. **One retry, maximum.** A command that fails twice is a finding. Do not permute flags looking for one that works. 6. **No payload dumps.** Never run `tea issues -o json`, never `cat` a pulled diff --git a/skills/issue/references/format.md b/skills/issue/references/format.md index ab7b1e2..a44f420 100644 --- a/skills/issue/references/format.md +++ b/skills/issue/references/format.md @@ -33,6 +33,7 @@ labels: [type/task, tech/sql] assignees: [naudachu] milestone: v0.2 depends: [migrate-schema] +wiki: [Simple Chains/Ideas/Chain core] origin: gitea branch: feat/wire-sqlc gitea: claude-skills/tea#42 @@ -54,6 +55,7 @@ url: https://git.noodles.cam/claude-skills/tea/issues/42 | `assignees` | domain | logins; may be empty | | `milestone` | domain | title, or `none` | | `depends` | domain | ids this issue depends on — **the authoritative graph** | +| `wiki` | domain | page **titles** this issue is written up in; may be empty. Titles, not URLs — a title is a name for a document and stays in this layer, a URL is tracker bookkeeping. `/tea:page` owns what those titles mean; `page_ls.py --titles` prints them | | `origin` | domain | `local`, or the name of a tracker this also lives in | | `gitea` | sync | the handle in that tracker: `owner/repo#N` | | `branch` | sync | the tracker's branch link (Gitea `ref`); push fills an empty one with the current git branch, and never overwrites a filled one | diff --git a/skills/page/SKILL.md b/skills/page/SKILL.md new file mode 100644 index 0000000..d9b3550 --- /dev/null +++ b/skills/page/SKILL.md @@ -0,0 +1,88 @@ +--- +name: page +description: Organize a discussion's artifacts into a named, ordered tree of wiki pages — import a directory of markdown, give every file a title, build the index, see what a space holds. Entirely offline; pages are local markdown files and need no wiki. Load when the user asks to turn notes/artifacts into wiki pages, organize or re-title a page tree, or rebuild a table of contents. For fetching from or publishing to a Gitea wiki, load /tea:wiki instead. +--- + +# /tea:page — discussion artifacts as a page tree + +A discussion produces artifacts wherever the discussion happened — a directory +of markdown with numbered files and subdirectories. This skill turns that into +a **space**: a named, ordered tree of pages with a manifest, living under +`tmp/wiki/`. + +**Nothing here touches the network.** No `tea`, no Gitea, no login. A space that +never leaves this machine is a finished thing, not a draft waiting for an +upload. Publishing is a separate, optional layer — `/tea:wiki`. + +Read [`references/pages.md`](references/pages.md) before importing or +re-titling. It is the single source of truth for titles, ordering, paths, the +manifest, and the index. + +## Identity: the title + +`Simple Chains/Ideas/Chain core`. The `/` is the only hierarchy there is — the +wiki this feeds is flat and has no directories. The local path is derived from +the title (`Simple-Chains/Ideas/Chain-core.md`); the reverse never happens. + +A title is chosen **once**, at import or at pull, and then it is a fact in the +manifest. Editing a heading does not rename a page. Renaming is `--retitle`, +and on a published page it orphans the old one. + +## Scripts + +All offline, all in `/scripts/`. + +| Script | What it does | +|---|---| +| `page_import.py --from DIR [--space S] [--prefix T]` | copy a directory of markdown into a space, titling every file | +| `page_index.py [--space S] [--prefix T]` | write the table-of-contents page — the navigation the flat wiki cannot provide | +| `page_ls.py [--space S] [--prefix T]` | the tree, the titles, and one sync-state tag per page | +| `page.py` | the domain module the others import — not a command | + +``` +tmp/wiki/claude-skills/tea/ a space + .pages.json the manifest — titles, order, sync bookkeeping + Simple-Chains.md the index page + Simple-Chains/Ideas.md title: Simple Chains/Ideas + Simple-Chains/Ideas/Chain-core.md title: Simple Chains/Ideas/Chain core +``` + +## The usual run + +```bash +python3 scripts/page_import.py \ + --from ~/proj/tmp/simple-chains \ + --space claude-skills/tea --prefix "Simple Chains" --dry-run +``` + +`--dry-run` first, always: it prints every path and the title it would get, and +that listing is the only chance to catch a heading that titles a page badly +before the name becomes a decision. Drop the flag to write. + +Then the index, then look at it: + +```bash +python3 scripts/page_index.py --space claude-skills/tea --prefix "Simple Chains" +python3 scripts/page_ls.py --space claude-skills/tea --prefix "Simple Chains" +``` + +`page_ls.py` tags each page `local` (never published), `synced` (published and +unchanged), or `ahead` (edited since it was published). `local` is a complete +state. + +## Where the cache is + +`/tmp/wiki` — **not** `tmp/wiki` relative to wherever you are +standing. The scripts resolve it by walking up from their own file to the +nearest `.git` or `AGENTS.md`, so they all see one cache no matter which +directory they are run from. + +`--out` overrides that and is taken **literally**: an absolute path is used as +given, a relative one stays relative to the current directory. + +## Re-importing is the normal refresh + +The discussion continues, the artifacts change, run the same import again. +Bodies are replaced, titles are kept, `sub_url` and the rest of the wiki +bookkeeping survive — so the next push updates the pages that already exist +instead of publishing a second copy of each. diff --git a/skills/page/references/pages.md b/skills/page/references/pages.md new file mode 100644 index 0000000..54fbadf --- /dev/null +++ b/skills/page/references/pages.md @@ -0,0 +1,173 @@ +# The page-tree format + +Canonical. Everything about how a discussion's artifacts become named, ordered, +navigable pages lives here. The scripts implement this document; when they +disagree, this document is right. + +## The one fact that shapes everything: the wiki is flat + +Gitea's wiki has no directories. It has a list of pages, each stored as one +file whose name Gitea escapes from the title: + +| title | file Gitea writes | `sub_url` | +|---|---|---| +| `Abstract Issue` | `Abstract-Issue.md` | `Abstract-Issue` | +| `zz-probe/child` | `zz-probe%2Fchild.-.md` | `zz-probe%2Fchild.-` | +| `Simple Chains/Parked/Chain decisions — DC` | `Simple-Chains%2FParked%2FChain-decisions-%E2%80%94-DC.md` | same, minus `.md` | + +Three rules are visible in that table, and all three are Gitea's to change: +space becomes `-`; `/` becomes `%2F`; a **literal** `-` in the title forces a +trailing `.-` marker so it stays distinguishable from a space. + +Two consequences run through the whole design. + +**Hierarchy lives in the title and nowhere else.** `/` inside a title is the +only nesting there is. A real subdirectory committed into the wiki's git +repository — `folder/page.md` — is invisible to the API and to the web UI. It +is a ghost file. Never create one. + +**`sub_url` is identity and is never constructed.** It is read back from +whatever the API returned and stored in the manifest. A hand-built one that is +almost right does not fail loudly; it creates a second page and abandons the +first. + +## The space + +``` +tmp/wiki/claude-skills/tea/ a SPACE + .pages.json the manifest + Simple-Chains.md title: Simple Chains (the index) + Simple-Chains/ + Ideas.md title: Simple Chains/Ideas + Ideas/ + Chain-core.md title: Simple Chains/Ideas/Chain core +``` + +A space is a directory holding a page tree and one manifest. Its name is +normally the `owner/repo` it syncs with, and to the domain layer that is an +opaque relative path — `--space docs` and `--space a/b/c` are equally valid. + +The path is `/tmp/wiki`, resolved from `page.py`'s own location and +not from the working directory. `--out` overrides it and is used exactly as +typed. Nothing creates a space as a side effect of a write: the scripts say so +on stderr when they make one. + +## The manifest + +`.pages.json`, one entry per page, keyed by the file's path inside the space. + +```json +{ + "space": "claude-skills/tea", + "pages": { + "Simple-Chains/Ideas/Chain-core.md": { + "title": "Simple Chains/Ideas/Chain core", + "order": 2, + "pushed": "9a1ab2e3bfd45f7c7ba323d9d8cd59642d6f0540", + "remote-updated": "2026-08-10T11:15:39Z", + "sha": "fc8ec1779d910850f49bfef60dd5a0e737bbdc8a", + "sub_url": "Simple-Chains%2FIdeas%2FChain-core", + "synced": "2026-08-10T11:15:39Z", + "url": "https://git.noodles.cam/…/wiki/Simple-Chains%2FIdeas%2FChain-core" + } + } +} +``` + +| key | owner | meaning | +|---|---|---| +| `title` | domain | the page's name; `/` is hierarchy | +| `order` | domain | sort key from a `NN-` file-name prefix; absent when there was none | +| `sub_url` | wiki | Gitea's address for the page — **the identity** | +| `pushed` | wiki | sha1 of the bytes last published; the whole of change detection | +| `sha` | wiki | the wiki commit the local copy came from | +| `synced` | wiki | when this copy was fetched or pushed | +| `url` | wiki | browser link | +| `remote-updated` | wiki | the page's last commit date in the wiki | + +The domain layer writes `title` and `order`, carries everything else through +load and save verbatim, and never reads it. A page with no `sub_url` has never +been published — a complete state, not a pending one, exactly as `origin: local` +is for an issue. + +## How a source file gets its title + +Applied at import, once. Three rules, in order: + +1. **`order 0`, or a file literally named `index` / `readme`, is the page for + the directory it sits in.** `ideas/00-intro.md` becomes `…/Ideas`, not a + child of it. Its title comes from the **directory name**, never from its own + heading — a child's title has to extend its parent's exactly, and that file + opens with "Ideas for chain business requirements", which no child would + ever be prefixed by. +2. **Otherwise the file's first markdown heading**, sanitized. It is what a + human wrote for a human: there is no mechanical route from + `03-q-01-do-we-know-the-chain-participant-by-name.md` to + `Q-01. Do We Know the Chain Participant by Name`. +3. **No heading: the file name**, made readable — `NN-` stripped, `-` and `_` + to spaces, first letter raised. Only the first letter: title-casing would + wreck `Q-01`, `sqlc`, and `APNs`. + +Sanitizing a title drops markdown markup (`` ` ``, `*`, `_` — a page list does +not render markdown) and turns `/` into `-`, because a slash inside a heading +would silently invent a level of hierarchy the author did not ask for. + +### A title is a decision, not a derivation + +Once a page is in the manifest its title stays put. Re-importing replaces the +body and leaves the title alone, so editing a heading cannot rename a page — +which matters because renaming a **published** page does not move it, it +creates a second one and orphans the first. `--retitle` opts into that +explicitly. + +The reverse direction does not exist. A path is derived from a title; a title +is never derived from a path. `02-chain-core` proves why: those dashes are +real, and undoing "space became dash" would eat them. + +## Ordering + +A leading `NN-` on a file name is sort order and nothing else — it never +reaches the title. `00` is special and means "this is the directory's own +page". Pages with an order sort before pages without one: an explicit `NN-` is +a decision, its absence is not. + +The wiki cannot hold ordering, so `order` is local-only and survives a pull. + +## Paths + +A path is one component per title segment, spaces to `-`, with characters a +shell has to quote dropped — apostrophes and quotes and commas. `Don't send to +this one` keeps its apostrophe in the title and loses it in +`Dont-send-to-this-one.md`. + +Two titles can land on one path. That is reported and never resolved +automatically: picking a winner is how a discussion loses a document. Rename a +source, or rename the page in the wiki, and run it again. + +## The index page + +The wiki will not draw a tree from titles, so an index page is the navigation, +not a nicety. `page_index.py` writes one as an ordinary page in the space — it +is pushed by the same command as everything else. + +Nesting follows the **titles**, not the manifest's path order; those two +disagree, because on disk `Simple-Chains/System.md` sorts before +`Simple-Chains/Ideas/Scale.md` while in the hierarchy System is a child and +Scale a grandchild. A parent with no page of its own still gets a node, so its +children are not hidden. + +Links: a published page is linked by its `sub_url`, the only address Gitea +guarantees. A page that has never been pushed gets Gitea's `[[Title|label]]` +wiki-link syntax, which resolves the escaping on the server at render time. +Rebuilding the index after a push upgrades those links to exact ones — so the +order is **push, rebuild the index, push again**. + +## What the sync does not do + +- **No merge.** A pull overwrites the local body. `synced` tells you how old + your copy is; re-pull when it matters. +- **No drift tracking.** `pushed` answers one question — is the local file + different from what was published — and answers it with a hash. +- **No deletes.** Pushing is additive. A page removed locally stays in the + wiki; removing a published page is an explicit act, done in the web UI or + with a `DELETE` through `/tea:use`. diff --git a/skills/page/scripts/page.py b/skills/page/scripts/page.py new file mode 100644 index 0000000..8067d8b --- /dev/null +++ b/skills/page/scripts/page.py @@ -0,0 +1,523 @@ +#!/usr/bin/env python3 +r""" +page.py — what a PAGE TREE is. The domain layer for wiki artifacts. + +Not a command; the module the other page scripts build on. It knows how a +directory of markdown becomes a named, ordered tree of pages, and it knows +NOTHING about any wiki: no Gitea, no `tea`, no logins, no HTTP, no `sub_url`. +The layering rule is mechanically checkable — every import in this directory is +stdlib, and `subprocess` is not among them: + + grep -rh '^import \|^from ' skills/page/scripts/ | sort -u + +Delete skills/wiki/ entirely and this layer keeps working: a discussion's +artifacts organized into a tree on this machine are a finished thing, not a +draft waiting for an upload. + + tmp/wiki/claude-skills/tea/ <- a SPACE + .pages.json <- the manifest + Simple-Chains/ + Ideas.md title: Simple Chains/Ideas + Ideas/ + Chain-core.md title: Simple Chains/Ideas/Chain core + +A space is a directory holding a page tree and one manifest. The space's name +("claude-skills/tea") is an opaque relative path to this module — it happens to +be an owner/repo pair, and this layer never learns that. + +Why a manifest at all +--------------------- +Because the wiki's own page identity is not derivable from a file path, and +guessing at it is how you get duplicate pages. The manifest is the record of +what each local file IS, written once at import or pull and never re-derived. + +Domain keys in a manifest entry are `title` and `order`. Everything else — +`sub_url`, `sha`, `synced`, `pushed` — is written by the wiki layer, carried +through load/save verbatim, and never read here. That passthrough is what lets +one manifest describe both a local-only tree and a published one without the +domain learning a second vocabulary. + +Titles +------ +The title is the identity that matters, and `/` inside it is the ONLY +hierarchy there is — the wiki this feeds has no directories. A local path is +derived from the title, never the reverse: + + title "Simple Chains/Ideas/Chain core" + path "Simple-Chains/Ideas/Chain-core.md" + +That direction is deliberate. Deriving a title back from a path would have to +undo `-`-for-space, and `02-chain-core` proves it cannot: the dashes there are +real. So a title is chosen ONCE, at import or at pull, and then it is a fact in +the manifest. Renaming is an explicit act, not a side effect of editing a +heading. + +Ordering +-------- +A leading `NN-` on a file name is sort order and nothing else — it never +reaches the title. `order 0` is special: it is the directory's own page, so +`ideas/00-intro.md` becomes the page "…/Ideas" rather than a child of it. +""" +import hashlib +import json +import os +import re + +# -------------------------------------------------------------------------- +# where the cache lives +# -------------------------------------------------------------------------- +# `/tmp/wiki`, absolute, resolved once at import — the same anchoring +# rule the issue store uses, and for the same reason: a script's own location is +# a fact about the installation, cwd is a fact about the last `cd`. Walking up +# from __file__ hands every script in both layers one answer no matter where it +# is invoked from. +# +# The twenty lines below are duplicated from the issue domain rather than +# imported from it. Two domains that do not know about each other is worth more +# than the duplication is worth saving: skills/page must keep working with +# skills/issue deleted, exactly as skills/issue keeps working with skills/sync +# deleted. + +STORE_PARTS = ("tmp", "wiki") + +# `.git` is a directory in a normal clone and a FILE in a worktree — hence +# exists(), not isdir(). AGENTS.md is the fallback for a plugin copied out of +# git; the agents-sync hook only ever puts one at a repository root. +REPO_MARKERS = (".git", "AGENTS.md") + +_HERE = os.path.dirname(os.path.abspath(__file__)) + +MANIFEST = ".pages.json" + +# Written here; read here. Everything else in an entry belongs to the wiki +# layer and is passed through untouched. +DOMAIN_KEYS = ("title", "order", "source") + + +def repo_root(start): + """Nearest ancestor of `start` (inclusive) carrying a repo marker, or None.""" + d = os.path.abspath(start) + while True: + if any(os.path.exists(os.path.join(d, m)) for m in REPO_MARKERS): + return d + parent = os.path.dirname(d) + if parent == d: + return None + d = parent + + +def store_root(start=None): + """Absolute path of the wiki cache root. + + `start` overrides the anchor so the resolution can be exercised against a + scratch tree. Outside a repository, cwd gets a turn, then the historical + cwd-relative location stands — made absolute so an error can name the + directory it really looked in.""" + for anchor in ([start] if start is not None else [_HERE, os.getcwd()]): + root = repo_root(anchor) + if root: + return os.path.join(root, *STORE_PARTS) + return os.path.abspath(os.path.join(*STORE_PARTS)) + + +WIKI_ROOT = store_root() + + +def space_root(space, root=None): + """Directory of one space. `space` is an opaque relative path — it may + contain `/` (it usually does) and is used as typed.""" + return os.path.join(root or WIKI_ROOT, *space.split("/")) + + +# -------------------------------------------------------------------------- +# names, titles, order +# -------------------------------------------------------------------------- +# Characters a title may not carry into a path. `/` is absent on purpose: it is +# the hierarchy separator and is split on before this ever applies. +_UNSAFE = re.compile(r'[\\:*?"<>|\x00-\x1f]+') +# Inline code in a heading is markup, not a name: `Inventory — \`P-NN\`` is a +# page called "Inventory — P-NN", and a page list does not render markdown. +_MARKUP = re.compile(r"[`*_]+") +# Dropped from a PATH but kept in a title. An apostrophe in "Don't send to this +# one" belongs in the name and does not belong in something a shell has to +# quote. +_PATH_NOISE = re.compile(r"['‘’\"“”,]+") +_DASHES = re.compile(r"-{2,}") +_ORDER = re.compile(r"^(\d+)[-_. ]+(.*)$") +_HEADING = re.compile(r"^\s{0,3}#{1,6}\s+(.+?)\s*#*\s*$") + + +def order_of(name): + """The `NN-` sort key on a file or directory name, or None. + + `00-intro.md` -> 0, `02-chain-core.md` -> 2, `handoff.md` -> None. Zero is + a real answer and not None; callers distinguish them.""" + m = _ORDER.match(strip_ext(name)) + return int(m.group(1)) if m else None + + +def strip_ext(name): + stem, ext = os.path.splitext(name) + return stem if ext.lower() in (".md", ".markdown") else name + + +def strip_order(name): + """`02-chain-core` -> `chain-core`; a name that is only digits is left + alone, because stripping it would leave nothing to call the page.""" + m = _ORDER.match(strip_ext(name)) + return m.group(2) if m and m.group(2) else strip_ext(name) + + +def title_from_name(name): + """Fallback title: the file or directory name made readable. + + `02-chain-core.md` -> `Chain core`. Only the first letter is raised — + title-casing would wreck `Q-01`, `sqlc`, `APNs`, and every other name that + already knows how it is spelled.""" + t = strip_order(name).replace("_", " ").replace("-", " ").strip() + t = re.sub(r"\s+", " ", t) + return t[:1].upper() + t[1:] if t else t + + +def title_from_body(text): + """The document's first markdown heading, or None. + + Preferred over the file name because it is what a human wrote for a human: + `03-q-01-do-we-know-the-chain-participant-by-name.md` opens with + `## Q-01. Do We Know the Chain Participant by Name`, and there is no + mechanical route from the first string to the second. Only the first + heading is consulted, and only before any prose — a heading further down is + a section, not a name.""" + for line in text.splitlines(): + if not line.strip(): + continue + m = _HEADING.match(line) + return m.group(1).strip() if m else None + return None + + +def sanitize_title(title): + """Make a string safe to be one title SEGMENT. + + `/` becomes `-`: a slash inside a heading would silently invent a level of + hierarchy that the author did not ask for, and inventing structure is worse + than losing a slash.""" + t = _MARKUP.sub("", _UNSAFE.sub("", title.replace("/", "-"))) + return re.sub(r"\s+", " ", t).strip(" .-") or "untitled" + + +def join_title(*parts): + """Join title segments with the hierarchy separator, dropping empties.""" + return "/".join(p for p in parts if p) + + +def path_segment(segment): + """One title segment as one path component.""" + s = _PATH_NOISE.sub("", _MARKUP.sub("", _UNSAFE.sub("", segment))) + s = re.sub(r"\s+", "-", s.replace("/", "-").strip()) + return _DASHES.sub("-", s).strip("-.") or "untitled" + + +def path_for_title(title): + """Relative path, inside a space, for a title. Always ends in `.md`.""" + parts = [path_segment(p) for p in title.split("/") if p.strip()] + if not parts: + parts = ["untitled"] + return os.path.join(*parts) + ".md" + + +# -------------------------------------------------------------------------- +# the manifest +# -------------------------------------------------------------------------- + +def blank_manifest(space): + return {"space": space, "pages": {}} + + +def manifest_path(space, root=None): + return os.path.join(space_root(space, root), MANIFEST) + + +def load_manifest(space, root=None): + """The space's manifest, or a blank one. + + A missing manifest and an empty one are the same thing to every caller here + — but they are NOT the same thing to a caller deciding whether to print + "no such space". That distinction is `os.path.isdir(space_root(...))`, and + the commands make it themselves rather than reading it out of a dict.""" + p = manifest_path(space, root) + if not os.path.isfile(p): + return blank_manifest(space) + with open(p, encoding="utf-8") as f: + m = json.load(f) + m.setdefault("space", space) + m.setdefault("pages", {}) + return m + + +def save_manifest(manifest, root=None): + """Write the manifest, keys sorted, one page per line-block. + + Sorted and indented because this file lands in a diff every time anything + syncs, and a diff nobody can read is a diff nobody checks.""" + p = manifest_path(manifest["space"], root) + os.makedirs(os.path.dirname(p), exist_ok=True) + ordered = {"space": manifest["space"], "pages": {}} + for path, e in sorted(manifest.get("pages", {}).items()): + ordered["pages"][path] = {k: e[k] for k in DOMAIN_KEYS if k in e} + ordered["pages"][path].update( + {k: v for k, v in sorted(e.items()) if k not in DOMAIN_KEYS}) + with open(p, "w", encoding="utf-8") as f: + json.dump(ordered, f, ensure_ascii=False, indent=2, sort_keys=False) + f.write("\n") + return p + + +def entry(title, order=None, source=None, **extra): + """A manifest entry. Domain keys first, passthrough after — the same + render order the issue layer uses, for the same reason: it makes a diff of + the file readable.""" + e = {"title": title} + if order is not None: + e["order"] = order + if source is not None: + e["source"] = source + e.update({k: v for k, v in extra.items() if v is not None}) + return e + + +def find_by_source(manifest, source, prefix=""): + """(relpath, entry) for the page imported from this source file, or + (None, None). + + The path is derived from the title, so a retitle moves it — and looking a + page up by its new path would find nothing, treat it as new, and publish a + duplicate beside the page it was meant to rename. Source is the one link + that survives a rename, which is why it is recorded at all. + + Scoped by title prefix, so importing the same directory twice under two + prefixes gives two independent trees rather than one fighting over itself. + """ + for path, e in manifest.get("pages", {}).items(): + if e.get("source") != source: + continue + if prefix and not (e.get("title", "") == prefix + or e.get("title", "").startswith(prefix + "/")): + continue + return path, e + return None, None + + +def sort_key(relpath, e): + """Order a tree for display and for an index. + + Directory by directory, `order` first and unnumbered pages after — an + explicit `NN-` is a decision, its absence is not. Ties break on title so + the output is stable.""" + d = os.path.dirname(relpath) + o = e.get("order") + return (d, 0 if o is not None else 1, o if o is not None else 0, + e.get("title", relpath)) + + +def sorted_pages(manifest): + """[(relpath, entry)] in tree order.""" + return sorted(manifest.get("pages", {}).items(), + key=lambda kv: sort_key(kv[0], kv[1])) + + +def by_title(manifest): + return {e["title"]: (p, e) for p, e in manifest.get("pages", {}).items() + if e.get("title")} + + +def children_of(manifest, prefix): + """Every page at or under a title prefix. + + The wiki this feeds is flat, so "children" is a prefix test on the title + and nothing more — there is no tree to walk, only a naming convention to + trust.""" + out = [] + for p, e in sorted_pages(manifest): + t = e.get("title", "") + if t == prefix or t.startswith(prefix + "/"): + out.append((p, e)) + return out + + +def body_hash(text): + """sha1 of the exact bytes a page would be published as. + + This is the whole of change detection: a page is worth pushing when what is + on disk hashes differently from what was pushed last. No timestamps, no + drift model — the same stance the issue store takes.""" + if isinstance(text, str): + text = text.encode("utf-8") + return hashlib.sha1(text).hexdigest() + + +# -------------------------------------------------------------------------- +# importing a directory of markdown +# -------------------------------------------------------------------------- + +SKIP_DIRS = {".git", ".svn", "__pycache__", "node_modules"} +MD_EXT = (".md", ".markdown") + + +def walk_markdown(src): + """Every markdown file under `src`, as paths relative to it, depth first + and sorted so an import is reproducible.""" + out = [] + for dirpath, dirnames, filenames in os.walk(src): + dirnames[:] = sorted(d for d in dirnames + if d not in SKIP_DIRS and not d.startswith(".")) + rel = os.path.relpath(dirpath, src) + rel = "" if rel == "." else rel + for f in sorted(filenames): + if f.lower().endswith(MD_EXT) and not f.startswith("."): + out.append(os.path.join(rel, f) if rel else f) + return out + + +def title_for_source(relpath, text, prefix=""): + """The title a source file gets on import. + + Three rules, in this order, and the reference doc spells out why: + + 1. `order 0` (`00-intro.md`, or a literal `index`/`readme`) is the page for + the directory it sits in. Its title comes from the DIRECTORY name, not + from its own heading — a child's title must extend its parent's exactly, + and `ideas/00-intro.md` opens with "Ideas for chain business + requirements", which no child would ever be prefixed by. + 2. Any other file takes its first heading, sanitized. + 3. No heading: the file name, made readable. + """ + parts = relpath.replace(os.sep, "/").split("/") + name = parts[-1] + dirs = [sanitize_title(title_from_name(d)) for d in parts[:-1]] + + stem = strip_ext(name).lower() + if order_of(name) == 0 or stem in ("index", "readme"): + # The directory's own page. At the root of the import that is the + # prefix itself. + return join_title(prefix, *dirs) + + own = title_from_body(text) + own = sanitize_title(own) if own else sanitize_title(title_from_name(name)) + return join_title(prefix, *dirs, own) + + +def plan_import(src, prefix="", read=None): + """Work out what an import would produce, without writing anything. + + Returns (pages, collisions): + pages [{"source", "path", "title", "order", "text"}] in tree order + collisions [(path, [title, title, ...])] — two sources landing on one + file. Reported, never resolved: the wiki would end up with + two pages fighting over one local copy, and picking a winner + for the operator is how a discussion loses a document.""" + def default_read(p): + with open(p, encoding="utf-8") as f: + return f.read() + + read = read or default_read + pages, seen = [], {} + for rel in walk_markdown(src): + source = os.path.join(src, rel) + text = read(source) + title = title_for_source(rel, text, prefix) + path = path_for_title(title) + seen.setdefault(path, []).append(title) + # `source` is kept relative to the import root, not absolute: it is the + # only durable link between a file on the far side and the page it + # became, and it has to survive the artifacts directory being moved. + pages.append({"source": source, "rel": rel.replace(os.sep, "/"), + "path": path, "title": title, + "order": order_of(os.path.basename(rel)), "text": text}) + pages.sort(key=lambda p: sort_key(p["path"], p)) + collisions = [(p, t) for p, t in sorted(seen.items()) if len(t) > 1] + return pages, collisions + + +# -------------------------------------------------------------------------- +# rendering +# -------------------------------------------------------------------------- + +def title_tree(manifest, prefix=""): + """Group pages into a parent -> children map keyed by title. + + Built from the titles, not from the manifest's path order. Those two + disagree: on disk `Simple-Chains/System.md` sorts before + `Simple-Chains/Ideas/Scale.md`, while in the hierarchy Scale is a + grandchild of Simple Chains and System is a child. Nesting has to follow + the titles, because the titles are the only hierarchy there is. + + A parent with no page of its own still gets a node: `Simple Chains/Parked` + can have children while nothing is published at that title, and dropping + its children because it is missing would hide them entirely.""" + kids, entries = {}, {} + for _, e in manifest.get("pages", {}).items(): + title = e.get("title") + if not title: + continue + if prefix and not (title == prefix or title.startswith(prefix + "/")): + continue + entries[title] = e + parts = title.split("/") + # Every ancestor gets a node, so a gap in the chain does not orphan a + # subtree. + for i in range(len(parts), 0, -1): + kids.setdefault("/".join(parts[:i - 1]), set()).add("/".join(parts[:i])) + return kids, entries + + +def render_index(manifest, prefix="", heading=None): + """A table-of-contents page for a space or a subtree. + + Nested markdown list, indented by title depth. The wiki is flat and will + not draw this for you, so the index IS the navigation. + + Links: a published page is linked by its `sub_url`, which is the only + address Gitea guarantees. A page that has never been pushed has no sub_url + yet, so it gets Gitea's own `[[Title]]` wiki-link syntax — which resolves + the escaping itself, at render time, on the server. Rebuilding the index + after a push upgrades those links to exact ones.""" + kids, entries = title_tree(manifest, prefix) + lines = ["# %s" % (heading or prefix or "Contents"), ""] + + def order_key(title): + e = entries.get(title) or {} + o = e.get("order") + return (0 if o is not None else 1, o if o is not None else 0, title) + + def walk(node, depth): + for child in sorted(kids.get(node, ()), key=order_key): + e = entries.get(child) or {} + label = child.split("/")[-1] + sub = e.get("sub_url") + link = "[%s](%s)" % (label, sub) if sub else "[[%s|%s]]" % (child, label) + lines.append("%s- %s" % (" " * depth, link)) + walk(child, depth + 1) + + walk(prefix, 0) + lines.append("") + return "\n".join(lines) + + +def tree_lines(manifest, mark=None): + """The space as an ascii tree, for a terminal. + + `mark(relpath, entry)` returns a short state tag shown after the title — + the wiki layer passes sync state through it, and this module stays unaware + of what the tags mean.""" + out, last_dir = [], None + for path, e in sorted_pages(manifest): + d = os.path.dirname(path) + if d != last_dir: + out.append("%s/" % d if d else ".") + last_dir = d + tag = mark(path, e) if mark else "" + out.append(" %-40s %s%s" % (os.path.basename(path), + e.get("title", ""), + (" " + tag) if tag else "")) + return out diff --git a/skills/page/scripts/page_import.py b/skills/page/scripts/page_import.py new file mode 100644 index 0000000..9945c47 --- /dev/null +++ b/skills/page/scripts/page_import.py @@ -0,0 +1,159 @@ +#!/usr/bin/env python3 +""" +page_import.py — pull a directory of markdown into a space. Offline. + +This is the "wiki organization" step, and it is the only step where a page gets +its name. A discussion produces artifacts wherever the discussion happened: + + ~/…/mpns/feat/simple-chains/tmp/simple-chains/ + handoff.md scope.md + ideas/00-intro.md ideas/02-chain-core.md + questions/03-q-01-do-we-know-the-chain-participant-by-name.md + +Import copies that tree into a space under `tmp/wiki/`, gives every file a +title, and records both in the manifest. Nothing here talks to a wiki; the +result is a complete, readable, greppable tree whether or not it is ever +published. + + page_import.py --from DIR --space claude-skills/tea --prefix "Simple Chains" + + Simple-Chains/Handoff.md Simple Chains/Handoff + Simple-Chains/Ideas.md Simple Chains/Ideas + Simple-Chains/Ideas/Chain-core.md Simple Chains/Ideas/Chain core + +Re-importing is safe and is the normal way to refresh: a page already in the +manifest keeps its title (a title is a decision, not a derivation) and only its +body is replaced. `--retitle` opts into re-deriving titles, which is a rename +and, for pages already published, will orphan the old ones — so it is never the +default. + +Usage: + page_import.py --from DIR [--space SPACE] [--prefix TITLE] + [--retitle] [--dry-run] [--out DIR] +""" +import argparse +import os +import shutil +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import page # noqa: E402 + + +def die(msg, code=1): + sys.stderr.write("%s: %s\n" % (os.path.basename(sys.argv[0]), msg)) + sys.exit(code) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--from", dest="src", required=True, + help="directory of markdown to import") + ap.add_argument("--space", default="local", + help="space to import into (default: local)") + ap.add_argument("--prefix", default="", + help="title every imported page hangs under") + ap.add_argument("--retitle", action="store_true", + help="re-derive titles of pages already in the manifest " + "(a rename; orphans published pages)") + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--out", help="wiki cache root (default: /tmp/wiki)") + a = ap.parse_args() + + src = os.path.abspath(a.src) + if not os.path.isdir(src): + die("not a directory: %s" % a.src) + + root = a.out or page.WIKI_ROOT + prefix = page.sanitize_title(a.prefix) if a.prefix else "" + + pages, collisions = page.plan_import(src, prefix) + if not pages: + die("no markdown found under %s" % src) + if collisions: + for path, titles in collisions: + sys.stderr.write("collision: %s <- %s\n" % (path, " | ".join(titles))) + die("%d path collision(s); rename the sources and retry" % len(collisions)) + + manifest = page.load_manifest(a.space, root) + known = manifest["pages"] + dest_root = page.space_root(a.space, root) + # Asked before anything is written: nothing should create a space as a + # silent side effect of a write, and saying so on stderr is how the + # operator learns a typo in --space made a second one. + created = not os.path.isdir(dest_root) + + new = changed = same = moved = 0 + for p in pages: + # Looked up by SOURCE, not by path: a retitle moves the path, and a + # lookup that missed would treat the page as new and publish a + # duplicate beside the one it was meant to rename. + prior_path, prior = page.find_by_source(manifest, p["rel"], prefix) + if prior is None: + prior_path, prior = p["path"], known.get(p["path"]) + + # A title already in the manifest is a decision that was made once. + # Re-deriving it on every import would let an edited heading silently + # rename a published page — which does not rename it, it creates a + # second one and abandons the first. + title = p["title"] if (a.retitle or not prior) else prior["title"] + relpath = page.path_for_title(title) + dest = os.path.join(dest_root, relpath) + + state = "new" + if prior and relpath != prior_path: + state = "moved" + elif prior and os.path.isfile(dest): + with open(dest, encoding="utf-8") as f: + state = "same" if f.read() == p["text"] else "changed" + elif prior: + state = "changed" + + new += state == "new" + changed += state == "changed" + same += state == "same" + moved += state == "moved" + + print("%-7s %-44s %s" % (state, relpath, title)) + if a.dry_run: + continue + + os.makedirs(os.path.dirname(dest), exist_ok=True) + shutil.copyfile(p["source"], dest) + # Passthrough keys survive: a re-import must not cost a page its + # sub_url, or the next push would publish a duplicate. + e = dict(prior or {}) + e.update(page.entry(title, p["order"], p["rel"])) + if state == "moved": + # The old copy goes, the entry moves with its bookkeeping intact. + # The page in the wiki is still at its old sub_url; the next push + # sends the new title, which is what renames it there. + old = os.path.join(dest_root, prior_path) + if os.path.isfile(old): + os.remove(old) + known.pop(prior_path, None) + # A rename can leave the body byte-identical, and push decides by + # body hash alone. Clearing it is what makes the next push send the + # new title instead of skipping the page as unchanged. + e.pop("pushed", None) + known[relpath] = e + + if a.dry_run: + print("\ndry run — nothing written") + return 0 + + path = page.save_manifest(manifest, root) + if created: + sys.stderr.write("created space %s\n" % dest_root) + print("\n%d new, %d changed, %d unchanged%s -> %s" + % (new, changed, same, + ", %d renamed" % moved if moved else "", os.path.dirname(path))) + if moved: + sys.stderr.write( + "%d page(s) renamed. A published page is renamed in the wiki by " + "the next push, not by this import.\n" % moved) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/page/scripts/page_index.py b/skills/page/scripts/page_index.py new file mode 100644 index 0000000..94e8e27 --- /dev/null +++ b/skills/page/scripts/page_index.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +""" +page_index.py — write a table-of-contents page into a space. Offline. + +The wiki this feeds is flat: a title like `Simple Chains/Ideas/Chain core` has +hierarchy in its name and nowhere else, and Gitea will not draw you a tree from +it. An index page is therefore not a nicety, it is the navigation. + +Written as an ordinary page in the space, so it is pushed by the same command +as everything else and needs no special case anywhere downstream. Links are +written by TITLE rather than by URL — the wiki resolves those itself, and a +link written that way survives every filename-escaping rule this layer +deliberately refuses to model. + + page_index.py --space claude-skills/tea --prefix "Simple Chains" + -> Simple-Chains.md, title `Simple Chains` + + page_index.py --space claude-skills/tea --title Home + -> Home.md, title `Home`, listing the whole space + +Usage: + page_index.py [--space SPACE] [--prefix TITLE] [--title TITLE] + [--dry-run] [--out DIR] +""" +import argparse +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import page # noqa: E402 + + +def die(msg, code=1): + sys.stderr.write("%s: %s\n" % (os.path.basename(sys.argv[0]), msg)) + sys.exit(code) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--space", default="local") + ap.add_argument("--prefix", default="", + help="index only this subtree; also the index's own title") + ap.add_argument("--title", help="title for the index page " + "(default: --prefix, else Home)") + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--out", help="wiki cache root (default: /tmp/wiki)") + a = ap.parse_args() + + root = a.out or page.WIKI_ROOT + space_dir = page.space_root(a.space, root) + if not os.path.isdir(space_dir): + die("no such space: %s (looked in %s)" % (a.space, space_dir)) + + manifest = page.load_manifest(a.space, root) + prefix = page.sanitize_title(a.prefix) if a.prefix else "" + title = a.title or prefix or "Home" + + body = page.render_index(manifest, prefix, heading=title) + relpath = page.path_for_title(title) + + if a.dry_run: + sys.stdout.write(body) + print("-> %s (%s)" % (relpath, title)) + return 0 + + dest = os.path.join(space_dir, relpath) + os.makedirs(os.path.dirname(dest), exist_ok=True) + with open(dest, "w", encoding="utf-8") as f: + f.write(body) + + # Carries the entry's passthrough keys forward: rebuilding an index must + # update the page that is already published, never publish a second one. + prior = manifest["pages"].get(relpath, {}) + e = dict(prior) + e.update(page.entry(title, prior.get("order"))) + manifest["pages"][relpath] = e + page.save_manifest(manifest, root) + + n = len(page.children_of(manifest, prefix) if prefix + else page.sorted_pages(manifest)) + print("%s -> %s (%d entr%s)" % (title, relpath, n - 1, + "y" if n - 1 == 1 else "ies")) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/page/scripts/page_ls.py b/skills/page/scripts/page_ls.py new file mode 100644 index 0000000..8d212d9 --- /dev/null +++ b/skills/page/scripts/page_ls.py @@ -0,0 +1,88 @@ +#!/usr/bin/env python3 +""" +page_ls.py — show what a space holds. Offline. + +The tree, the titles, and one state tag per page. The tag is the only place +this layer acknowledges that a wiki exists, and it reads it the way the issue +index reads `origin:` — as an opaque fact recorded by somebody else: + + local never published; a complete state, not a pending one + synced published, and the file matches what was pushed + ahead published, and the local file has changed since + ? published, but nothing recorded what was pushed + +Usage: + page_ls.py [--space SPACE] [--prefix TITLE] [--titles] [--out DIR] +""" +import argparse +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import page # noqa: E402 + + +def die(msg, code=1): + sys.stderr.write("%s: %s\n" % (os.path.basename(sys.argv[0]), msg)) + sys.exit(code) + + +def state_of(space_dir, relpath, e): + if not e.get("sub_url"): + return "local" + pushed = e.get("pushed") + if not pushed: + return "?" + full = os.path.join(space_dir, relpath) + if not os.path.isfile(full): + return "missing" + with open(full, encoding="utf-8") as f: + return "synced" if page.body_hash(f.read()) == pushed else "ahead" + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--space", default="local") + ap.add_argument("--prefix", default="", help="only titles at or under this") + ap.add_argument("--titles", action="store_true", + help="print one title per line and nothing else") + ap.add_argument("--out", help="wiki cache root (default: /tmp/wiki)") + a = ap.parse_args() + + root = a.out or page.WIKI_ROOT + space_dir = page.space_root(a.space, root) + # "Does not exist" and "is empty" are different answers and get different + # messages — an empty space is a space somebody made on purpose. + if not os.path.isdir(space_dir): + die("no such space: %s (looked in %s)" % (a.space, space_dir)) + + manifest = page.load_manifest(a.space, root) + pages = (page.children_of(manifest, a.prefix) if a.prefix + else page.sorted_pages(manifest)) + if not pages: + print("space %s is empty" % a.space if not a.prefix + else "nothing at or under %r" % a.prefix) + return 0 + + if a.titles: + for _, e in pages: + print(e.get("title", "")) + return 0 + + sub = {p: e for p, e in pages} + view = dict(manifest, pages=sub) + for line in page.tree_lines(view, mark=lambda p, e: state_of(space_dir, p, e)): + print(line) + + counts = {} + for p, e in pages: + s = state_of(space_dir, p, e) + counts[s] = counts.get(s, 0) + 1 + print("\n%d page(s): %s" % (len(pages), + ", ".join("%d %s" % (v, k) + for k, v in sorted(counts.items())))) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/wiki/SKILL.md b/skills/wiki/SKILL.md new file mode 100644 index 0000000..31addf4 --- /dev/null +++ b/skills/wiki/SKILL.md @@ -0,0 +1,119 @@ +--- +name: wiki +description: Move wiki pages between a local space and Gitea — fetch a page and everything under it as a local cache, publish a page tree with an update message, list what the wiki holds. Load when the user asks to read/fetch a wiki page, cache a wiki subtree for a discussion, or publish artifacts to the wiki. Organizing artifacts into a page tree (titles, ordering, the index) is /tea:page and needs no network. +--- + +# /tea:wiki — the bridge between a local space and a Gitea wiki + +One job: translate between `tmp/wiki//` and Gitea's wiki JSON, and carry +the result over the wire. Everything about **what a page tree is** — titles, +ordering, paths, the index — belongs to `/tea:page` and is imported from there, +never redefined here. + +Transport is `tea api` through `skills/sync/scripts/_gitea.py`: the same login +pin, the same pagination, the same payload files. There is no second transport. + +## The wiki is flat, and that is the whole design + +Gitea's wiki is a list of pages, not a tree. Nesting exists only inside a +title, as `/`, and Gitea escapes that title into a filename by rules that are +its own: + +| title | `sub_url` | +|---|---| +| `Abstract Issue` | `Abstract-Issue` | +| `zz-probe/child` | `zz-probe%2Fchild.-` | +| `Simple Chains/Parked/Chain decisions — DC` | `Simple-Chains%2FParked%2FChain-decisions-%E2%80%94-DC` | + +**`sub_url` is the identity and is never constructed.** It is read back from +the API and stored in the manifest. Building one by hand that is almost right +does not fail loudly — it creates a second page and abandons the first. + +**Never commit a subdirectory into the wiki's git repository.** A real +`folder/page.md` is invisible to the API and to the web UI. It is a ghost file. +Do not clone the wiki repo to work in; use these scripts. + +## Scripts + +In `/scripts/`. + +| Script | What it does | +|---|---| +| `wiki_ls.py [--prefix T]` | what the wiki actually holds — titles, `sub_url`, last commit. One call, no bodies | +| `wiki_pull.py [--prefix T] [--space S]` | fetch a page and everything under it into a local space | +| `wiki_push.py -m MSG [--prefix T] [PATH…]` | publish; create what is new, update what changed, skip what is not | +| `wikimap.py` | md ↔ wiki JSON, pure — not a command | + +## Fetching a subtree as a cache + +"A page and its children" is a prefix test on the title, run against one +listing call, followed by one GET per page. There is no tree endpoint and no +bulk-body endpoint. + +```bash +python3 scripts/wiki_ls.py --prefix "Simple Chains" # what is there +python3 scripts/wiki_pull.py --prefix "Simple Chains" # cache it locally +``` + +A pull **overwrites** the local body — a fetch, not a merge. `synced` tells you +how old your copy is; re-pull when it matters. Nothing tracks drift. + +The space defaults to the repo's own `owner/repo`, so a pull with no flags +caches this repo's whole wiki into `tmp/wiki///`. + +## Publishing + +```bash +python3 scripts/wiki_push.py -m "Import the simple-chains discussion" --dry-run +python3 scripts/wiki_push.py -m "Import the simple-chains discussion" +``` + +`-m` is required and is the wiki commit message — the only record of why a page +changed, and it shows up in `wiki/revisions/`. One operation, one +message. + +Change detection is a hash: a page whose file matches `pushed` is skipped. +A page with no `sub_url` is created; one with a `sub_url` is edited in place, +using the title **from the manifest** — sending a different title to the edit +endpoint is a rename and leaves nothing at the old address. + +Selection, narrowest first: positional `PATH`-or-`TITLE` arguments (matched +exactly), then `--prefix`, then the whole space. + +**Pushing is additive.** A page deleted locally is not deleted in the wiki. +Removing a published page is an explicit act — the web UI, or +`tea api --login "$GITEA_LOGIN" -X DELETE repos/{owner}/{repo}/wiki/page/`. + +## Order of operations for a fresh tree + +The index links published pages by `sub_url`, which does not exist until the +first push. So: + +```bash +python3 ../page/scripts/page_import.py --from DIR --prefix "Simple Chains" +python3 scripts/wiki_push.py -m "Import the simple-chains discussion" +python3 ../page/scripts/page_index.py --prefix "Simple Chains" # now with real links +python3 scripts/wiki_push.py -m "Index" +``` + +## Linking an issue to a page + +An issue's `wiki:` field holds page **titles**, not URLs — a title is a name for +a document and stays in the domain; the URL is bookkeeping. `wiki_ls.py +--titles` prints them one per line, which is what to paste. + +## Endpoints, for when a script is not enough + +Reach for `/tea:use` and `tea api` directly only for what has no script — a +delete, or a page's history. + +| | | +|---|---| +| list | `GET repos/{owner}/{repo}/wiki/pages` | +| read | `GET repos/{owner}/{repo}/wiki/page/{sub_url}` | +| create | `POST repos/{owner}/{repo}/wiki/new` — `{title, content_base64, message}` | +| edit | `PATCH repos/{owner}/{repo}/wiki/page/{sub_url}` — same body | +| delete | `DELETE repos/{owner}/{repo}/wiki/page/{sub_url}` | +| history | `GET repos/{owner}/{repo}/wiki/revisions/{sub_url}` | + +The `tea` CLI has no wiki subcommand. `tea api` is the only route. diff --git a/skills/wiki/scripts/wiki_ls.py b/skills/wiki/scripts/wiki_ls.py new file mode 100644 index 0000000..3000b76 --- /dev/null +++ b/skills/wiki/scripts/wiki_ls.py @@ -0,0 +1,77 @@ +#!/usr/bin/env python3 +""" +wiki_ls.py — list what is actually in a wiki. One call, no bodies. + +Cheap enough to run before a pull: it tells you what titles exist, which is the +only thing a prefix filter can be built from, and it shows the `sub_url` Gitea +settled on for each — worth a look the first time a title contains a dash or a +slash, because the escaping is not what anyone guesses. + + wiki_ls.py + wiki_ls.py --prefix "Simple Chains" + wiki_ls.py --repo other/repo --titles + +Usage: + wiki_ls.py [--repo owner/repo] [--prefix TITLE] [--titles] [--urls] +""" +import argparse +import os +import sys + +_HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, _HERE) +sys.path.insert(0, os.path.join(_HERE, "..", "..", "sync", "scripts")) +import _gitea # noqa: E402 +import wikimap # noqa: E402 + + +def cell(v): + return (str(v or "").strip().replace("|", "\\|")) or "—" + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--repo", help="owner/repo (default: the repo in CWD)") + ap.add_argument("--prefix", default="", help="only titles at or under this") + ap.add_argument("--titles", action="store_true", + help="print one title per line and nothing else") + ap.add_argument("--urls", action="store_true", help="add the browser URL") + a = ap.parse_args() + + login = _gitea.require_login() + base = _gitea.repo_base(a.repo) + slug = _gitea.repo_slug(login, a.repo) + + listing = _gitea.paginate(login, "%s/wiki/pages" % base) + if not isinstance(listing, list): + _gitea.die("unexpected listing from %s/wiki/pages" % base) + + rows = sorted((p for p in listing + if wikimap.matches_prefix(p.get("title") or "", a.prefix)), + key=lambda p: p.get("title") or "") + if not rows: + print("no page at or under %r in %s" % (a.prefix, slug) if a.prefix + else "%s has no wiki pages" % slug) + return 0 + + if a.titles: + for p in rows: + print(p.get("title") or "") + return 0 + + head = ["title", "sub_url", "updated", "by"] + (["url"] if a.urls else []) + print("| %s |" % " | ".join(head)) + print("|%s|" % "|".join("---" for _ in head)) + for p in rows: + c = (p.get("last_commit") or {}).get("author") or {} + row = [cell(p.get("title")), "`%s`" % cell(p.get("sub_url")), + cell((c.get("date") or "")[:10]), cell(c.get("name"))] + if a.urls: + row.append(cell(p.get("html_url"))) + print("| %s |" % " | ".join(row)) + print("\n%d page(s) in %s" % (len(rows), slug)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/wiki/scripts/wiki_pull.py b/skills/wiki/scripts/wiki_pull.py new file mode 100644 index 0000000..2d9d47d --- /dev/null +++ b/skills/wiki/scripts/wiki_pull.py @@ -0,0 +1,127 @@ +#!/usr/bin/env python3 +""" +wiki_pull.py — fetch wiki pages into a local space. + + wiki_pull.py the whole wiki + wiki_pull.py --prefix "Simple Chains" a page and everything under it + wiki_pull.py --repo other/repo --space docs from elsewhere, into a named space + +The wiki is flat, so "a page and its children" is a prefix test on the title, +run against one listing call. One GET per page follows. There is no tree +endpoint to ask for a subtree, and no way to fetch bodies in bulk. + +Pulling OVERWRITES the local body — a fetch, not a merge, the same stance the +issue store takes. `sha` and `synced` tell you how old your copy is; re-pull +when it matters. Nothing tracks drift. + +Usage: + wiki_pull.py [--repo owner/repo] [--prefix TITLE] [--space SPACE] + [--dry-run] [--out DIR] +""" +import argparse +import os +import sys + +_HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, _HERE) +sys.path.insert(0, os.path.join(_HERE, "..", "..", "sync", "scripts")) +sys.path.insert(0, os.path.join(_HERE, "..", "..", "page", "scripts")) +import _gitea # noqa: E402 +import page # noqa: E402 +import wikimap # noqa: E402 + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--repo", help="owner/repo (default: the repo in CWD)") + ap.add_argument("--prefix", default="", help="only titles at or under this") + ap.add_argument("--space", help="local space (default: the owner/repo slug)") + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--out", help="wiki cache root (default: /tmp/wiki)") + a = ap.parse_args() + + login = _gitea.require_login() + base = _gitea.repo_base(a.repo) + slug = _gitea.repo_slug(login, a.repo) + space = a.space or slug + root = a.out or page.WIKI_ROOT + space_dir = page.space_root(space, root) + + listing = _gitea.paginate(login, "%s/wiki/pages" % base) + if not isinstance(listing, list): + _gitea.die("unexpected listing from %s/wiki/pages" % base) + + wanted = [p for p in listing + if wikimap.matches_prefix(p.get("title") or "", a.prefix)] + if not wanted: + if a.prefix: + _gitea.die("no page at or under %r in %s (%d page(s) in the wiki)" + % (a.prefix, slug, len(listing))) + _gitea.die("%s has no wiki pages" % slug) + + manifest = page.load_manifest(space, root) + + # Two remote titles can land on one local path — Gitea keeps them apart with + # its `.-` marker, a filesystem does not. Caught before anything is written, + # because the failure mode otherwise is one page silently overwriting + # another and the manifest pointing both entries at the survivor. + seen = {} + for p in wanted: + seen.setdefault(page.path_for_title(p["title"]), []).append(p["title"]) + clashes = {k: v for k, v in seen.items() if len(v) > 1} + for path, titles in sorted(clashes.items()): + sys.stderr.write("collision: %s <- %s\n" % (path, " | ".join(titles))) + + created = not os.path.isdir(space_dir) + n = 0 + for p in sorted(wanted, key=lambda x: x.get("title") or ""): + title = p["title"] + relpath = page.path_for_title(title) + if relpath in clashes: + continue + + if a.dry_run: + print("%-44s %s" % (relpath, title)) + n += 1 + continue + + full = _gitea.api(login, wikimap.page_endpoint(base, p["sub_url"])) + if not isinstance(full, dict): + _gitea.warn("could not read %r; skipped" % title) + continue + text = wikimap.decode(full) + + dest = os.path.join(space_dir, relpath) + os.makedirs(os.path.dirname(dest), exist_ok=True) + with open(dest, "w", encoding="utf-8") as f: + f.write(text) + + # The prior entry is the base so `order` — a local decision the wiki + # cannot hold — survives a pull. + e = dict(manifest["pages"].get(relpath, {})) + e.update(wikimap.from_payload(full)) + e["synced"] = _gitea.now_iso() + # What is on disk is now exactly what is published, so push has nothing + # to do until the file is edited. + e["pushed"] = page.body_hash(text) + manifest["pages"][relpath] = e + print("%-44s %s" % (relpath, title)) + n += 1 + + if a.dry_run: + print("\ndry run — %d page(s) would be written to %s" % (n, space_dir)) + return 1 if clashes else 0 + + page.save_manifest(manifest, root) + if created: + sys.stderr.write("created space %s\n" % space_dir) + print("\n%d page(s) from %s -> %s" % (n, slug, space_dir)) + if clashes: + sys.stderr.write("%d collision(s) skipped — rename them in the wiki\n" + % len(clashes)) + return 1 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/wiki/scripts/wiki_push.py b/skills/wiki/scripts/wiki_push.py new file mode 100644 index 0000000..6fb085e --- /dev/null +++ b/skills/wiki/scripts/wiki_push.py @@ -0,0 +1,152 @@ +#!/usr/bin/env python3 +""" +wiki_push.py — publish a local space to a wiki. + + wiki_push.py -m "Import the simple-chains discussion" + wiki_push.py --prefix "Simple Chains/Ideas" -m "Rework B-04" + wiki_push.py -m "Fix the send-timing table" Simple-Chains/Ideas/Send-timing.md + +Every page in the selection is compared against `pushed` — the hash of what was +last published — and only the ones that differ are sent. That is the whole of +change detection: no timestamps, no drift model. + +A page with no `sub_url` is created; a page with one is edited in place. The +title comes from the manifest, never re-derived from the file, because sending +a different title to the edit endpoint is a RENAME and leaves nothing behind at +the old address. + +Pushing is additive. A page deleted locally is NOT deleted in the wiki — the +manifest simply stops mentioning it. Removing a published page is an explicit +act; do it in the web UI or with a DELETE through /tea:use. + +Usage: + wiki_push.py -m MESSAGE [--space SPACE] [--repo owner/repo] + [--prefix TITLE] [--dry-run] [--out DIR] [PATH-or-TITLE ...] +""" +import argparse +import os +import sys + +_HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, _HERE) +sys.path.insert(0, os.path.join(_HERE, "..", "..", "sync", "scripts")) +sys.path.insert(0, os.path.join(_HERE, "..", "..", "page", "scripts")) +import _gitea # noqa: E402 +import page # noqa: E402 +import wikimap # noqa: E402 + + +def select(manifest, prefix, targets): + """The pages to consider, in tree order. + + A positional argument matches a manifest path or a title, exactly. Exact + because a near-miss that silently selects nothing is indistinguishable from + a clean no-op run, and the operator finds out only when the page never + appears.""" + pages = (page.children_of(manifest, prefix) if prefix + else page.sorted_pages(manifest)) + if not targets: + return pages, [] + want, chosen, hit = set(targets), [], set() + for p, e in pages: + if p in want or e.get("title") in want: + chosen.append((p, e)) + hit.add(p if p in want else e.get("title")) + return chosen, sorted(want - hit) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("targets", nargs="*", metavar="PATH-or-TITLE") + ap.add_argument("-m", "--message", required=True, + help="wiki commit message for this push") + ap.add_argument("--repo", help="owner/repo (default: the repo in CWD)") + ap.add_argument("--space", help="local space (default: the owner/repo slug)") + ap.add_argument("--prefix", default="", help="only titles at or under this") + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--out", help="wiki cache root (default: /tmp/wiki)") + a = ap.parse_args() + + login = _gitea.require_login() + base = _gitea.repo_base(a.repo) + slug = _gitea.repo_slug(login, a.repo) + space = a.space or slug + root = a.out or page.WIKI_ROOT + space_dir = page.space_root(space, root) + if not os.path.isdir(space_dir): + _gitea.die("no such space: %s (looked in %s). Import or pull first." + % (space, space_dir)) + + manifest = page.load_manifest(space, root) + if not manifest["pages"]: + _gitea.die("space %s has no pages in its manifest" % space) + + chosen, missing = select(manifest, a.prefix, a.targets) + for t in missing: + _gitea.warn("not in the manifest: %s" % t) + if not chosen: + _gitea.die("nothing selected") + + created = updated = skipped = 0 + for relpath, e in chosen: + title = e.get("title") + full = os.path.join(space_dir, relpath) + if not title: + _gitea.warn("%s has no title in the manifest; skipped" % relpath) + continue + if not os.path.isfile(full): + _gitea.warn("%s is in the manifest but not on disk; skipped" % relpath) + continue + with open(full, encoding="utf-8") as f: + text = f.read() + h = page.body_hash(text) + + if e.get("sub_url") and h == e.get("pushed"): + skipped += 1 + continue + + verb = "create" if not e.get("sub_url") else "update" + print("%-7s %-44s %s" % (verb, relpath, title)) + if a.dry_run: + created += verb == "create" + updated += verb == "update" + continue + + if verb == "create": + payload = wikimap.new_payload(title, text, a.message) + got = _gitea.api(login, "%s/wiki/new" % base, method="POST", + payload=payload, payload_name="wiki-new", + out_root=space_dir) + else: + payload = wikimap.edit_payload(title, text, a.message) + got = _gitea.api(login, wikimap.page_endpoint(base, e["sub_url"]), + method="PATCH", payload=payload, + payload_name="wiki-edit", out_root=space_dir) + + if not isinstance(got, dict) or not got.get("sub_url"): + _gitea.warn("%s: no page returned; the manifest is unchanged for it" + % title) + continue + + # sub_url comes back from Gitea and is stored as given. It is the only + # address this page has, and it is not something we could have computed. + e.update(wikimap.from_payload(got)) + e["synced"] = _gitea.now_iso() + e["pushed"] = h + manifest["pages"][relpath] = e + created += verb == "create" + updated += verb == "update" + + if a.dry_run: + print("\ndry run — %d to create, %d to update, %d unchanged" + % (created, updated, skipped)) + return 0 + + page.save_manifest(manifest, root) + print("\n%d created, %d updated, %d unchanged -> %s wiki" + % (created, updated, skipped, slug)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/wiki/scripts/wikimap.py b/skills/wiki/scripts/wikimap.py new file mode 100644 index 0000000..d150df6 --- /dev/null +++ b/skills/wiki/scripts/wikimap.py @@ -0,0 +1,124 @@ +#!/usr/bin/env python3 +""" +wikimap.py — md <-> Gitea wiki JSON. The whole translation, and only the +translation. + +Pure functions: no network, no filesystem, no argparse. Give it a payload and +it hands back a page; give it a page and it hands back a request body. That +purity is the point — it can be reasoned about and tested without a Gitea +anywhere, and it is the single file to open when the two representations +disagree. + +Direction of knowledge: this module imports the domain (page.py) and is +imported by the transport's callers. The domain never imports this. + +What crosses the boundary, and what does not: + + domain Gitea note + ---------------------------------------------------------------------- + title title verbatim, both ways; `/` is the + only hierarchy either side has + path — local only; derived from the title + order — local only; the wiki cannot sort + body content_base64 base64, utf-8, verbatim + — sub_url lands in the manifest as sub_url + — last_commit.sha lands as sha + — html_url lands as url + +sub_url is the identity, and it is NOT derivable +------------------------------------------------ +Gitea stores a wiki page as one flat file whose name it escapes from the title, +and the escaping is not a mapping worth reimplementing: + + "Abstract Issue" -> Abstract-Issue.md space -> dash + "zz-probe/child" -> zz-probe%2Fchild.-.md / -> %2F, and a + LITERAL dash forces a + `.-` marker so the two + cases stay distinct + +Every rule there is Gitea's to change. So `sub_url` is read back from whatever +the API returned and stored; it is never constructed here, and a caller that +needs to address a page fetches the listing rather than guessing. Building one +by hand is how you get a second page instead of an edit. + +The wiki is flat, and only titles are structured +------------------------------------------------ +There are no directories. A real subdirectory committed into the wiki's git +repository is invisible to the API and to the web UI — a ghost file. All nesting +lives in the title, which is why `page.py` treats `/` as its only separator. +""" +import base64 + +# A page's whole shape on the wire, for reference and for tests. Gitea also +# returns `commit_count`, `sidebar` and `footer` on a single-page GET; none of +# them describe the page itself, so none of them cross. +WIRE_KEYS = ("title", "sub_url", "html_url", "content_base64", "last_commit") + + +def decode(payload): + """content_base64 -> text. Missing content is "" and not None: a page that + exists with an empty body is a real state, and the caller writing a file + should not have to tell the two apart.""" + b = payload.get("content_base64") or "" + return base64.b64decode(b).decode("utf-8", "replace") if b else "" + + +def encode(text): + return base64.b64encode(text.encode("utf-8")).decode("ascii") + + +def from_payload(payload): + """Gitea JSON -> the manifest fields the wiki layer owns, plus the title + the domain owns. The caller merges this into the existing entry so that + domain keys it does not mention (`order`) survive.""" + commit = payload.get("last_commit") or {} + author = commit.get("author") or {} + return { + "title": payload.get("title") or "", + "sub_url": payload.get("sub_url") or "", + "url": payload.get("html_url") or "", + "sha": commit.get("sha") or "", + "remote-updated": author.get("date") or "", + } + + +def new_payload(title, text, message): + """POST /repos/{owner}/{repo}/wiki/new. + + `title` carries the hierarchy; Gitea derives the filename from it and + returns the sub_url it settled on. `message` is the wiki commit message — + the operator's words, not a generated one, because this is the only record + of why a page changed.""" + return {"title": title, "content_base64": encode(text), "message": message} + + +def edit_payload(title, text, message): + """PATCH /repos/{owner}/{repo}/wiki/page/{sub_url}. + + The same shape as a create. Sending the unchanged title is a no-op; sending + a different one is a RENAME, which moves the file and leaves nothing at the + old sub_url — so callers pass the title from the manifest unless the + operator asked for a rename.""" + return {"title": title, "content_base64": encode(text), "message": message} + + +def page_endpoint(base, sub_url): + """The address of one page. `sub_url` goes in verbatim — Gitea hands it + back already escaped (`%2F` and all), and re-encoding it here would produce + a path that resolves to nothing.""" + return "%s/wiki/page/%s" % (base, sub_url) + + +def revisions_endpoint(base, sub_url): + return "%s/wiki/revisions/%s" % (base, sub_url) + + +def matches_prefix(title, prefix): + """Is this page at, or under, a title prefix? + + The wiki being flat, "children" is exactly this test and nothing more: + there is no tree to walk, only a naming convention to trust. An empty + prefix matches everything.""" + if not prefix: + return True + return title == prefix or title.startswith(prefix + "/") diff --git a/tests/test_page_tree.py b/tests/test_page_tree.py new file mode 100644 index 0000000..032e4f7 --- /dev/null +++ b/tests/test_page_tree.py @@ -0,0 +1,523 @@ +#!/usr/bin/env python3 +""" +How a directory of markdown becomes a page tree, and that the tree survives a +round trip through the wiki layer's bookkeeping. + + python3 -m unittest discover -s tests -v + +Stdlib unittest, no third-party anything. `skills/*/scripts/` are not packages, +so the modules under test are imported by path. + +Nothing here touches tmp/wiki/. The subprocess cases build a throwaway +repository in a temp directory — a `.git` marker, a copy of both script layers, +a directory of fixture artifacts — and run the real scripts inside it. That is +the only honest way to test behaviour that depends on where a script is run +from, and it keeps the developer's own cache out of the blast radius. +""" +import json +import os +import shutil +import subprocess +import sys +import tempfile +import unittest + +REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +PAGE_SCRIPTS = os.path.join(REPO, "skills", "page", "scripts") +WIKI_SCRIPTS = os.path.join(REPO, "skills", "wiki", "scripts") +SYNC_SCRIPTS = os.path.join(REPO, "skills", "sync", "scripts") + +sys.path.insert(0, PAGE_SCRIPTS) +sys.path.insert(0, WIKI_SCRIPTS) +import page # noqa: E402 +import wikimap # noqa: E402 + + +# The fixture mirrors the shape a real discussion leaves behind: numbered files +# for ordering, a `00-` file standing in for its directory, headings that no +# mechanical rule could derive from the file names. +FIXTURE = { + "handoff.md": "# handoff — notification chains\n\nEntry point.\n", + "ideas/00-intro.md": "# Ideas for chain business requirements\n\nFlat list.\n", + "ideas/02-chain-core.md": "## Chain core\n\n- **B-01.** Something.\n", + "ideas/01-relations.md": "## Relations\n\nHow they relate.\n", + "questions/00-intro.md": "# Questions\n\nOpen questions.\n", + "questions/03-q-01-do-we-know-the-participant.md": + "## Q-01. Do We Know the Chain Participant by Name\n\n**Question.** …\n", + "notes/plain.md": "No heading here, only prose.\n", +} + + +def build_artifacts(root): + for rel, text in FIXTURE.items(): + p = os.path.join(root, rel.replace("/", os.sep)) + os.makedirs(os.path.dirname(p), exist_ok=True) + with open(p, "w", encoding="utf-8") as f: + f.write(text) + return root + + +# -------------------------------------------------------------------------- +# names, titles, order — pure +# -------------------------------------------------------------------------- + +class TestNames(unittest.TestCase): + + def test_order_comes_from_a_numeric_prefix(self): + self.assertEqual(page.order_of("02-chain-core.md"), 2) + self.assertEqual(page.order_of("00-intro.md"), 0) + self.assertIsNone(page.order_of("handoff.md")) + + def test_zero_is_an_order_and_not_a_missing_one(self): + """`00-` means "this is the directory's own page", so the difference + between 0 and None decides where a page lands in the tree.""" + self.assertIsNot(page.order_of("00-intro.md"), None) + + def test_the_prefix_never_reaches_the_title(self): + self.assertEqual(page.title_from_name("02-chain-core.md"), "Chain core") + + def test_only_the_first_letter_is_raised(self): + """Title-casing would wreck every name that already knows how it is + spelled.""" + self.assertEqual(page.title_from_name("sqlc-and-APNs.md"), "Sqlc and APNs") + + def test_a_heading_beats_a_file_name(self): + text = "## Q-01. Do We Know the Chain Participant by Name\n" + self.assertEqual(page.title_from_body(text), + "Q-01. Do We Know the Chain Participant by Name") + + def test_only_the_first_heading_counts(self): + self.assertEqual(page.title_from_body("# One\n\n## Two\n"), "One") + + def test_a_heading_after_prose_is_a_section_not_a_name(self): + self.assertIsNone(page.title_from_body("Prose first.\n\n# Late\n")) + + def test_markup_is_stripped_from_a_title(self): + """A page list does not render markdown, so inline code in a heading is + noise in the name.""" + self.assertEqual(page.sanitize_title("Inventory — `P-NN`"), + "Inventory — P-NN") + + def test_a_slash_in_a_heading_does_not_invent_hierarchy(self): + self.assertEqual(page.sanitize_title("Send/receive timing"), + "Send-receive timing") + + +class TestPaths(unittest.TestCase): + + def test_a_title_becomes_one_path_component_per_segment(self): + self.assertEqual(page.path_for_title("Simple Chains/Ideas/Chain core"), + os.path.join("Simple-Chains", "Ideas", "Chain-core.md")) + + def test_shell_hostile_characters_leave_the_path_but_not_the_title(self): + title = "Simple Chains/Don't send to this one" + self.assertEqual(page.path_for_title(title), + os.path.join("Simple-Chains", "Dont-send-to-this-one.md")) + self.assertIn("'", title) + + def test_an_empty_title_still_produces_a_file(self): + self.assertEqual(page.path_for_title(""), "untitled.md") + + +# -------------------------------------------------------------------------- +# importing +# -------------------------------------------------------------------------- + +class TestPlanImport(unittest.TestCase): + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.src = build_artifacts(os.path.join(self.tmp.name, "artifacts")) + self.pages, self.collisions = page.plan_import(self.src, "Simple Chains") + self.titles = {p["title"] for p in self.pages} + + def tearDown(self): + self.tmp.cleanup() + + def test_nothing_collides(self): + self.assertEqual(self.collisions, []) + + def test_an_order_zero_file_becomes_the_directorys_own_page(self): + self.assertIn("Simple Chains/Ideas", self.titles) + + def test_that_page_is_named_for_the_directory_not_its_heading(self): + """`ideas/00-intro.md` opens with "Ideas for chain business + requirements". A child's title must extend its parent's exactly, and no + child would ever be prefixed by that.""" + self.assertNotIn("Simple Chains/Ideas for chain business requirements", + self.titles) + + def test_every_child_extends_its_parents_title(self): + self.assertIn("Simple Chains/Ideas/Chain core", self.titles) + self.assertIn("Simple Chains/Questions/" + "Q-01. Do We Know the Chain Participant by Name", + self.titles) + + def test_a_file_without_a_heading_falls_back_to_its_name(self): + self.assertIn("Simple Chains/Notes/Plain", self.titles) + + def test_the_prefix_hangs_everything_under_one_title(self): + self.assertTrue(all(t.startswith("Simple Chains/") for t in self.titles)) + + def test_numeric_prefixes_order_siblings(self): + ideas = [p for p in self.pages + if p["title"].startswith("Simple Chains/Ideas/")] + self.assertEqual([p["title"].split("/")[-1] for p in ideas], + ["Relations", "Chain core"]) + + def test_a_collision_is_reported_and_not_resolved(self): + """Two headings that sanitize to one path. Picking a winner is how a + discussion loses a document.""" + d = os.path.join(self.tmp.name, "clash") + os.makedirs(d) + for name, heading in (("a.md", "# Send timing"), ("b.md", "# Send/timing")): + with open(os.path.join(d, name), "w") as f: + f.write(heading + "\n") + _, collisions = page.plan_import(d) + self.assertEqual(len(collisions), 1) + + +# -------------------------------------------------------------------------- +# the index +# -------------------------------------------------------------------------- + +class TestIndex(unittest.TestCase): + + def setUp(self): + self.m = page.blank_manifest("s") + for title, order in (("Top", None), + ("Top/Ideas", 0), + ("Top/Ideas/Relations", 1), + ("Top/Ideas/Chain core", 2), + ("Top/Zeta", None), + ("Top/Parked/Decisions", None)): + self.m["pages"][page.path_for_title(title)] = page.entry(title, order) + + def test_nesting_follows_titles_not_manifest_path_order(self): + """On disk `Top/Zeta.md` sorts before `Top/Ideas/Chain-core.md`; in the + hierarchy Zeta is a child and Chain core a grandchild.""" + body = page.render_index(self.m, "Top") + lines = [l for l in body.splitlines() if l.strip().startswith("- ") + or l.strip().startswith(" - ")] + ideas = next(i for i, l in enumerate(lines) if "|Ideas]]" in l) + core = next(i for i, l in enumerate(lines) if "Chain core]]" in l) + zeta = next(i for i, l in enumerate(lines) if "|Zeta]]" in l) + self.assertLess(ideas, core) + self.assertLess(core, zeta) + + def test_a_parent_with_no_page_still_holds_its_children(self): + """Nothing is published at `Top/Parked`; dropping it would hide + Decisions entirely.""" + body = page.render_index(self.m, "Top") + self.assertIn("- [[Top/Parked|Parked]]", body) + self.assertIn(" - [[Top/Parked/Decisions|Decisions]]", body) + + def test_an_unpublished_page_is_linked_by_wiki_syntax(self): + self.assertIn("[[Top/Ideas|Ideas]]", page.render_index(self.m, "Top")) + + def test_a_published_page_is_linked_by_its_sub_url(self): + """sub_url is the only address Gitea guarantees, and it appears only + after a push — so rebuilding the index after publishing upgrades the + links.""" + rel = page.path_for_title("Top/Ideas") + self.m["pages"][rel]["sub_url"] = "Top%2FIdeas" + self.assertIn("- [Ideas](Top%2FIdeas)", page.render_index(self.m, "Top")) + + def test_the_prefix_itself_is_not_listed_inside_its_own_index(self): + self.assertNotIn("|Top]]", page.render_index(self.m, "Top")) + + +# -------------------------------------------------------------------------- +# the manifest +# -------------------------------------------------------------------------- + +class TestManifest(unittest.TestCase): + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + + def tearDown(self): + self.tmp.cleanup() + + def test_a_missing_manifest_loads_blank(self): + m = page.load_manifest("a/b", self.tmp.name) + self.assertEqual(m["pages"], {}) + + def test_wiki_bookkeeping_survives_a_round_trip(self): + """The domain never reads sub_url, and must never drop it either — a + lost sub_url is a duplicate page on the next push.""" + m = page.blank_manifest("a/b") + m["pages"]["X.md"] = page.entry("X", 1, sub_url="X", pushed="deadbeef") + page.save_manifest(m, self.tmp.name) + back = page.load_manifest("a/b", self.tmp.name) + self.assertEqual(back["pages"]["X.md"]["sub_url"], "X") + self.assertEqual(back["pages"]["X.md"]["pushed"], "deadbeef") + self.assertEqual(back["pages"]["X.md"]["order"], 1) + + def test_domain_keys_are_written_first(self): + """The manifest lands in a diff on every sync; a readable one gets + checked.""" + m = page.blank_manifest("a/b") + m["pages"]["X.md"] = page.entry("X", 1, sub_url="X") + with open(page.save_manifest(m, self.tmp.name), encoding="utf-8") as f: + raw = f.read() + self.assertLess(raw.index('"title"'), raw.index('"sub_url"')) + + def test_children_of_is_a_prefix_test_and_not_a_substring_one(self): + m = page.blank_manifest("s") + for t in ("Top", "Top/A", "Topaz", "Topaz/B"): + m["pages"][page.path_for_title(t)] = page.entry(t) + got = {e["title"] for _, e in page.children_of(m, "Top")} + self.assertEqual(got, {"Top", "Top/A"}) + + +# -------------------------------------------------------------------------- +# md <-> wiki JSON +# -------------------------------------------------------------------------- + +class TestWikiMap(unittest.TestCase): + + def test_a_body_survives_encode_and_decode(self): + text = "# Заголовок — DC\n\n- [x] пункт\n" + self.assertEqual(wikimap.decode({"content_base64": wikimap.encode(text)}), + text) + + def test_an_empty_page_decodes_to_an_empty_string(self): + """A page that exists with no body is a real state; the caller writing + a file should not have to tell it from a missing key.""" + self.assertEqual(wikimap.decode({}), "") + self.assertEqual(wikimap.decode({"content_base64": None}), "") + + def test_from_payload_takes_the_address_gitea_returned(self): + got = wikimap.from_payload({ + "title": "A/B", "sub_url": "A%2FB.-", "html_url": "https://x/A%2FB.-", + "last_commit": {"sha": "abc", "author": {"date": "2026-08-10T11:15:39Z"}}, + }) + self.assertEqual(got["sub_url"], "A%2FB.-") + self.assertEqual(got["sha"], "abc") + self.assertEqual(got["remote-updated"], "2026-08-10T11:15:39Z") + + def test_a_sub_url_goes_into_the_endpoint_verbatim(self): + """Gitea hands it back already escaped; re-encoding it produces a path + that resolves to nothing.""" + self.assertEqual( + wikimap.page_endpoint("repos/o/r", "A%2FB.-"), + "repos/o/r/wiki/page/A%2FB.-") + + def test_prefix_matching_needs_a_separator(self): + self.assertTrue(wikimap.matches_prefix("Top", "Top")) + self.assertTrue(wikimap.matches_prefix("Top/A", "Top")) + self.assertFalse(wikimap.matches_prefix("Topaz", "Top")) + + def test_an_empty_prefix_matches_everything(self): + self.assertTrue(wikimap.matches_prefix("anything", "")) + + def test_a_payload_carries_the_operators_message(self): + p = wikimap.new_payload("A/B", "body", "why it changed") + self.assertEqual(p["message"], "why it changed") + self.assertEqual(wikimap.decode(p), "body") + + +# -------------------------------------------------------------------------- +# the scripts, in a throwaway repository +# -------------------------------------------------------------------------- + +class TestImportScript(unittest.TestCase): + """The real scripts, run as subprocesses inside a scratch repo.""" + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.root = self.tmp.name + os.makedirs(os.path.join(self.root, ".git")) + for layer in ("page", "wiki"): + shutil.copytree(os.path.join(REPO, "skills", layer, "scripts"), + os.path.join(self.root, "skills", layer, "scripts"), + ignore=shutil.ignore_patterns("__pycache__")) + shutil.copytree(SYNC_SCRIPTS, + os.path.join(self.root, "skills", "sync", "scripts"), + ignore=shutil.ignore_patterns("__pycache__")) + self.src = build_artifacts(os.path.join(self.root, "artifacts")) + self.scripts = os.path.join(self.root, "skills", "page", "scripts") + self.space = os.path.join(self.root, "tmp", "wiki", "s") + + def tearDown(self): + self.tmp.cleanup() + + def run_script(self, name, *args, cwd=None): + return subprocess.run( + [sys.executable, os.path.join(self.scripts, name)] + list(args), + capture_output=True, text=True, cwd=cwd or self.root) + + def manifest(self): + with open(os.path.join(self.space, ".pages.json"), encoding="utf-8") as f: + return json.load(f) + + def do_import(self, *extra): + return self.run_script("page_import.py", "--from", self.src, + "--space", "s", "--prefix", "Top", *extra) + + def test_dry_run_writes_nothing(self): + r = self.do_import("--dry-run") + self.assertEqual(r.returncode, 0, r.stderr) + self.assertFalse(os.path.exists(self.space)) + + def test_import_writes_the_tree_and_the_manifest(self): + self.assertEqual(self.do_import().returncode, 0) + self.assertTrue(os.path.isfile( + os.path.join(self.space, "Top", "Ideas", "Chain-core.md"))) + self.assertIn("Top/Ideas/Chain core", + {e["title"] for e in self.manifest()["pages"].values()}) + + def test_creating_a_space_is_announced(self): + """Nothing creates a store as a silent side effect of a write — that is + how a typo in --space makes a second one nobody notices.""" + self.assertIn("created space", self.do_import().stderr) + + def test_the_cache_is_found_from_a_subdirectory(self): + """The anchor is the script's own location, not cwd. A `cd` outlives + the command that ran it.""" + self.do_import() + deep = os.path.join(self.src, "ideas") + r = self.run_script("page_ls.py", "--space", "s", cwd=deep) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertIn("Chain core", r.stdout) + + def test_a_reimport_keeps_the_title_and_the_wiki_bookkeeping(self): + self.do_import() + m = self.manifest() + rel = "Top/Ideas/Chain-core.md" + m["pages"][rel]["sub_url"] = "Top%2FIdeas%2FChain-core" + with open(os.path.join(self.space, ".pages.json"), "w") as f: + json.dump(m, f) + + # The heading changes. Without the manifest that would rename a + # published page, which does not rename it — it publishes a second one. + with open(os.path.join(self.src, "ideas", "02-chain-core.md"), "w") as f: + f.write("## A completely different heading\n\nchanged\n") + self.do_import() + + after = self.manifest()["pages"][rel] + self.assertEqual(after["title"], "Top/Ideas/Chain core") + self.assertEqual(after["sub_url"], "Top%2FIdeas%2FChain-core") + with open(os.path.join(self.space, rel), encoding="utf-8") as f: + self.assertIn("A completely different heading", f.read()) + + def test_retitle_moves_the_page_and_keeps_its_address(self): + """A retitle changes the path, so the entry has to be found by source. + Found by path it would look new, and the next push would publish a + duplicate beside the page it was meant to rename.""" + self.do_import() + m = self.manifest() + m["pages"]["Top/Ideas/Chain-core.md"]["sub_url"] = "Top%2FIdeas%2FChain-core" + m["pages"]["Top/Ideas/Chain-core.md"]["pushed"] = "deadbeef" + with open(os.path.join(self.space, ".pages.json"), "w") as f: + json.dump(m, f) + + with open(os.path.join(self.src, "ideas", "02-chain-core.md"), "w") as f: + f.write("## Chain core, renamed\n") + self.do_import("--retitle") + + pages = self.manifest()["pages"] + self.assertNotIn("Top/Ideas/Chain-core.md", pages) + moved = pages["Top/Ideas/Chain-core-renamed.md"] + self.assertEqual(moved["title"], "Top/Ideas/Chain core, renamed") + self.assertEqual(moved["sub_url"], "Top%2FIdeas%2FChain-core") + self.assertFalse(os.path.exists( + os.path.join(self.space, "Top", "Ideas", "Chain-core.md"))) + + def test_a_rename_makes_the_next_push_send_the_page(self): + """The body can be byte-identical after a rename, and push decides by + body hash alone — so a stale `pushed` would skip the rename forever.""" + self.do_import() + m = self.manifest() + rel = "Top/Ideas/Chain-core.md" + with open(os.path.join(self.space, rel), encoding="utf-8") as f: + body = f.read() + m["pages"][rel]["sub_url"] = "x" + m["pages"][rel]["pushed"] = __import__("hashlib").sha1( + body.encode()).hexdigest() + with open(os.path.join(self.space, ".pages.json"), "w") as f: + json.dump(m, f) + + src = os.path.join(self.src, "ideas", "02-chain-core.md") + with open(src, encoding="utf-8") as f: + text = f.read() + with open(src, "w") as f: + f.write(text.replace("## Chain core", "## Chain core renamed")) + self.do_import("--retitle") + + moved = self.manifest()["pages"]["Top/Ideas/Chain-core-renamed.md"] + self.assertNotIn("pushed", moved) + + def test_ls_reports_an_unpublished_page_as_local(self): + self.do_import() + r = self.run_script("page_ls.py", "--space", "s") + self.assertIn("local", r.stdout) + self.assertNotIn("synced", r.stdout) + + def test_ls_distinguishes_a_missing_space_from_an_empty_one(self): + r = self.run_script("page_ls.py", "--space", "nope") + self.assertNotEqual(r.returncode, 0) + self.assertIn("no such space", r.stderr) + + def test_index_is_written_as_an_ordinary_page(self): + self.do_import() + r = self.run_script("page_index.py", "--space", "s", "--prefix", "Top") + self.assertEqual(r.returncode, 0, r.stderr) + self.assertIn("Top.md", self.manifest()["pages"]) + with open(os.path.join(self.space, "Top.md"), encoding="utf-8") as f: + body = f.read() + self.assertIn("- [[Top/Ideas|Ideas]]", body) + self.assertIn(" - [[Top/Ideas/Chain core|Chain core]]", body) + + +# -------------------------------------------------------------------------- +# the layering rule, mechanically +# -------------------------------------------------------------------------- + +class TestLayering(unittest.TestCase): + + def test_the_page_layer_is_stdlib_only(self): + """skills/page must keep working with skills/wiki deleted — so no + transport, and above all no subprocess, in the domain layer.""" + imported = set() + for name in sorted(os.listdir(PAGE_SCRIPTS)): + if not name.endswith(".py"): + continue + with open(os.path.join(PAGE_SCRIPTS, name)) as f: + for line in f: + if line.startswith(("import ", "from ")): + imported.add(line.split()[1].split(".")[0]) + foreign = imported - {"page"} - sys.stdlib_module_names + self.assertEqual(foreign, set(), + "non-stdlib import in the page layer: %s" + % ", ".join(sorted(foreign))) + self.assertNotIn("subprocess", imported) + + def test_the_page_layer_never_mentions_a_tracker(self): + """A sub_url, a login, an HTTP verb in skills/page means the concept is + in the wrong layer.""" + banned = ("tea api", "_gitea", "GITEA_LOGIN", "content_base64") + for name in sorted(os.listdir(PAGE_SCRIPTS)): + if not name.endswith(".py"): + continue + with open(os.path.join(PAGE_SCRIPTS, name)) as f: + body = f.read() + for word in banned: + self.assertNotIn(word, body, + "%s mentions %r" % (name, word)) + + def test_wikimap_is_pure(self): + """The translation layer holds no transport and no I/O: give it a + payload, get a page; give it a page, get a request body. Checked on the + imports, not on the prose — the docstring names the things it refuses + to do.""" + with open(os.path.join(WIKI_SCRIPTS, "wikimap.py")) as f: + imported = {line.split()[1].split(".")[0] for line in f + if line.startswith(("import ", "from "))} + self.assertEqual(imported, {"base64"}, + "wikimap.py imports more than the translation needs") + + +if __name__ == "__main__": + unittest.main()