mirror of
https://github.com/akitaonrails/ai-memory.git
synced 2026-10-02 03:24:46 +08:00
Merge upstream/main into fork main (post-#86 reindex-from-wiki)
Brings the fork's main current with upstream — now includes reindex-from-wiki (#86) + the audit hardening on top. Keeps the fork-only CLAUDE.md routing doc.
This commit is contained in:
+21
-1
@@ -7,6 +7,25 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [0.13.0] - 2026-06-08
|
||||
### Added
|
||||
- New `ai-memory reindex` lifecycle command rebuilds the derived SQLite page
|
||||
index from the on-disk wiki. It recreates workspace/project rows from
|
||||
per-scope `_meta.md` manifests while preserving the UUIDs encoded in the
|
||||
wiki tree, then reindexes page markdown into pages, links, and FTS. The
|
||||
command refuses to run unless the SQLite store is clean; operators should
|
||||
stop the server, back up data, move/remove `db/memory.sqlite`, run
|
||||
`reindex`, and recompute embeddings separately with `embed` when needed.
|
||||
- Wiki startup now backfills per-workspace and per-project `_meta.md` manifests
|
||||
containing the human workspace/project names (and `repo_path` for projects)
|
||||
so the markdown tree is self-describing enough to rebuild the derived DB.
|
||||
|
||||
### Fixed
|
||||
- Wiki reindexing now treats `log.md` / `log-YYYY-MM.md` as raw hook ledgers
|
||||
only when their content opens with the hook log prefix, so ordinary markdown
|
||||
pages with reserved-looking names are no longer silently dropped. `_meta.md`
|
||||
manifests and direct watcher events also reject symlinks before reading.
|
||||
|
||||
## [0.12.3] - 2026-06-07
|
||||
|
||||
## [0.12.2] - 2026-06-07
|
||||
@@ -876,7 +895,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- Consolidator used server startup default project instead of the
|
||||
session's actual project.
|
||||
|
||||
[Unreleased]: https://github.com/akitaonrails/ai-memory/compare/v0.12.3...HEAD
|
||||
[Unreleased]: https://github.com/akitaonrails/ai-memory/compare/v0.13.0...HEAD
|
||||
[0.13.0]: https://github.com/akitaonrails/ai-memory/releases/tag/v0.13.0
|
||||
[0.12.3]: https://github.com/akitaonrails/ai-memory/releases/tag/v0.12.3
|
||||
[0.12.2]: https://github.com/akitaonrails/ai-memory/releases/tag/v0.12.2
|
||||
[0.12.1]: https://github.com/akitaonrails/ai-memory/releases/tag/v0.12.1
|
||||
|
||||
Generated
+10
-10
@@ -31,7 +31,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-cli"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-consolidate",
|
||||
"ai-memory-core",
|
||||
@@ -71,7 +71,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-consolidate"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-core",
|
||||
"ai-memory-llm",
|
||||
@@ -92,7 +92,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-core"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"jiff",
|
||||
"regex",
|
||||
@@ -106,7 +106,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-eval"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-consolidate",
|
||||
"ai-memory-core",
|
||||
@@ -125,7 +125,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-hooks"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-consolidate",
|
||||
"ai-memory-core",
|
||||
@@ -147,7 +147,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-llm"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-core",
|
||||
"anyhow",
|
||||
@@ -170,7 +170,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-mcp"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-consolidate",
|
||||
"ai-memory-core",
|
||||
@@ -197,7 +197,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-store"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-core",
|
||||
"anyhow",
|
||||
@@ -220,7 +220,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-web"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-core",
|
||||
"ai-memory-store",
|
||||
@@ -243,7 +243,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ai-memory-wiki"
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
dependencies = [
|
||||
"ai-memory-core",
|
||||
"ai-memory-llm",
|
||||
|
||||
+9
-9
@@ -16,7 +16,7 @@ members = [
|
||||
]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.12.3"
|
||||
version = "0.13.0"
|
||||
edition = "2024"
|
||||
rust-version = "1.95"
|
||||
license = "MIT"
|
||||
@@ -25,14 +25,14 @@ authors = ["Fabio Akita <boss@akitaonrails.com>"]
|
||||
|
||||
[workspace.dependencies]
|
||||
# Inter-crate dependencies.
|
||||
ai-memory-core = { path = "crates/ai-memory-core", version = "0.12.3" }
|
||||
ai-memory-store = { path = "crates/ai-memory-store", version = "0.12.3" }
|
||||
ai-memory-wiki = { path = "crates/ai-memory-wiki", version = "0.12.3" }
|
||||
ai-memory-mcp = { path = "crates/ai-memory-mcp", version = "0.12.3" }
|
||||
ai-memory-hooks = { path = "crates/ai-memory-hooks", version = "0.12.3" }
|
||||
ai-memory-llm = { path = "crates/ai-memory-llm", version = "0.12.3" }
|
||||
ai-memory-consolidate = { path = "crates/ai-memory-consolidate", version = "0.12.3" }
|
||||
ai-memory-web = { path = "crates/ai-memory-web", version = "0.12.3" }
|
||||
ai-memory-core = { path = "crates/ai-memory-core", version = "0.13.0" }
|
||||
ai-memory-store = { path = "crates/ai-memory-store", version = "0.13.0" }
|
||||
ai-memory-wiki = { path = "crates/ai-memory-wiki", version = "0.13.0" }
|
||||
ai-memory-mcp = { path = "crates/ai-memory-mcp", version = "0.13.0" }
|
||||
ai-memory-hooks = { path = "crates/ai-memory-hooks", version = "0.13.0" }
|
||||
ai-memory-llm = { path = "crates/ai-memory-llm", version = "0.13.0" }
|
||||
ai-memory-consolidate = { path = "crates/ai-memory-consolidate", version = "0.13.0" }
|
||||
ai-memory-web = { path = "crates/ai-memory-web", version = "0.13.0" }
|
||||
|
||||
# (Workspace shared deps follow below)
|
||||
|
||||
|
||||
@@ -500,7 +500,7 @@ diagram, crate breakdown, schema notes, and invariants.
|
||||
| [`docs/deploy.md`](docs/deploy.md) | Homelab deploy: bin/deploy, bearer-token auth, pointers to the TLS guide. |
|
||||
| [`docs/users.md`](docs/users.md) | **Multi-user attribution (v0.8).** Four-rung auth ladder, `ai-memory user add/list/expire/revive/rotate-token` walkthrough, backward-compat migration for pre-v0.8 installs, token storage rationale. |
|
||||
| [`docs/https-via-proxy.md`](docs/https-via-proxy.md) | **HTTPS via a reverse proxy.** When you need TLS (multi-user, non-loopback) and when you don't (loopback / stdio). Copy-paste docker compose templates for Caddy + Let's Encrypt, Caddy + internal CA (LAN-only), Cloudflare Tunnel (no open ports), and external cert files; plus native-Caddy + nginx recipes. The "thinking you're secure when you're not" failure modes explicitly called out. |
|
||||
| [`docs/lifecycle-ops.md`](docs/lifecycle-ops.md) | **Read before running purge / rename / backup / restore / reset.** Safety matrix for the state-touching commands, per-project disk layout (how isolation actually works), and operator workflows for "fresh start", "snapshot before risky op", "drop one project". |
|
||||
| [`docs/lifecycle-ops.md`](docs/lifecycle-ops.md) | **Read before running purge / rename / backup / restore / reset / reindex.** Safety matrix for the state-touching commands, per-project disk layout (how isolation actually works), and operator workflows for "fresh start", "snapshot before risky op", "drop one project", and rebuilding SQLite from wiki files. |
|
||||
| [`docs/llm-provider-comparison.md`](docs/llm-provider-comparison.md) | Empirical notes behind the recommended LLM defaults. |
|
||||
| [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) | Operational summary: data flow, crate layout, cross-cutting invariants, schema. |
|
||||
| [`docs/design-decisions.md`](docs/design-decisions.md) | The full v1 spec. |
|
||||
|
||||
@@ -52,6 +52,11 @@ pub enum Command {
|
||||
Backup(BackupArgs),
|
||||
/// Restore a backup tarball into the data directory.
|
||||
Restore(RestoreArgs),
|
||||
/// Rebuild the SQLite index from the wiki/ markdown (the "DB is
|
||||
/// rebuildable from files" guarantee). Recreates workspaces/projects from
|
||||
/// each scope's `_meta.md` manifest and reindexes every page. Run with the
|
||||
/// server stopped, against a freshly-migrated (clean) data dir.
|
||||
Reindex(ReindexArgs),
|
||||
/// Print (or apply) lifecycle-hook configuration for an agent CLI.
|
||||
InstallHooks(InstallHooksArgs),
|
||||
/// Emit a single lifecycle hook natively (reads the event payload
|
||||
@@ -625,6 +630,10 @@ pub struct RestoreArgs {
|
||||
pub force: bool,
|
||||
}
|
||||
|
||||
/// Arguments for `reindex`.
|
||||
#[derive(Debug, Args)]
|
||||
pub struct ReindexArgs {}
|
||||
|
||||
/// Agent CLI to install hooks/extensions for. For MCP-only clients
|
||||
/// (Claude Desktop), use `install-mcp --client <name>` instead.
|
||||
#[derive(Debug, Clone, Copy, clap::ValueEnum)]
|
||||
|
||||
@@ -26,6 +26,7 @@ pub mod move_project;
|
||||
pub mod openclaw_plugin;
|
||||
pub mod purge_project;
|
||||
pub mod read_page;
|
||||
pub mod reindex;
|
||||
pub mod rename_project;
|
||||
pub mod render_shared;
|
||||
pub mod reorg;
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
//! `ai-memory reindex` — rebuild the SQLite index from the wiki/ markdown.
|
||||
//!
|
||||
//! Walks the wiki tree, recreates workspaces/projects from each scope's
|
||||
//! self-describing `_meta.md` manifest (preserving the ids the tree is keyed
|
||||
//! by), and reindexes every page. This is the "DB is rebuildable from files"
|
||||
//! guarantee made operable — e.g. to move a data dir onto a clean migration
|
||||
//! lineage: drop the old `db/`, then `reindex` rebuilds it from the markdown.
|
||||
//!
|
||||
//! # Exception to invariant §16
|
||||
//!
|
||||
//! Like `reset`/`restore`, `reindex` is a documented exception to the
|
||||
//! thin-HTTP-client rule: it opens the store directly and must run with the
|
||||
//! server stopped (a live SQLite writer would race the rebuild). The sysinfo
|
||||
//! guard enforces this.
|
||||
//!
|
||||
//! Episodic DB-only state (sessions, observations, handoffs, decay counters)
|
||||
//! is NOT in the markdown and is not reconstructed; embeddings can be
|
||||
//! recomputed afterwards via `ai-memory embed`.
|
||||
|
||||
use ai_memory_store::Store;
|
||||
use ai_memory_wiki::Wiki;
|
||||
use anyhow::{Context, Result, bail};
|
||||
use tracing::info;
|
||||
|
||||
use crate::cli::ReindexArgs;
|
||||
use crate::config::Config;
|
||||
use crate::process_guard::{busy_message, sibling_processes};
|
||||
|
||||
/// Run the `reindex` subcommand.
|
||||
///
|
||||
/// # Errors
|
||||
/// Returns an error if another `ai-memory` process is running, the store
|
||||
/// fails to open (e.g. a divergent migration lineage — rebuild onto a fresh
|
||||
/// `db/` instead), or a scope directory lacks its `_meta.md` manifest.
|
||||
pub async fn run(config: &Config, _args: ReindexArgs) -> Result<()> {
|
||||
let siblings = sibling_processes();
|
||||
if !siblings.is_empty() {
|
||||
bail!(busy_message("reindex", &siblings));
|
||||
}
|
||||
|
||||
let store = Store::open(&config.data_dir).context("opening store for reindex")?;
|
||||
let target = store
|
||||
.reader
|
||||
.reindex_target_status()
|
||||
.await
|
||||
.context("checking whether the SQLite store is clean for reindex")?;
|
||||
if !target.is_clean() {
|
||||
bail!(
|
||||
"refusing to reindex into a non-empty SQLite store ({counts}). \
|
||||
`ai-memory reindex` rebuilds a clean derived DB from wiki files; \
|
||||
it is not an in-place dirty-index repair. Stop the server, take a \
|
||||
backup, move or remove {db_path}, then re-run reindex. DB-only \
|
||||
state (sessions, observations, handoffs, users, audit rows, \
|
||||
embeddings, decay counters) is not reconstructed from markdown.",
|
||||
counts = target.nonzero_summary(),
|
||||
db_path = store.db_path().display(),
|
||||
);
|
||||
}
|
||||
let wiki = Wiki::new(&config.data_dir, store.writer.clone())
|
||||
.context("opening wiki")?
|
||||
// Reindex is index-rebuild only; embeddings are recomputed separately
|
||||
// via `embed` so a missing/paid provider never blocks a rebuild.
|
||||
.without_embedder();
|
||||
|
||||
let summary = wiki
|
||||
.reindex_all()
|
||||
.await
|
||||
.context("rebuilding index from wiki/")?;
|
||||
info!(
|
||||
workspaces = summary.workspaces,
|
||||
projects = summary.projects,
|
||||
pages = summary.pages,
|
||||
"reindex complete"
|
||||
);
|
||||
println!(
|
||||
"reindexed {} pages across {} project(s) in {} workspace(s) from {}",
|
||||
summary.pages,
|
||||
summary.projects,
|
||||
summary.workspaces,
|
||||
config.data_dir.join("wiki").display(),
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use ai_memory_core::{NewPage, PagePath, Tier};
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[tokio::test]
|
||||
async fn reindex_refuses_dirty_store() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let config = Config {
|
||||
data_dir: tmp.path().to_path_buf(),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let store = Store::open(tmp.path()).unwrap();
|
||||
let ws = store
|
||||
.writer
|
||||
.get_or_create_workspace("default")
|
||||
.await
|
||||
.unwrap();
|
||||
let proj = store
|
||||
.writer
|
||||
.get_or_create_project(ws, "scratch", None)
|
||||
.await
|
||||
.unwrap();
|
||||
store
|
||||
.writer
|
||||
.upsert_page(NewPage {
|
||||
workspace_id: ws,
|
||||
project_id: proj,
|
||||
path: PagePath::new("notes/stale.md").unwrap(),
|
||||
title: "stale".into(),
|
||||
body: "stale body".into(),
|
||||
tier: Tier::Semantic,
|
||||
frontmatter_json: serde_json::json!({}),
|
||||
pinned: false,
|
||||
links: Vec::new(),
|
||||
author_id: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
drop(store);
|
||||
|
||||
let err = run(&config, ReindexArgs {}).await.unwrap_err();
|
||||
assert!(
|
||||
err.to_string()
|
||||
.contains("refusing to reindex into a non-empty SQLite store"),
|
||||
"dirty DB must be rejected, got {err:#}"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -96,7 +96,12 @@ pub async fn run(config: &Config, args: ServeArgs) -> Result<()> {
|
||||
// operators discover misconfiguration immediately.
|
||||
let sanitizer = Sanitizer::new(&config.sanitize)
|
||||
.context("compiling sanitizer.extra_patterns from config")?;
|
||||
let wiki = Wiki::new(&config.data_dir, store.writer.clone())?.with_sanitizer(sanitizer.clone());
|
||||
let wiki = Wiki::new(&config.data_dir, store.writer.clone())?
|
||||
.with_sanitizer(sanitizer.clone())
|
||||
// Reader attached unconditionally: admission name-resolution uses it
|
||||
// when a chain is configured, and the startup scope-manifest backfill
|
||||
// (below) always needs it to enumerate scopes.
|
||||
.with_store_reader(store.reader.clone());
|
||||
// Attach the admission webhook chain (operator-configured via
|
||||
// `[[admission_webhooks]]` in config.toml or `AI_MEMORY_ADMISSION_WEBHOOKS__N__*`
|
||||
// env vars). Empty config = no chain attached, zero overhead. The store
|
||||
@@ -112,11 +117,19 @@ pub async fn run(config: &Config, args: ServeArgs) -> Result<()> {
|
||||
"admission webhook chain attached"
|
||||
);
|
||||
wiki.with_admission_chain(chain)
|
||||
.with_store_reader(store.reader.clone())
|
||||
};
|
||||
let provider_health = ProviderHealth::default();
|
||||
let (wiki, embedder) = configure_embedder(config, &store, wiki, &provider_health).await?;
|
||||
|
||||
// Make the wiki tree self-describing: write each scope's `_meta.md`
|
||||
// (workspace/project name + repo_path) if missing, so the markdown alone
|
||||
// can rebuild the index via `ai-memory reindex`. Idempotent; non-fatal.
|
||||
match wiki.backfill_scope_manifests().await {
|
||||
Ok(0) => {}
|
||||
Ok(n) => tracing::info!(count = n, "wrote _meta.md scope manifests"),
|
||||
Err(e) => tracing::warn!(error = %e, "scope-manifest backfill failed (non-fatal)"),
|
||||
}
|
||||
|
||||
// Keep the guard alive for the lifetime of `serve`.
|
||||
let _watcher = start_watcher(&args, &wiki)?;
|
||||
|
||||
|
||||
@@ -62,6 +62,7 @@ async fn main() -> Result<()> {
|
||||
Command::Reset(args) => commands::reset::run(&config, args),
|
||||
Command::Backup(args) => commands::backup::run(&config, args).await,
|
||||
Command::Restore(args) => commands::restore::run(&config, args),
|
||||
Command::Reindex(args) => commands::reindex::run(&config, args).await,
|
||||
Command::InstallHooks(args) => commands::install_hooks::run(&config, args),
|
||||
// `Hook` is handled in the fast-path above (before config/tracing).
|
||||
Command::Hook(_) => unreachable!("hook handled before config load"),
|
||||
|
||||
@@ -1513,7 +1513,7 @@ async fn handle_purge_project(
|
||||
.to_string();
|
||||
let mut files_deleted: Vec<String> = Vec::new();
|
||||
let mut files_failed: Vec<String> = Vec::new();
|
||||
match state.wiki.remove_project_dir(ws_id, proj_id) {
|
||||
match state.wiki.remove_project_dir(ws_id, proj_id).await {
|
||||
Ok(()) => {
|
||||
files_deleted.push(proj_root_str);
|
||||
}
|
||||
@@ -2393,7 +2393,7 @@ async fn copy_purge_merge(
|
||||
.to_string();
|
||||
let mut files_deleted: Vec<String> = Vec::new();
|
||||
let mut files_failed: Vec<String> = Vec::new();
|
||||
match state.wiki.remove_project_dir(src_ws, src_proj) {
|
||||
match state.wiki.remove_project_dir(src_ws, src_proj).await {
|
||||
Ok(()) => files_deleted.push(proj_root_str),
|
||||
Err(e) => {
|
||||
warn!(path = %proj_root_str, error = %e, "move-project: failed to remove source dir");
|
||||
|
||||
@@ -29,8 +29,9 @@ pub use ops::{EmbeddingWrite, MoveSummary, PurgeSummary, ReorgSummary};
|
||||
pub use reader::{
|
||||
ActivityWindow, BriefingPage, BriefingSnapshot, DecayCandidate, DerivedIndexStatus,
|
||||
EmbeddingTripleCount, HealthDetail, HealthPage, ObservationHit, PageAuthor, PageHit,
|
||||
PageHitWithMeta, PageLinks, PageMeta, PageSummary, ProjectSummary, ReaderPool, RelatedPage,
|
||||
StatusCounts, StoredEmbedding, StoredPageBody, WorkspaceSummary, f32_vec_to_bytes,
|
||||
PageHitWithMeta, PageLinks, PageMeta, PageSummary, ProjectSummary, ReaderPool,
|
||||
ReindexTargetStatus, RelatedPage, ScopeRow, StatusCounts, StoredEmbedding, StoredPageBody,
|
||||
WorkspaceScopeRow, WorkspaceSummary, f32_vec_to_bytes,
|
||||
};
|
||||
pub use users::{TOKEN_HASH_LEN, TOKEN_RAW_LEN, TokenPepper, generate_token, hash_token};
|
||||
pub use writer::WriterHandle;
|
||||
@@ -392,6 +393,38 @@ mod tests {
|
||||
assert_eq!(counts.observations, 0);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reindex_target_status_tracks_clean_and_dirty_store() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = Store::open(tmp.path()).unwrap();
|
||||
|
||||
let clean = store.reader.reindex_target_status().await.unwrap();
|
||||
assert!(clean.is_clean(), "fresh migrated DB must be reindex-clean");
|
||||
|
||||
let ws = store
|
||||
.writer
|
||||
.get_or_create_workspace("default")
|
||||
.await
|
||||
.unwrap();
|
||||
let proj = store
|
||||
.writer
|
||||
.get_or_create_project(ws, "ai-memory", None)
|
||||
.await
|
||||
.unwrap();
|
||||
store
|
||||
.writer
|
||||
.upsert_page(sample_page(ws, proj, "alpha.md", "body"))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let dirty = store.reader.reindex_target_status().await.unwrap();
|
||||
assert!(
|
||||
!dirty.is_clean(),
|
||||
"existing rows must block lifecycle reindex"
|
||||
);
|
||||
assert!(dirty.nonzero_summary().contains("pages=1"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn search_finds_inserted_page_and_counts_reflect_supersession() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -142,6 +142,90 @@ pub fn get_or_create_project(
|
||||
Ok(id)
|
||||
}
|
||||
|
||||
/// Insert a workspace with an **explicit id**, idempotent. Unlike
|
||||
/// [`get_or_create_workspace`] (which mints a fresh id), this preserves the id
|
||||
/// the caller already holds — used by `reindex`, which recovers the id from the
|
||||
/// wiki directory name so the rebuilt index keys pages by the same
|
||||
/// `(workspace_id, project_id)` the on-disk tree is laid out under. Re-running
|
||||
/// is a no-op (`ON CONFLICT(id)`). `created_at` is the rebuild time.
|
||||
pub fn ensure_workspace_with_id(
|
||||
conn: &mut Connection,
|
||||
id: ai_memory_core::WorkspaceId,
|
||||
name: &str,
|
||||
) -> StoreResult<()> {
|
||||
conn.execute(
|
||||
"INSERT INTO workspaces (id, name, created_at) VALUES (?1, ?2, ?3) \
|
||||
ON CONFLICT(id) DO NOTHING",
|
||||
params![id.as_bytes(), name, Timestamp::now().as_microsecond()],
|
||||
)?;
|
||||
let existing: Option<String> = conn
|
||||
.query_row(
|
||||
"SELECT name FROM workspaces WHERE id = ?1",
|
||||
params![id.as_bytes()],
|
||||
|row| row.get(0),
|
||||
)
|
||||
.optional()?;
|
||||
match existing {
|
||||
Some(existing) if existing == name => Ok(()),
|
||||
Some(existing) => Err(StoreError::Duplicate(format!(
|
||||
"workspace id {id} already exists as name '{existing}', not manifest name '{name}'"
|
||||
))),
|
||||
None => Err(StoreError::NotFound(format!(
|
||||
"workspace id {id} was not inserted"
|
||||
))),
|
||||
}?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Insert a project with an **explicit id** under `workspace_id`, idempotent.
|
||||
/// The reindex counterpart of [`ensure_workspace_with_id`].
|
||||
pub fn ensure_project_with_id(
|
||||
conn: &mut Connection,
|
||||
id: ai_memory_core::ProjectId,
|
||||
workspace_id: ai_memory_core::WorkspaceId,
|
||||
name: &str,
|
||||
repo_path: Option<&str>,
|
||||
) -> StoreResult<()> {
|
||||
conn.execute(
|
||||
"INSERT INTO projects (id, workspace_id, name, repo_path, created_at) \
|
||||
VALUES (?1, ?2, ?3, ?4, ?5) ON CONFLICT(id) DO NOTHING",
|
||||
params![
|
||||
id.as_bytes(),
|
||||
workspace_id.as_bytes(),
|
||||
name,
|
||||
repo_path,
|
||||
Timestamp::now().as_microsecond()
|
||||
],
|
||||
)?;
|
||||
type ProjectRow = (Vec<u8>, String, Option<String>);
|
||||
let existing: Option<ProjectRow> = conn
|
||||
.query_row(
|
||||
"SELECT workspace_id, name, repo_path FROM projects WHERE id = ?1",
|
||||
params![id.as_bytes()],
|
||||
|row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)),
|
||||
)
|
||||
.optional()?;
|
||||
match existing {
|
||||
Some((existing_ws, existing_name, existing_repo_path))
|
||||
if existing_ws.as_slice() == workspace_id.as_bytes()
|
||||
&& existing_name == name
|
||||
&& existing_repo_path.as_deref() == repo_path =>
|
||||
{
|
||||
Ok(())
|
||||
}
|
||||
Some((existing_ws, existing_name, existing_repo_path)) => {
|
||||
Err(StoreError::Duplicate(format!(
|
||||
"project id {id} already exists with workspace_id bytes length {}, name='{existing_name}', repo_path={existing_repo_path:?}; manifest has workspace={workspace_id}, name='{name}', repo_path={repo_path:?}",
|
||||
existing_ws.len(),
|
||||
)))
|
||||
}
|
||||
None => Err(StoreError::NotFound(format!(
|
||||
"project id {id} was not inserted"
|
||||
))),
|
||||
}?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Assert that `project_id` currently belongs to `workspace_id`.
|
||||
///
|
||||
/// Wiki writes call this before touching the filesystem so a stale hook/cache
|
||||
@@ -1203,7 +1287,7 @@ mod tests {
|
||||
//! one-line diff instead of a cascading e2e failure.
|
||||
use super::*;
|
||||
use ai_memory_core::{
|
||||
LinkTarget, NewHandoff, NewPage, NewSession, PagePath, Tier, WorkspaceId,
|
||||
LinkTarget, NewHandoff, NewPage, NewSession, PagePath, ProjectId, Tier, WorkspaceId,
|
||||
};
|
||||
use rusqlite::Connection;
|
||||
use tempfile::TempDir;
|
||||
@@ -2108,6 +2192,35 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ensure_workspace_with_id_rejects_id_name_mismatch() {
|
||||
let (_tmp, mut conn, _ws, _proj) = fresh_db();
|
||||
let id = WorkspaceId::new();
|
||||
|
||||
ensure_workspace_with_id(&mut conn, id, "from-manifest").unwrap();
|
||||
let err = ensure_workspace_with_id(&mut conn, id, "other-name").unwrap_err();
|
||||
|
||||
assert!(
|
||||
matches!(err, StoreError::Duplicate(_)),
|
||||
"same workspace id with different name must fail loudly; got {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ensure_project_with_id_rejects_existing_id_mismatch() {
|
||||
let (_tmp, mut conn, ws, _proj) = fresh_db();
|
||||
let id = ProjectId::new();
|
||||
|
||||
ensure_project_with_id(&mut conn, id, ws, "from-manifest", Some("/repo/a")).unwrap();
|
||||
let err =
|
||||
ensure_project_with_id(&mut conn, id, ws, "renamed", Some("/repo/a")).unwrap_err();
|
||||
|
||||
assert!(
|
||||
matches!(err, StoreError::Duplicate(_)),
|
||||
"same project id with different manifest data must fail loudly; got {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// V19 data-repair migration: observations whose `project_id`
|
||||
/// disagrees with their session's `project_id` are re-attributed
|
||||
/// to the session's project. Handoffs that carry a session id are
|
||||
|
||||
@@ -110,6 +110,73 @@ pub struct StatusCounts {
|
||||
pub observations: u64,
|
||||
}
|
||||
|
||||
/// Counts that must all be zero before `ai-memory reindex` rebuilds the
|
||||
/// derived SQLite store from wiki files.
|
||||
#[derive(Debug, Clone, Default, Serialize)]
|
||||
pub struct ReindexTargetStatus {
|
||||
/// Workspace rows already present in SQLite.
|
||||
pub workspaces: u64,
|
||||
/// Project rows already present in SQLite.
|
||||
pub projects: u64,
|
||||
/// Page rows, including superseded versions.
|
||||
pub pages: u64,
|
||||
/// Link rows derived from latest page bodies.
|
||||
pub links: u64,
|
||||
/// Stored embedding rows derived from latest pages.
|
||||
pub page_embeddings: u64,
|
||||
/// Session rows. These are DB-only episodic state and are not rebuilt.
|
||||
pub sessions: u64,
|
||||
/// Observation rows. These are DB-only episodic state and are not rebuilt.
|
||||
pub observations: u64,
|
||||
/// Handoff rows. These are DB-only episodic state and are not rebuilt.
|
||||
pub handoffs: u64,
|
||||
/// User rows and token hashes. These are DB-only state and are not rebuilt.
|
||||
pub users: u64,
|
||||
/// Audit rows. These are DB-only state and are not rebuilt.
|
||||
pub audit_log: u64,
|
||||
}
|
||||
|
||||
impl ReindexTargetStatus {
|
||||
/// True when the store has no user data or derived rows.
|
||||
#[must_use]
|
||||
pub fn is_clean(&self) -> bool {
|
||||
self.workspaces == 0
|
||||
&& self.projects == 0
|
||||
&& self.pages == 0
|
||||
&& self.links == 0
|
||||
&& self.page_embeddings == 0
|
||||
&& self.sessions == 0
|
||||
&& self.observations == 0
|
||||
&& self.handoffs == 0
|
||||
&& self.users == 0
|
||||
&& self.audit_log == 0
|
||||
}
|
||||
|
||||
/// Render a compact list of non-zero counters for operator errors.
|
||||
#[must_use]
|
||||
pub fn nonzero_summary(&self) -> String {
|
||||
let mut parts = Vec::new();
|
||||
macro_rules! push_nonzero {
|
||||
($field:ident) => {
|
||||
if self.$field != 0 {
|
||||
parts.push(format!("{}={}", stringify!($field), self.$field));
|
||||
}
|
||||
};
|
||||
}
|
||||
push_nonzero!(workspaces);
|
||||
push_nonzero!(projects);
|
||||
push_nonzero!(pages);
|
||||
push_nonzero!(links);
|
||||
push_nonzero!(page_embeddings);
|
||||
push_nonzero!(sessions);
|
||||
push_nonzero!(observations);
|
||||
push_nonzero!(handoffs);
|
||||
push_nonzero!(users);
|
||||
push_nonzero!(audit_log);
|
||||
parts.join(", ")
|
||||
}
|
||||
}
|
||||
|
||||
/// Derived-index health counters surfaced by admin status.
|
||||
#[derive(Debug, Clone, Default, Serialize)]
|
||||
pub struct DerivedIndexStatus {
|
||||
@@ -230,6 +297,34 @@ pub struct ProjectSummary {
|
||||
pub last_updated: Option<String>,
|
||||
}
|
||||
|
||||
/// One workspace scope with the id + name needed to write its
|
||||
/// self-describing `_meta.md` manifest. Returned by
|
||||
/// [`ReaderPool::list_all_workspace_scopes`].
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct WorkspaceScopeRow {
|
||||
/// Workspace id — matches the level-1 wiki directory name.
|
||||
pub workspace_id: WorkspaceId,
|
||||
/// Human-readable workspace name.
|
||||
pub workspace_name: String,
|
||||
}
|
||||
|
||||
/// One `(workspace, project)` scope with the ids + repo_path needed to write
|
||||
/// its self-describing `_meta.md` manifest. Returned by
|
||||
/// [`ReaderPool::list_all_scopes`]; consumed by `Wiki::backfill_scope_manifests`.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ScopeRow {
|
||||
/// Workspace id — matches the level-1 wiki directory name.
|
||||
pub workspace_id: WorkspaceId,
|
||||
/// Human-readable workspace name.
|
||||
pub workspace_name: String,
|
||||
/// Project id — matches the level-2 wiki directory name.
|
||||
pub project_id: ProjectId,
|
||||
/// Human-readable project name.
|
||||
pub project_name: String,
|
||||
/// Filesystem path the project's cwd-based routing resolves to, if any.
|
||||
pub repo_path: Option<String>,
|
||||
}
|
||||
|
||||
/// One row per workspace with aggregate stats.
|
||||
/// Returned by [`ReaderPool::list_workspaces_with_stats`].
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
@@ -2465,6 +2560,114 @@ impl ReaderPool {
|
||||
.await
|
||||
}
|
||||
|
||||
/// Return every `(workspace, project)` scope with its ids, names and
|
||||
/// `repo_path` — the data needed to write each scope's self-describing
|
||||
/// `_meta.md` manifest. Unlike [`list_projects_with_stats`], this carries
|
||||
/// the surrogate ids (not just names), so a rebuild can key the wiki tree.
|
||||
///
|
||||
/// # Errors
|
||||
/// Propagates any SQL or pool error.
|
||||
pub async fn list_all_scopes(&self) -> StoreResult<Vec<ScopeRow>> {
|
||||
// (ws_id, ws_name, proj_id, proj_name, repo_path) as raw SQL columns.
|
||||
type RawScope = (Vec<u8>, String, Vec<u8>, String, Option<String>);
|
||||
let raw: Vec<RawScope> = self
|
||||
.with_conn(|conn| {
|
||||
let mut stmt = conn.prepare(
|
||||
"SELECT w.id, w.name, p.id, p.name, p.repo_path \
|
||||
FROM projects p JOIN workspaces w ON w.id = p.workspace_id",
|
||||
)?;
|
||||
let rows = stmt.query_map([], |row| {
|
||||
Ok((
|
||||
row.get(0)?,
|
||||
row.get(1)?,
|
||||
row.get(2)?,
|
||||
row.get(3)?,
|
||||
row.get(4)?,
|
||||
))
|
||||
})?;
|
||||
let out: rusqlite::Result<Vec<_>> = rows.collect();
|
||||
Ok(out?)
|
||||
})
|
||||
.await?;
|
||||
raw.into_iter()
|
||||
.map(|(wi, wn, pi, pn, rp)| {
|
||||
Ok(ScopeRow {
|
||||
workspace_id: WorkspaceId::from_slice(&wi)?,
|
||||
workspace_name: wn,
|
||||
project_id: ProjectId::from_slice(&pi)?,
|
||||
project_name: pn,
|
||||
repo_path: rp,
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return every workspace with its id and name so manifest backfill can
|
||||
/// describe even empty workspaces that have no project rows yet.
|
||||
///
|
||||
/// # Errors
|
||||
/// Propagates any SQL or pool error.
|
||||
pub async fn list_all_workspace_scopes(&self) -> StoreResult<Vec<WorkspaceScopeRow>> {
|
||||
let raw: Vec<(Vec<u8>, String)> = self
|
||||
.with_conn(|conn| {
|
||||
let mut stmt = conn.prepare("SELECT id, name FROM workspaces")?;
|
||||
let rows = stmt.query_map([], |row| Ok((row.get(0)?, row.get(1)?)))?;
|
||||
let out: rusqlite::Result<Vec<_>> = rows.collect();
|
||||
Ok(out?)
|
||||
})
|
||||
.await?;
|
||||
raw.into_iter()
|
||||
.map(|(wi, wn)| {
|
||||
Ok(WorkspaceScopeRow {
|
||||
workspace_id: WorkspaceId::from_slice(&wi)?,
|
||||
workspace_name: wn,
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return counts that must be zero before `ai-memory reindex` runs.
|
||||
///
|
||||
/// `reindex` is a rebuild-from-files operation, not an in-place repair of a
|
||||
/// dirty DB. If rows already exist, stale derived rows or DB-only episodic
|
||||
/// state could survive and contradict the "rebuilt from wiki" contract.
|
||||
///
|
||||
/// # Errors
|
||||
/// Propagates any SQL or pool error.
|
||||
pub async fn reindex_target_status(&self) -> StoreResult<ReindexTargetStatus> {
|
||||
self.with_conn(|conn| {
|
||||
Ok(conn.query_row(
|
||||
"SELECT \
|
||||
(SELECT COUNT(*) FROM workspaces), \
|
||||
(SELECT COUNT(*) FROM projects), \
|
||||
(SELECT COUNT(*) FROM pages), \
|
||||
(SELECT COUNT(*) FROM links), \
|
||||
(SELECT COUNT(*) FROM page_embeddings), \
|
||||
(SELECT COUNT(*) FROM sessions), \
|
||||
(SELECT COUNT(*) FROM observations), \
|
||||
(SELECT COUNT(*) FROM handoffs), \
|
||||
(SELECT COUNT(*) FROM users), \
|
||||
(SELECT COUNT(*) FROM audit_log)",
|
||||
[],
|
||||
|row| {
|
||||
Ok(ReindexTargetStatus {
|
||||
workspaces: row.get(0)?,
|
||||
projects: row.get(1)?,
|
||||
pages: row.get(2)?,
|
||||
links: row.get(3)?,
|
||||
page_embeddings: row.get(4)?,
|
||||
sessions: row.get(5)?,
|
||||
observations: row.get(6)?,
|
||||
handoffs: row.get(7)?,
|
||||
users: row.get(8)?,
|
||||
audit_log: row.get(9)?,
|
||||
})
|
||||
},
|
||||
)?)
|
||||
})
|
||||
.await
|
||||
}
|
||||
|
||||
/// Return one row per workspace with project/page-count and
|
||||
/// last-updated aggregates. Used by custom frontends that need a
|
||||
/// workspace chooser before narrowing into projects.
|
||||
|
||||
@@ -38,6 +38,18 @@ pub(crate) enum WriteCmd {
|
||||
project_id: ProjectId,
|
||||
reply: oneshot::Sender<StoreResult<()>>,
|
||||
},
|
||||
EnsureWorkspaceWithId {
|
||||
id: WorkspaceId,
|
||||
name: String,
|
||||
reply: oneshot::Sender<StoreResult<()>>,
|
||||
},
|
||||
EnsureProjectWithId {
|
||||
id: ProjectId,
|
||||
workspace_id: WorkspaceId,
|
||||
name: String,
|
||||
repo_path: Option<String>,
|
||||
reply: oneshot::Sender<StoreResult<()>>,
|
||||
},
|
||||
UpsertPage {
|
||||
page: NewPage,
|
||||
reply: oneshot::Sender<StoreResult<PageId>>,
|
||||
@@ -269,6 +281,51 @@ impl WriterHandle {
|
||||
rx.await.map_err(|_| StoreError::WriterClosed)?
|
||||
}
|
||||
|
||||
/// Insert a workspace with an **explicit id** (idempotent). Used by
|
||||
/// `reindex` to recreate a scope under the id recovered from the wiki
|
||||
/// directory name. See [`ops::ensure_workspace_with_id`].
|
||||
///
|
||||
/// # Errors
|
||||
/// Returns [`StoreError::WriterClosed`] or propagates SQL errors.
|
||||
pub async fn ensure_workspace_with_id(
|
||||
&self,
|
||||
id: WorkspaceId,
|
||||
name: impl Into<String>,
|
||||
) -> StoreResult<()> {
|
||||
let (tx, rx) = oneshot::channel();
|
||||
self.send(WriteCmd::EnsureWorkspaceWithId {
|
||||
id,
|
||||
name: name.into(),
|
||||
reply: tx,
|
||||
})
|
||||
.await?;
|
||||
rx.await.map_err(|_| StoreError::WriterClosed)?
|
||||
}
|
||||
|
||||
/// Insert a project with an **explicit id** under `workspace_id`
|
||||
/// (idempotent). See [`ops::ensure_project_with_id`].
|
||||
///
|
||||
/// # Errors
|
||||
/// Returns [`StoreError::WriterClosed`] or propagates SQL errors.
|
||||
pub async fn ensure_project_with_id(
|
||||
&self,
|
||||
id: ProjectId,
|
||||
workspace_id: WorkspaceId,
|
||||
name: impl Into<String>,
|
||||
repo_path: Option<String>,
|
||||
) -> StoreResult<()> {
|
||||
let (tx, rx) = oneshot::channel();
|
||||
self.send(WriteCmd::EnsureProjectWithId {
|
||||
id,
|
||||
workspace_id,
|
||||
name: name.into(),
|
||||
repo_path,
|
||||
reply: tx,
|
||||
})
|
||||
.await?;
|
||||
rx.await.map_err(|_| StoreError::WriterClosed)?
|
||||
}
|
||||
|
||||
/// Begin a session (idempotent on the supplied id).
|
||||
///
|
||||
/// # Errors
|
||||
@@ -786,6 +843,26 @@ fn worker_loop(mut conn: Connection, mut rx: mpsc::Receiver<WriteCmd>) {
|
||||
let result = ops::ensure_project_workspace(&conn, &workspace_id, &project_id);
|
||||
send_or_warn(reply, result, "ensure_project_workspace");
|
||||
}
|
||||
WriteCmd::EnsureWorkspaceWithId { id, name, reply } => {
|
||||
let result = ops::ensure_workspace_with_id(&mut conn, id, &name);
|
||||
send_or_warn(reply, result, "ensure_workspace_with_id");
|
||||
}
|
||||
WriteCmd::EnsureProjectWithId {
|
||||
id,
|
||||
workspace_id,
|
||||
name,
|
||||
repo_path,
|
||||
reply,
|
||||
} => {
|
||||
let result = ops::ensure_project_with_id(
|
||||
&mut conn,
|
||||
id,
|
||||
workspace_id,
|
||||
&name,
|
||||
repo_path.as_deref(),
|
||||
);
|
||||
send_or_warn(reply, result, "ensure_project_with_id");
|
||||
}
|
||||
WriteCmd::UpsertPage { page, reply } => {
|
||||
let result = ops::upsert_page(&mut conn, &page);
|
||||
send_or_warn(reply, result, "upsert_page");
|
||||
|
||||
@@ -8,7 +8,8 @@
|
||||
//! Own-writes are absorbed by the store's sha256 short-circuit, so
|
||||
//! the loop terminates after one no-op reindex.
|
||||
//! 2. **Reconciliation tick** every 30s walks the entire wiki tree and
|
||||
//! reindexes every markdown file. Catches any events the OS dropped
|
||||
//! reindexes page markdown files (excluding `_meta.md`, bootstrap files,
|
||||
//! raw event ledgers, and symlinks). Catches any events the OS dropped
|
||||
//! (basic-memory #580 — file watchers go stale under FSEvents buffer
|
||||
//! overflow, hidden-dir globs, etc.). Hidden-directory paths are
|
||||
//! explicitly NOT skipped (#798 lesson).
|
||||
@@ -203,13 +204,24 @@ async fn handle_event(wiki: &Wiki, event: notify_debouncer_full::DebouncedEvent)
|
||||
return;
|
||||
}
|
||||
for raw_path in &event.paths {
|
||||
if raw_path.is_dir() {
|
||||
let Ok(metadata) = std::fs::symlink_metadata(raw_path) else {
|
||||
// Likely a transient state (mv, atomic rename in flight).
|
||||
continue;
|
||||
};
|
||||
let ft = metadata.file_type();
|
||||
if ft.is_symlink() {
|
||||
continue;
|
||||
}
|
||||
if ft.is_dir() {
|
||||
let Some((ws, proj, proj_root)) = extract_project_dir_ids(wiki.root(), raw_path) else {
|
||||
continue;
|
||||
};
|
||||
reindex_project_dir(wiki, ws, proj, proj_root).await;
|
||||
continue;
|
||||
}
|
||||
if !ft.is_file() {
|
||||
continue;
|
||||
}
|
||||
if !is_markdown(raw_path) {
|
||||
continue;
|
||||
}
|
||||
@@ -219,11 +231,7 @@ async fn handle_event(wiki: &Wiki, event: notify_debouncer_full::DebouncedEvent)
|
||||
let Some((ws, proj, page_path)) = extract_project_ids(wiki.root(), raw_path) else {
|
||||
continue;
|
||||
};
|
||||
if is_reserved_filename(&page_path) {
|
||||
continue;
|
||||
}
|
||||
if !raw_path.is_file() {
|
||||
// Likely a transient state (mv, atomic rename in flight).
|
||||
if is_reserved_page_file(raw_path, &page_path) {
|
||||
continue;
|
||||
}
|
||||
match wiki.reindex_page(ws, proj, page_path.clone()).await {
|
||||
@@ -284,7 +292,7 @@ async fn reconcile(wiki: &Wiki) -> WikiResult<()> {
|
||||
|
||||
/// Walk `<wiki_root>` and return all `(WorkspaceId, ProjectId, proj_root)` tuples
|
||||
/// whose first two path segments parse as valid UUIDs.
|
||||
fn walk_project_dirs(
|
||||
pub(crate) fn walk_project_dirs(
|
||||
wiki_root: &Path,
|
||||
) -> WikiResult<Vec<(WorkspaceId, ProjectId, std::path::PathBuf)>> {
|
||||
let mut out = Vec::new();
|
||||
@@ -376,7 +384,7 @@ fn extract_project_dir_ids(
|
||||
Some((ws_id, proj_id, wiki_root.join(ws_seg).join(proj_seg)))
|
||||
}
|
||||
|
||||
fn walk_markdown(root: &Path) -> WikiResult<Vec<PagePath>> {
|
||||
pub(crate) fn walk_markdown(root: &Path) -> WikiResult<Vec<PagePath>> {
|
||||
let mut out = Vec::new();
|
||||
let mut stack = vec![root.to_path_buf()];
|
||||
while let Some(dir) = stack.pop() {
|
||||
@@ -404,7 +412,7 @@ fn walk_markdown(root: &Path) -> WikiResult<Vec<PagePath>> {
|
||||
&& is_markdown(&path)
|
||||
&& !is_tempfile(&path)
|
||||
&& let Some(pp) = page_path_relative_to(root, &path)
|
||||
&& !is_reserved_filename(&pp)
|
||||
&& !is_reserved_page_file(&path, &pp)
|
||||
{
|
||||
out.push(pp);
|
||||
}
|
||||
@@ -423,19 +431,75 @@ fn is_tempfile(path: &Path) -> bool {
|
||||
.is_some_and(|n| n.starts_with(".ai-memory-tmp."))
|
||||
}
|
||||
|
||||
/// Returns `true` for the per-project reserved files that are NOT wiki
|
||||
/// pages: `log-YYYY-MM.md` (per-month rolling event ledger — see
|
||||
/// `ai-memory-hooks::log::log_filename_for`), the legacy `log.md`
|
||||
/// filename for backwards compatibility, and `bootstrap.md` (manifest).
|
||||
/// The watcher must skip them to avoid supersession loops — every
|
||||
/// `append_event` write triggers a watcher event, triggering an
|
||||
/// `upsert_page`, creating a spurious new supersession version.
|
||||
fn is_reserved_filename(page_path: &PagePath) -> bool {
|
||||
/// `_meta.md` is the per-scope manifest the engine writes (workspace/project
|
||||
/// name + repo_path) so the wiki tree is self-describing. It describes the
|
||||
/// scope, it is never a wiki page.
|
||||
fn is_manifest_filename(page_path: &PagePath) -> bool {
|
||||
page_path
|
||||
.as_str()
|
||||
.rsplit('/')
|
||||
.next()
|
||||
.is_some_and(|name| name == "_meta.md")
|
||||
}
|
||||
|
||||
/// `log.md` / `log-YYYY-MM.md` are the raw per-project event ledger the hooks
|
||||
/// append to (see `ai-memory-hooks::log::log_filename_for`): `## [ts] ...`
|
||||
/// entries, never YAML frontmatter.
|
||||
fn is_log_ledger_filename(page_path: &PagePath) -> bool {
|
||||
let s = page_path.as_str();
|
||||
s == "log.md"
|
||||
|| s == "bootstrap.md"
|
||||
// log-YYYY-MM.md — 7 chars between the dash and the dot.
|
||||
|| (s.starts_with("log-") && s.ends_with(".md") && s.len() == "log-YYYY-MM.md".len())
|
||||
s == "log.md" || is_rotated_log_filename(s)
|
||||
}
|
||||
|
||||
fn is_rotated_log_filename(s: &str) -> bool {
|
||||
let Some(stem) = s.strip_prefix("log-").and_then(|v| v.strip_suffix(".md")) else {
|
||||
return false;
|
||||
};
|
||||
let bytes = stem.as_bytes();
|
||||
bytes.len() == "YYYY-MM".len()
|
||||
&& bytes[4] == b'-'
|
||||
&& bytes[..4].iter().all(|b| b.is_ascii_digit())
|
||||
&& bytes[5..].iter().all(|b| b.is_ascii_digit())
|
||||
}
|
||||
|
||||
/// Cheap peek: does the file open with a `---` YAML frontmatter fence?
|
||||
/// Used to tell a real page apart from the raw event ledger.
|
||||
fn opens_with_frontmatter(abs: &Path) -> bool {
|
||||
use std::io::{BufRead, BufReader};
|
||||
let Ok(file) = std::fs::File::open(abs) else {
|
||||
return false;
|
||||
};
|
||||
let mut line = String::new();
|
||||
BufReader::new(file).read_line(&mut line).is_ok() && line.trim_end() == "---"
|
||||
}
|
||||
|
||||
/// Cheap check for the raw hook event ledger shape. Real page markdown can be
|
||||
/// frontmatter-free; a reserved-looking filename is only a ledger when the
|
||||
/// content starts with the hook log prefix.
|
||||
fn opens_with_log_ledger(abs: &Path) -> bool {
|
||||
use std::io::{BufRead, BufReader};
|
||||
let Ok(file) = std::fs::File::open(abs) else {
|
||||
return false;
|
||||
};
|
||||
let mut line = String::new();
|
||||
BufReader::new(file)
|
||||
.read_line(&mut line)
|
||||
.is_ok_and(|_| line.starts_with("## ["))
|
||||
}
|
||||
|
||||
/// Returns `true` for markdown files that are NOT wiki pages and must be
|
||||
/// skipped by the indexer:
|
||||
/// - `_meta.md` (the self-describing scope manifest) and `bootstrap.md` —
|
||||
/// always; and
|
||||
/// - the raw event ledger (`log.md` / exact `log-YYYY-MM.md`) — skipping which
|
||||
/// avoids supersession loops, since every `append_event` write triggers a
|
||||
/// watcher event. A reserved-looking filename is skipped only when its
|
||||
/// content opens with the raw hook log prefix; ordinary markdown pages with
|
||||
/// those names are indexed.
|
||||
fn is_reserved_page_file(abs: &Path, page_path: &PagePath) -> bool {
|
||||
if is_manifest_filename(page_path) || page_path.as_str() == "bootstrap.md" {
|
||||
return true;
|
||||
}
|
||||
is_log_ledger_filename(page_path) && !opens_with_frontmatter(abs) && opens_with_log_ledger(abs)
|
||||
}
|
||||
|
||||
fn page_path_relative_to(root: &Path, abs: &Path) -> Option<PagePath> {
|
||||
@@ -669,10 +733,19 @@ mod tests {
|
||||
// Single-word unique tokens so FTS5 (which parses hyphens as
|
||||
// operators) can match them.
|
||||
std::fs::write(proj_dir.join("real.md"), "real content\n").unwrap();
|
||||
std::fs::write(proj_dir.join("log.md"), "sessionstarted logtoken unique\n").unwrap();
|
||||
std::fs::write(
|
||||
proj_dir.join("log.md"),
|
||||
"## [2026-06-08T12:34:56Z] session-start | logtoken unique\n",
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(
|
||||
proj_dir.join("log-2026-05.md"),
|
||||
"sessionstarted rotatedlogtoken unique\n",
|
||||
"## [2026-05-01T00:00:00Z] user-prompt | rotatedlogtoken unique\n",
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(
|
||||
proj_dir.join("log-summary.md"),
|
||||
"ordinary markdown summaries regularlogtoken unique\n",
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(
|
||||
@@ -712,6 +785,18 @@ mod tests {
|
||||
"log-YYYY-MM.md (rotated) must not be indexed"
|
||||
);
|
||||
|
||||
let regular_hits = store
|
||||
.reader
|
||||
.search_pages("regularlogtoken".into(), 5)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
regular_hits.len(),
|
||||
1,
|
||||
"ordinary log-looking markdown must still be indexed"
|
||||
);
|
||||
assert_eq!(regular_hits[0].path.as_str(), "log-summary.md");
|
||||
|
||||
let boot_hits = store
|
||||
.reader
|
||||
.search_pages("boottoken".into(), 5)
|
||||
@@ -723,6 +808,80 @@ mod tests {
|
||||
drop(store);
|
||||
}
|
||||
|
||||
/// A page that *collides* with a reserved ledger name (`log.md`) but
|
||||
/// carries YAML frontmatter is a real page and MUST be indexed — not
|
||||
/// silently dropped. Regression for a prod data anomaly (a page lived at
|
||||
/// `log.md`) that a filename-only skip would lose on every reindex. The
|
||||
/// `_meta.md` manifest, by contrast, must NEVER be indexed even though it
|
||||
/// also has frontmatter.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn reindex_page_by_content_not_filename() {
|
||||
let (tmp, store, wiki, ws, proj) = setup().await;
|
||||
let proj_dir = tmp
|
||||
.path()
|
||||
.join("wiki")
|
||||
.join(ws.to_string())
|
||||
.join(proj.to_string());
|
||||
std::fs::create_dir_all(&proj_dir).unwrap();
|
||||
|
||||
// A genuine page that happens to live at `log.md` (has frontmatter).
|
||||
std::fs::write(
|
||||
proj_dir.join("log.md"),
|
||||
"---\ntitle: Collides With Ledger\n---\nframmaticpage uniquetoken\n",
|
||||
)
|
||||
.unwrap();
|
||||
// The self-describing manifest — never a page, even with frontmatter.
|
||||
std::fs::write(
|
||||
proj_dir.join("_meta.md"),
|
||||
"---\nworkspace: default\nproject: scratch\n---\nmanifesttoken here\n",
|
||||
)
|
||||
.unwrap();
|
||||
// A raw ledger (no frontmatter) — still skipped.
|
||||
std::fs::write(
|
||||
proj_dir.join("log-2026-06.md"),
|
||||
"## [t] evt | x\nrawledgertoken\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let handle = WatcherHandle::start(wiki.clone()).unwrap();
|
||||
reconcile(&wiki).await.unwrap();
|
||||
|
||||
let page_hits = store
|
||||
.reader
|
||||
.search_pages("uniquetoken".into(), 5)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
page_hits.len(),
|
||||
1,
|
||||
"frontmatter page named log.md must be indexed"
|
||||
);
|
||||
assert_eq!(page_hits[0].path.as_str(), "log.md");
|
||||
|
||||
let meta_hits = store
|
||||
.reader
|
||||
.search_pages("manifesttoken".into(), 5)
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(
|
||||
meta_hits.is_empty(),
|
||||
"_meta.md manifest must not be indexed"
|
||||
);
|
||||
|
||||
let ledger_hits = store
|
||||
.reader
|
||||
.search_pages("rawledgertoken".into(), 5)
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(
|
||||
ledger_hits.is_empty(),
|
||||
"raw ledger (no frontmatter) must not be indexed"
|
||||
);
|
||||
|
||||
handle.shutdown().await;
|
||||
drop(store);
|
||||
}
|
||||
|
||||
/// Defence: an attacker who can write to wiki/ shouldn't be able
|
||||
/// to make the watcher index arbitrary files via symlinks.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
@@ -752,4 +911,42 @@ mod tests {
|
||||
"symlink to outside file must be skipped; got: {names:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Direct notify events must use the same symlink guard as full-tree walks;
|
||||
/// otherwise a symlinked markdown file can be opened before reconciliation
|
||||
/// gets a chance to skip it.
|
||||
#[cfg(any(unix, windows))]
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn direct_file_event_skips_symlink() {
|
||||
let (tmp, store, wiki, ws, proj) = setup().await;
|
||||
let proj_dir = tmp
|
||||
.path()
|
||||
.join("wiki")
|
||||
.join(ws.to_string())
|
||||
.join(proj.to_string());
|
||||
std::fs::create_dir_all(&proj_dir).unwrap();
|
||||
|
||||
let secret = tmp.path().join("outside-secret.md");
|
||||
std::fs::write(&secret, "directsymlinksecret should not index\n").unwrap();
|
||||
|
||||
let symlink = proj_dir.join("symlinked.md");
|
||||
#[cfg(unix)]
|
||||
std::os::unix::fs::symlink(&secret, &symlink).unwrap();
|
||||
#[cfg(windows)]
|
||||
std::os::windows::fs::symlink_file(&secret, &symlink).unwrap();
|
||||
|
||||
let event = notify_debouncer_full::DebouncedEvent::new(
|
||||
notify::Event::new(EventKind::Create(notify::event::CreateKind::File))
|
||||
.add_path(symlink),
|
||||
std::time::Instant::now(),
|
||||
);
|
||||
handle_event(&wiki, event).await;
|
||||
|
||||
let hits = store
|
||||
.reader
|
||||
.search_pages("directsymlinksecret".into(), 5)
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(hits.is_empty(), "direct symlink event must not be indexed");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,10 +10,21 @@ use tokio::sync::RwLock;
|
||||
|
||||
use crate::admission::{AdmissionChain, AdmissionContext, AdmissionOp};
|
||||
use crate::atomic;
|
||||
use crate::error::WikiResult;
|
||||
use crate::error::{WikiError, WikiResult};
|
||||
use crate::git::GitAdapter;
|
||||
use crate::markdown::{Markdown, derive_title, emit, extract_links, parse};
|
||||
|
||||
/// Summary of a [`Wiki::reindex_all`] run.
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct ReindexSummary {
|
||||
/// Workspaces recreated from `_meta.md`.
|
||||
pub workspaces: usize,
|
||||
/// Projects recreated from `_meta.md`.
|
||||
pub projects: usize,
|
||||
/// Pages reindexed from the wiki tree.
|
||||
pub pages: usize,
|
||||
}
|
||||
|
||||
/// Wiki filesystem handle.
|
||||
///
|
||||
/// Owns the path of the wiki root (`<data_dir>/wiki/`) and a cloneable
|
||||
@@ -410,7 +421,7 @@ impl Wiki {
|
||||
let ctx = self
|
||||
.admit_purge_project(workspace_id, project_id, admission_ctx)
|
||||
.await?;
|
||||
self.remove_project_dir(workspace_id, project_id)?;
|
||||
self.remove_project_dir(workspace_id, project_id).await?;
|
||||
self.dispatch_purge_project(ctx.as_ref());
|
||||
Ok(())
|
||||
}
|
||||
@@ -443,11 +454,12 @@ impl Wiki {
|
||||
///
|
||||
/// # Errors
|
||||
/// Returns [`WikiError::Io`] on filesystem errors other than NotFound.
|
||||
pub fn remove_project_dir(
|
||||
pub async fn remove_project_dir(
|
||||
&self,
|
||||
workspace_id: WorkspaceId,
|
||||
project_id: ProjectId,
|
||||
) -> WikiResult<()> {
|
||||
let _guard = self.mutation_lock.write().await;
|
||||
let root = self.project_root(workspace_id, project_id);
|
||||
match std::fs::remove_dir_all(&root) {
|
||||
Ok(()) => Ok(()),
|
||||
@@ -513,6 +525,166 @@ impl Wiki {
|
||||
Ok(id)
|
||||
}
|
||||
|
||||
/// Read a `_meta.md` scope-manifest's frontmatter from `dir`.
|
||||
fn read_scope_meta(dir: &Path) -> WikiResult<serde_json::Value> {
|
||||
let path = dir.join("_meta.md");
|
||||
let meta = std::fs::symlink_metadata(&path)?;
|
||||
if meta.file_type().is_symlink() {
|
||||
return Err(WikiError::Io(std::io::Error::other(format!(
|
||||
"refusing to read symlinked scope manifest {}",
|
||||
path.display()
|
||||
))));
|
||||
}
|
||||
let raw = std::fs::read_to_string(path)?;
|
||||
Ok(parse(&raw)?.frontmatter)
|
||||
}
|
||||
|
||||
/// Rebuild the **entire** store index from the on-disk wiki tree — the
|
||||
/// "DB is rebuildable from files" guarantee made concrete. Walks every
|
||||
/// `<ws-uuid>/<proj-uuid>/` directory, recreates the workspace/project rows
|
||||
/// from each dir's self-describing `_meta.md` manifest (preserving the ids
|
||||
/// the tree is keyed by, via [`WriterHandle::ensure_workspace_with_id`] /
|
||||
/// [`ensure_project_with_id`]), then reindexes every page. Pages are
|
||||
/// detected by content (a frontmatter file named `log.md` is a page; the
|
||||
/// raw `## [..]` ledger, `_meta.md` and `bootstrap.md` are skipped).
|
||||
///
|
||||
/// Intended for a freshly-migrated (clean) store, e.g. to move a data dir
|
||||
/// onto a different migration lineage without carrying the old
|
||||
/// `refinery_schema_history`. DB-only episodic state (sessions,
|
||||
/// observations, handoffs, decay counters) is NOT reconstructed — it is not
|
||||
/// in the markdown; embeddings can be recomputed separately via `embed`.
|
||||
///
|
||||
/// # Errors
|
||||
/// Returns [`WikiError`] for filesystem/parse/store errors, including a
|
||||
/// scope directory that lacks its `_meta.md` (the wiki is not
|
||||
/// self-describing — newer engines write the manifest on scope creation).
|
||||
pub async fn reindex_all(&self) -> WikiResult<ReindexSummary> {
|
||||
let root = self.root().to_path_buf();
|
||||
let project_dirs =
|
||||
tokio::task::spawn_blocking(move || crate::watcher::walk_project_dirs(&root))
|
||||
.await
|
||||
.map_err(|e| WikiError::Io(std::io::Error::other(e.to_string())))??;
|
||||
|
||||
let mut summary = ReindexSummary::default();
|
||||
let mut seen_ws = std::collections::HashSet::new();
|
||||
|
||||
for (ws, proj, proj_root) in project_dirs {
|
||||
if seen_ws.insert(ws) {
|
||||
let ws_dir = proj_root
|
||||
.parent()
|
||||
.unwrap_or(proj_root.as_path())
|
||||
.to_path_buf();
|
||||
let meta = Self::read_scope_meta(&ws_dir)?;
|
||||
let name = meta
|
||||
.get("workspace")
|
||||
.and_then(|v| v.as_str())
|
||||
.ok_or_else(|| {
|
||||
WikiError::Io(std::io::Error::other(format!(
|
||||
"{}/_meta.md is missing the `workspace` name",
|
||||
ws_dir.display()
|
||||
)))
|
||||
})?;
|
||||
self.writer.ensure_workspace_with_id(ws, name).await?;
|
||||
summary.workspaces += 1;
|
||||
}
|
||||
|
||||
let meta = Self::read_scope_meta(&proj_root)?;
|
||||
let name = meta
|
||||
.get("project")
|
||||
.and_then(|v| v.as_str())
|
||||
.ok_or_else(|| {
|
||||
WikiError::Io(std::io::Error::other(format!(
|
||||
"{}/_meta.md is missing the `project` name",
|
||||
proj_root.display()
|
||||
)))
|
||||
})?;
|
||||
let repo_path = meta
|
||||
.get("repo_path")
|
||||
.and_then(|v| v.as_str())
|
||||
.map(String::from);
|
||||
self.writer
|
||||
.ensure_project_with_id(proj, ws, name, repo_path)
|
||||
.await?;
|
||||
summary.projects += 1;
|
||||
|
||||
let pr = proj_root.clone();
|
||||
let pages = tokio::task::spawn_blocking(move || crate::watcher::walk_markdown(&pr))
|
||||
.await
|
||||
.map_err(|e| WikiError::Io(std::io::Error::other(e.to_string())))??;
|
||||
for path in pages {
|
||||
self.reindex_page(ws, proj, path).await?;
|
||||
summary.pages += 1;
|
||||
}
|
||||
}
|
||||
Ok(summary)
|
||||
}
|
||||
|
||||
/// Write a `_meta.md` scope manifest under `dir` from `frontmatter`,
|
||||
/// idempotently — unchanged content is left untouched so a startup
|
||||
/// backfill never churns the wiki git history. Returns `true` if written.
|
||||
fn write_scope_manifest(dir: &Path, frontmatter: serde_json::Value) -> WikiResult<bool> {
|
||||
let content = emit(&Markdown {
|
||||
frontmatter,
|
||||
body: String::new(),
|
||||
})?;
|
||||
let path = dir.join("_meta.md");
|
||||
match std::fs::symlink_metadata(&path) {
|
||||
Ok(meta) if meta.file_type().is_symlink() => {
|
||||
return Err(WikiError::Io(std::io::Error::other(format!(
|
||||
"refusing to update symlinked scope manifest {}",
|
||||
path.display()
|
||||
))));
|
||||
}
|
||||
Ok(meta) if meta.is_file() => {
|
||||
if std::fs::read_to_string(&path).is_ok_and(|existing| existing == content) {
|
||||
return Ok(false);
|
||||
}
|
||||
}
|
||||
Ok(_) | Err(_) => {}
|
||||
}
|
||||
std::fs::create_dir_all(dir)?;
|
||||
crate::atomic::write_atomic(&path, content.as_bytes())?;
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
/// Ensure every workspace/project scope has its self-describing `_meta.md`
|
||||
/// manifest on disk (`workspace`/`project` name + `repo_path`), so the wiki
|
||||
/// tree alone is enough to rebuild the index via [`Self::reindex_all`] —
|
||||
/// the "DB is rebuildable from files" guarantee. Idempotent; safe to run on
|
||||
/// every startup. No-op without a store reader. Returns the count written.
|
||||
///
|
||||
/// # Errors
|
||||
/// Returns [`WikiError`] for store or filesystem errors.
|
||||
pub async fn backfill_scope_manifests(&self) -> WikiResult<usize> {
|
||||
let Some(reader) = &self.store_reader else {
|
||||
return Ok(0);
|
||||
};
|
||||
let _guard = self.mutation_lock.write().await;
|
||||
let workspaces = reader.list_all_workspace_scopes().await?;
|
||||
let scopes = reader.list_all_scopes().await?;
|
||||
let mut written = 0;
|
||||
for ws in workspaces {
|
||||
let ws_dir = self.root().join(ws.workspace_id.to_string());
|
||||
if Self::write_scope_manifest(
|
||||
&ws_dir,
|
||||
serde_json::json!({ "workspace": ws.workspace_name }),
|
||||
)? {
|
||||
written += 1;
|
||||
}
|
||||
}
|
||||
for s in scopes {
|
||||
let ws_dir = self.root().join(s.workspace_id.to_string());
|
||||
let mut fm = serde_json::json!({ "project": s.project_name });
|
||||
if let Some(rp) = s.repo_path {
|
||||
fm["repo_path"] = serde_json::Value::String(rp);
|
||||
}
|
||||
if Self::write_scope_manifest(&ws_dir.join(s.project_id.to_string()), fm)? {
|
||||
written += 1;
|
||||
}
|
||||
}
|
||||
Ok(written)
|
||||
}
|
||||
|
||||
/// Atomically apply a batch of page writes. Either all pages land
|
||||
/// (one SQL transaction) and their files are renamed into place,
|
||||
/// or no DB row changes and tempfiles are dropped.
|
||||
@@ -1654,4 +1826,144 @@ mod tests {
|
||||
);
|
||||
assert_eq!(md.frontmatter["title"], "Anon");
|
||||
}
|
||||
|
||||
fn copy_tree(src: &Path, dst: &Path) {
|
||||
std::fs::create_dir_all(dst).unwrap();
|
||||
for entry in std::fs::read_dir(src).unwrap() {
|
||||
let entry = entry.unwrap();
|
||||
let to = dst.join(entry.file_name());
|
||||
if entry.file_type().unwrap().is_dir() {
|
||||
copy_tree(&entry.path(), &to);
|
||||
} else {
|
||||
std::fs::copy(entry.path(), &to).unwrap();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// End-to-end "DB is rebuildable from files": `backfill_scope_manifests`
|
||||
/// makes the wiki self-describing, then `reindex_all` on a FRESH store
|
||||
/// (no DB carried over) recreates the named scopes + all pages from the
|
||||
/// wiki tree alone — including a page that lives at the reserved name
|
||||
/// `log.md` (kept because it has frontmatter).
|
||||
#[tokio::test]
|
||||
async fn backfill_then_reindex_rebuilds_from_wiki_alone() {
|
||||
// Source store: a named scope with two pages (one at `log.md`).
|
||||
let src = TempDir::new().unwrap();
|
||||
let s1 = Store::open(src.path()).unwrap();
|
||||
let ws = s1.writer.get_or_create_workspace("acme").await.unwrap();
|
||||
let proj = s1
|
||||
.writer
|
||||
.get_or_create_project(ws, "webapp", Some("/repo/webapp".into()))
|
||||
.await
|
||||
.unwrap();
|
||||
let w1 = Wiki::new(src.path(), s1.writer.clone())
|
||||
.unwrap()
|
||||
.with_store_reader(s1.reader.clone());
|
||||
w1.apply_batch(vec![
|
||||
req(
|
||||
ws,
|
||||
proj,
|
||||
"notes/a.md",
|
||||
"alpha uniquetoken",
|
||||
serde_json::json!({}),
|
||||
),
|
||||
req(
|
||||
ws,
|
||||
proj,
|
||||
"log.md",
|
||||
"a page that lives at the reserved log name",
|
||||
serde_json::json!({ "title": "Log Page" }),
|
||||
),
|
||||
])
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Make the wiki self-describing.
|
||||
let written = w1.backfill_scope_manifests().await.unwrap();
|
||||
assert!(written >= 2, "ws + proj manifests written, got {written}");
|
||||
let ws_dir = src.path().join("wiki").join(ws.to_string());
|
||||
assert!(ws_dir.join("_meta.md").is_file());
|
||||
assert!(ws_dir.join(proj.to_string()).join("_meta.md").is_file());
|
||||
drop(s1);
|
||||
|
||||
// Fresh store; copy ONLY the wiki tree (no db/); rebuild from it.
|
||||
let dst = TempDir::new().unwrap();
|
||||
let s2 = Store::open(dst.path()).unwrap();
|
||||
copy_tree(&src.path().join("wiki"), &dst.path().join("wiki"));
|
||||
let w2 = Wiki::new(dst.path(), s2.writer.clone()).unwrap();
|
||||
let summary = w2.reindex_all().await.unwrap();
|
||||
|
||||
assert_eq!(summary.projects, 1);
|
||||
assert_eq!(summary.pages, 2, "both pages incl. log.md reconstructed");
|
||||
assert_eq!(
|
||||
s2.reader.workspace_name_by_id(ws).await.unwrap().as_deref(),
|
||||
Some("acme"),
|
||||
"workspace name recovered from _meta.md"
|
||||
);
|
||||
let hits = s2
|
||||
.reader
|
||||
.search_pages("uniquetoken".into(), 5)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
hits.len(),
|
||||
1,
|
||||
"reindexed page is searchable in the fresh store"
|
||||
);
|
||||
drop(s2);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn backfill_writes_manifest_for_empty_workspace() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = Store::open(tmp.path()).unwrap();
|
||||
let ws = store
|
||||
.writer
|
||||
.get_or_create_workspace("empty-ws")
|
||||
.await
|
||||
.unwrap();
|
||||
let wiki = Wiki::new(tmp.path(), store.writer.clone())
|
||||
.unwrap()
|
||||
.with_store_reader(store.reader.clone());
|
||||
|
||||
let written = wiki.backfill_scope_manifests().await.unwrap();
|
||||
|
||||
assert_eq!(written, 1);
|
||||
let meta = std::fs::read_to_string(
|
||||
tmp.path()
|
||||
.join("wiki")
|
||||
.join(ws.to_string())
|
||||
.join("_meta.md"),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(meta.contains("workspace: empty-ws"));
|
||||
}
|
||||
|
||||
#[cfg(any(unix, windows))]
|
||||
#[tokio::test]
|
||||
async fn reindex_rejects_symlinked_scope_manifest() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = Store::open(tmp.path()).unwrap();
|
||||
let wiki = Wiki::new(tmp.path(), store.writer.clone()).unwrap();
|
||||
let ws = WorkspaceId::new();
|
||||
let proj = ProjectId::new();
|
||||
let ws_dir = tmp.path().join("wiki").join(ws.to_string());
|
||||
let proj_dir = ws_dir.join(proj.to_string());
|
||||
std::fs::create_dir_all(&proj_dir).unwrap();
|
||||
std::fs::write(ws_dir.join("_meta.md"), "---\nworkspace: acme\n---\n").unwrap();
|
||||
|
||||
let outside = tmp.path().join("outside-meta.md");
|
||||
std::fs::write(&outside, "---\nproject: webapp\n---\n").unwrap();
|
||||
let link = proj_dir.join("_meta.md");
|
||||
#[cfg(unix)]
|
||||
std::os::unix::fs::symlink(&outside, &link).unwrap();
|
||||
#[cfg(windows)]
|
||||
std::os::windows::fs::symlink_file(&outside, &link).unwrap();
|
||||
|
||||
let err = wiki.reindex_all().await.unwrap_err();
|
||||
assert!(
|
||||
err.to_string().contains("symlinked scope manifest"),
|
||||
"reindex must reject symlinked manifests, got {err:#}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+42
-4
@@ -14,11 +14,12 @@ on a homelab box where mistakes are harder to undo.
|
||||
| `backup --output-path` | ✅ yes | no | n/a | Streams a gzipped tarball from the server's online `sqlite3 .backup` plus the wiki tree. Safe alongside the live writer. |
|
||||
| `restore --from <tarball>` | ❌ **stop the server first** | overwrites the data dir | no (without prior backup) | Refuses if any sibling `ai-memory` process is alive (sysinfo guard). |
|
||||
| `reset --confirm` | ❌ **stop the server first** | yes, all data | no | Refuses if any sibling `ai-memory` process is alive (sysinfo guard). |
|
||||
| `reindex` | ❌ **stop the server first** | no wiki wipe; requires a clean DB | only with prior DB backup | Rebuilds pages/links/FTS from `wiki/` using `_meta.md` manifests. Refuses if SQLite already has rows so stale DB-only state cannot survive silently. |
|
||||
|
||||
All five commands route through the HTTP admin API except `reset` and
|
||||
`restore`, which are direct-disk operations that fundamentally cannot
|
||||
run while another process holds the SQLite WAL writer. See [CLAUDE.md
|
||||
§16](../CLAUDE.md) for the invariant.
|
||||
State-touching commands route through the HTTP admin API except `reset`,
|
||||
`restore`, and `reindex`, which are direct-disk lifecycle operations that
|
||||
fundamentally cannot run while another process holds the SQLite WAL writer. See
|
||||
[CLAUDE.md §16](../CLAUDE.md) for the invariant.
|
||||
|
||||
## What "project isolation" means here
|
||||
|
||||
@@ -28,12 +29,14 @@ Every project's data lives under an isolated, UUID-keyed root on disk:
|
||||
<wiki_root>/
|
||||
├── .git/
|
||||
├── <workspace_id>/
|
||||
│ ├── _meta.md # workspace name for rebuilds
|
||||
│ └── <project_id>/
|
||||
│ ├── concepts/
|
||||
│ ├── decisions/
|
||||
│ ├── gotchas/
|
||||
│ ├── sessions/
|
||||
│ ├── _rules/
|
||||
│ ├── _meta.md # project name + repo_path for rebuilds
|
||||
│ ├── log-YYYY-MM.md # rolling event log, one file per month
|
||||
│ └── bootstrap.md
|
||||
└── <other_workspace_id>/
|
||||
@@ -53,6 +56,11 @@ as subtrees). A `git log` from inside the wiki dir shows changes
|
||||
across every project; per-project diffs are also possible via
|
||||
`git log -- <workspace_id>/<project_id>/`.
|
||||
|
||||
Each workspace directory also carries `<workspace_id>/_meta.md`, and each
|
||||
project directory carries `<workspace_id>/<project_id>/_meta.md`. Those small
|
||||
frontmatter-only manifests store human names (plus `repo_path` for projects),
|
||||
so a clean SQLite DB can be rebuilt from the UUID-keyed wiki tree alone.
|
||||
|
||||
## Command-by-command
|
||||
|
||||
### `purge-project`
|
||||
@@ -325,6 +333,36 @@ container - but `ai-memory reset` is the cross-platform path that
|
||||
works whether the data dir is local, bind-mounted, or in a named
|
||||
volume.
|
||||
|
||||
### `reindex`
|
||||
|
||||
```bash
|
||||
ai-memory reindex --data-dir <path>
|
||||
```
|
||||
|
||||
Direct-disk lifecycle operation. Refuses if any sibling `ai-memory` process is
|
||||
alive, and also refuses if SQLite already contains rows. `reindex` is a
|
||||
rebuild-from-files path, not an in-place dirty-index repair.
|
||||
|
||||
Use it when the markdown wiki is intact but you intentionally want a fresh
|
||||
SQLite migration lineage:
|
||||
|
||||
1. Stop the server or container.
|
||||
2. Take a backup of the current data directory.
|
||||
3. Move or remove `<data-dir>/db/memory.sqlite` and its WAL/SHM siblings.
|
||||
4. Run `ai-memory reindex --data-dir <data-dir>`.
|
||||
5. Run `ai-memory embed` after restart if you need embeddings rebuilt.
|
||||
|
||||
What is rebuilt:
|
||||
|
||||
- Workspaces and projects from `_meta.md`, preserving the UUIDs encoded in the
|
||||
wiki directory names.
|
||||
- Latest page rows, page links, and FTS from markdown files.
|
||||
|
||||
What is not rebuilt:
|
||||
|
||||
- Sessions, observations, handoffs, users/tokens, audit rows, access counters,
|
||||
and embeddings. Those are DB-only state; keep a backup if you need them.
|
||||
|
||||
## Operator workflows
|
||||
|
||||
### "Fresh start" (wipe everything)
|
||||
|
||||
Reference in New Issue
Block a user