Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,9 @@

## Unreleased

- Group curation sessions by routed topic pages instead of workspace folder names.
The nightly curator asks the model which topic each container-folder session is
about before building the backlog.
- Label pages and links written by the local curator as `local-curator`.

## 0.13.0 - 2026-09-29
Expand Down
4 changes: 4 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -227,6 +227,10 @@ wikibricks curate --prune-archived-sessions-after-days 90

## Nightly curation (optional)

Sessions captured under container folders such as `~/code` do not identify their
topic. Before building the backlog, the curator asks the model to route those
sessions to an existing or new page, subject to `curation.generic_workspaces`.

`wikibricks-curator propose` asks one Databricks model for updates to the top
local backlog projects. It writes those proposals as a curation run, applies
low-risk groups with the safe policy, and leaves high-risk or conflicting
Expand Down
27 changes: 27 additions & 0 deletions src/wikibricks/config/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
"database": {"path": None, "url": None},
"search": {"default_results": None, "maximum_results": None},
"maintenance": {"prune_archived_sessions_after_days": None},
"curation": {"generic_workspaces": None},
"automation": {
"enabled": None,
"poll_seconds": None,
Expand All @@ -41,6 +42,7 @@ class WikiBricksConfig:
search_default_results: int
search_maximum_results: int
prune_archived_sessions_after_days: int | None
curation_generic_workspaces: tuple[str, ...]
automation_enabled: bool
automation_poll_seconds: int
automation_local_maintenance_hours: int
Expand Down Expand Up @@ -115,6 +117,21 @@ def _string(value: Any, path: str, *, optional: bool = False) -> str | None:
return value.strip()


def _string_list(value: Any, path: str) -> list[str]:
if not isinstance(value, list) or not all(isinstance(item, str) for item in value):
raise ValueError(f"{path} must be a list of strings")
if any(not item.strip() for item in value):
raise ValueError(f"{path} must not contain empty strings")
return [item.strip() for item in value]


def _comma_separated_strings(value: str) -> list[str]:
values = [item.strip() for item in value.split(",")]
if any(not item for item in values):
raise ValueError("must be a comma-separated list of non-empty strings")
return values


def _environment_overlay(environ: Mapping[str, str]) -> dict[str, Any]:
result: dict[str, Any] = {}
mappings = {
Expand All @@ -127,6 +144,11 @@ def _environment_overlay(environ: Mapping[str, str]) -> dict[str, Any]:
"prune_archived_sessions_after_days",
int,
),
"WIKIBRICKS_CURATION_GENERIC_WORKSPACES": (
"curation",
"generic_workspaces",
_comma_separated_strings,
),
"WIKIBRICKS_SYNC_BATCH_SIZE": ("sync", "batch_size", int),
"WIKIBRICKS_SYNC_APPLY_POLICY": ("sync", "apply_policy", str),
"WIKIBRICKS_AUTOMATION_ENABLED": ("automation", "enabled", str),
Expand Down Expand Up @@ -215,6 +237,10 @@ def load_config(
minimum=1,
maximum=36500,
)
generic_workspaces = _string_list(
value["curation"]["generic_workspaces"],
"curation.generic_workspaces",
)
automation_enabled = _boolean(
value["automation"]["enabled"],
"automation.enabled",
Expand Down Expand Up @@ -260,6 +286,7 @@ def load_config(
search_default_results=default_results,
search_maximum_results=maximum_results,
prune_archived_sessions_after_days=retention,
curation_generic_workspaces=tuple(generic_workspaces),
automation_enabled=automation_enabled,
automation_poll_seconds=poll_seconds,
automation_local_maintenance_hours=local_maintenance_hours,
Expand Down
12 changes: 12 additions & 0 deletions src/wikibricks/config/defaults.yml
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,18 @@ search:
maximum_results: 20
maintenance:
prune_archived_sessions_after_days: null
curation:
generic_workspaces:
- code
- emails
- work
- projects
- repos
- src
- documents
- desktop
- downloads
- tmp
automation:
enabled: true
poll_seconds: 300
Expand Down
173 changes: 147 additions & 26 deletions src/wikibricks/curation/backlog.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,11 +5,23 @@
from pathlib import Path
from typing import Any

from wikibricks.config import load_config


def _normalize(value: str) -> str:
return value.strip().lower().replace(" ", "-").replace("_", "-")


def _normalized_name(value: str) -> str:
return _normalize(Path(value).name)


def _unrouted_workspace(workspace: str | None, *, generic: set[str], home: Path) -> bool:
if workspace is None or not workspace or Path(workspace) == home:
return True
return _normalized_name(workspace) in generic


def _timestamp(value: str | datetime) -> datetime:
if isinstance(value, datetime):
parsed = value
Expand All @@ -21,56 +33,70 @@ def _timestamp(value: str | datetime) -> datetime:


def curation_backlog(
sessions: Iterable[tuple[str | None, str | datetime]],
sessions: Iterable[tuple[str, str | None, str | datetime]],
pages: Iterable[tuple[str, str, str | datetime]],
*,
now: datetime,
days: int = 7,
limit: int = 5,
home: Path | None = None,
targets: dict[str, str | None],
) -> list[dict[str, Any]]:
"""Return recent sessions that are newer than their covering pages."""
now = _timestamp(now)
home = home or Path.home()
cutoff = now - timedelta(days=days)
page_updates = {
path: _timestamp(updated_at) for path, _, updated_at in pages if not path.startswith("_meta/")
}
page_titles = {path: title for path, title, _ in pages}
projects: dict[str, dict[str, Any]] = {}

for workspace, updated_at in sessions:
if workspace is None or not workspace or Path(workspace) == home:
for session_id, workspace, updated_at in sessions:
target = targets.get(session_id)
if target is None:
continue
timestamp = _timestamp(updated_at)
if timestamp < cutoff or timestamp > now:
continue
project = _normalize(Path(workspace).name)
project = Path(target).name
item = projects.setdefault(
project,
target,
{
"project": project,
"living_page": target,
"workspace": workspace,
"workspaces": set(),
"session_timestamps": [],
},
)
if workspace is not None and workspace:
item["workspaces"].add(workspace)
if (
item["workspace"] is None
or not item["workspace"]
or workspace < item["workspace"]
):
item["workspace"] = workspace
item["session_timestamps"].append(timestamp)

covering: dict[str, list[tuple[str, datetime]]] = {}
for path, title, updated_at in pages:
if path.startswith("_meta/"):
continue
normalized_path = _normalize(path)
normalized_title = _normalize(title)
timestamp = _timestamp(updated_at)
for project, item in projects.items():
if project in normalized_path or project in normalized_title:
covering.setdefault(project, []).append((path, timestamp))

result: list[dict[str, Any]] = []
for project, item in projects.items():
project_pages = sorted(covering.get(project, []))
for target, item in projects.items():
project = item["project"]
living_page = item["living_page"]
last_page_update = page_updates.get(living_page)
project_pages: list[str] = []
if last_page_update is not None:
project_pages.append(living_page)
project_pages.extend(
path
for path in sorted(page_updates)
if path != living_page
and path.startswith("topics/")
and (
project in _normalize(path)
or project in _normalize(page_titles.get(path, ""))
)
)
last_session_at = max(item["session_timestamps"])
if project_pages:
last_page_update = max(timestamp for _, timestamp in project_pages)
else:
last_page_update = None
new_sessions = sum(
last_page_update is None or timestamp > last_page_update
for timestamp in item["session_timestamps"]
Expand All @@ -81,8 +107,10 @@ def curation_backlog(
result.append(
{
**item,
"workspaces": sorted(item["workspaces"]),
"workspace": item["workspace"],
"new_sessions": new_sessions,
"pages": [path for path, _ in project_pages[:3]],
"pages": project_pages[:3],
"last_page_update": (
last_page_update.isoformat() if last_page_update else None
),
Expand All @@ -95,16 +123,109 @@ def curation_backlog(
)[:limit]


def session_targets(
conn: Any,
*,
generic: Iterable[str] | None = None,
home: Path | None = None,
) -> dict[str, str | None]:
"""Route every session to the durable page that covers its work."""
home = home or Path.home()
normalized_generic = {
_normalize(value) for value in (
load_config().curation_generic_workspaces
if generic is None
else generic
)
}
session_rows = conn.execute(
"SELECT s.session_id, s.workspace, t.session_id, t.page_path FROM sessions s "
"LEFT JOIN session_topics t ON t.session_id = s.session_id"
).fetchall()
page_rows = conn.execute(
"SELECT p.path, v.title FROM pages p "
"JOIN page_versions v ON v.version_id = p.current_version_id "
"WHERE p.status = 'active' AND p.path LIKE 'topics/%' ORDER BY p.path"
).fetchall()
pages = [(row[0], row[1]) for row in page_rows]
targets: dict[str, str | None] = {}
for session_id, workspace, topic_session_id, routed_page in session_rows:
if topic_session_id is not None:
targets[session_id] = routed_page
continue
if _unrouted_workspace(workspace, generic=normalized_generic, home=home):
targets[session_id] = None
continue
key = _normalized_name(workspace)
target = next(
(
path
for path, title in pages
if key in _normalize(path) or key in _normalize(title)
),
f"topics/{key}",
)
targets[session_id] = target
return targets


def unrouted_sessions(
conn: Any,
*,
since: str | datetime,
generic: Iterable[str] | None = None,
home: Path | None = None,
) -> list[str]:
"""Return container-folder sessions still waiting for topic routing."""
home = home or Path.home()
normalized_generic = {
_normalize(value) for value in (
load_config().curation_generic_workspaces
if generic is None
else generic
)
}
cutoff = _timestamp(since)
rows = conn.execute(
"SELECT s.session_id, s.workspace, "
"COALESCE(s.source_updated_at, s.updated_at) FROM sessions s "
"LEFT JOIN session_topics t ON t.session_id = s.session_id "
"WHERE t.session_id IS NULL ORDER BY s.session_id"
).fetchall()
return [
session_id
for session_id, workspace, updated_at in rows
if _timestamp(updated_at) >= cutoff
and _unrouted_workspace(workspace, generic=normalized_generic, home=home)
]


def load_curation_backlog(conn: Any, **kwargs: Any) -> list[dict[str, Any]]:
"""Load sessions and active pages, then calculate the curation backlog."""
# Count sessions by when the work happened, not when a backfill imported them.
sessions = conn.execute(
"SELECT workspace, COALESCE(source_updated_at, updated_at) FROM sessions"
"SELECT s.session_id, s.workspace, COALESCE(s.source_updated_at, s.updated_at), "
"t.created_at "
"FROM sessions s LEFT JOIN session_topics t ON t.session_id = s.session_id"
).fetchall()
sessions = [
(
row[0],
row[1],
_timestamp(row[2]) if row[3] is None else max(_timestamp(row[2]), _timestamp(row[3])),
)
for row in sessions
]
pages = conn.execute(
"SELECT p.path, v.title, p.updated_at FROM pages p "
"JOIN page_versions v ON v.version_id = p.current_version_id "
"WHERE p.status = 'active'"
).fetchall()
home = kwargs.pop("home", None)
generic = kwargs.pop("generic", None)
if generic is None:
generic = load_config().curation_generic_workspaces
targets = session_targets(conn, generic=generic, home=home)
kwargs.setdefault("now", datetime.now(timezone.utc))
kwargs["targets"] = targets
return curation_backlog(sessions, pages, **kwargs)
6 changes: 6 additions & 0 deletions src/wikibricks/sql/migrations/0007_session_topics.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
CREATE TABLE session_topics (
session_id uuid PRIMARY KEY REFERENCES sessions(session_id) ON DELETE CASCADE,
page_path text,
origin text NOT NULL,
created_at timestamptz NOT NULL DEFAULT now()
);
6 changes: 6 additions & 0 deletions src/wikibricks/sql/sqlite/0004_session_topics.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
CREATE TABLE IF NOT EXISTS session_topics (
session_id TEXT PRIMARY KEY REFERENCES sessions(session_id) ON DELETE CASCADE,
page_path TEXT,
origin TEXT NOT NULL,
created_at TEXT NOT NULL
);
4 changes: 2 additions & 2 deletions src/wikibricks/storage/store.py
Original file line number Diff line number Diff line change
Expand Up @@ -76,8 +76,8 @@ def migrate(self) -> None:

def clear_all(self) -> None:
tables = (
"remote_maintenance_runs, curation_conflicts, curation_receipts, "
"page_aliases, curation_patches, "
"session_topics, remote_maintenance_runs, curation_conflicts, "
"curation_receipts, page_aliases, curation_patches, "
"curation_runs, sync_replicas, archive_batch_events, archive_events, "
"archive_batches, curated_pages, archive_pages, sync_state, sync_outbox, "
"session_search_chunks, session_event_versions, session_events, sessions, "
Expand Down
Loading
Loading