From dcc6bba61b0cc3ad3d5027fd61e04ae289e29890 Mon Sep 17 00:00:00 2001 From: Sorin Savu Date: Mon, 13 Jul 2026 09:51:28 +0300 Subject: [PATCH] upgrades --- db.py | 45 +++ diag.py | 25 ++ editor.html | 602 ++++++++++++++++++++++++++++++++++++++++ migrate.py | 77 ++++++ ocr.py | 56 ++++ requirements.txt | 8 + schema.sql | 57 ++++ server.py | 704 +++++++++++++++++++++++++++++++++++++++++++++++ wikilib.py | 25 ++ work-notes.md | 21 +- 10 files changed, 1612 insertions(+), 8 deletions(-) create mode 100644 db.py create mode 100644 diag.py create mode 100644 editor.html create mode 100644 migrate.py create mode 100644 ocr.py create mode 100644 requirements.txt create mode 100644 schema.sql create mode 100644 server.py create mode 100644 wikilib.py diff --git a/db.py b/db.py new file mode 100644 index 0000000..28980be --- /dev/null +++ b/db.py @@ -0,0 +1,45 @@ +"""MySQL connection helper (PyMySQL). + +Connection settings come from environment variables so nothing is hard-coded: + + WIKI_DB_HOST default 127.0.0.1 + WIKI_DB_PORT default 3306 + WIKI_DB_USER default root + WIKI_DB_PASSWORD default "" (empty) + WIKI_DB_NAME default wiki + +`connect()` selects the wiki database; `connect(select_db=False)` connects to the +server without a database (used by migrate.py to CREATE DATABASE). +""" + +from __future__ import annotations + +import os + +import pymysql +from pymysql.cursors import DictCursor + + +def _cfg() -> dict: + return { + "host": os.environ.get("WIKI_DB_HOST", "127.0.0.1"), + "port": int(os.environ.get("WIKI_DB_PORT", "3306")), + "user": os.environ.get("WIKI_DB_USER", "root"), + "password": os.environ.get("WIKI_DB_PASSWORD", ""), + } + + +def db_name() -> str: + return os.environ.get("WIKI_DB_NAME", "wiki") + + +def connect(select_db: bool = True) -> pymysql.connections.Connection: + kwargs = _cfg() + if select_db: + kwargs["database"] = db_name() + return pymysql.connect( + cursorclass=DictCursor, + autocommit=True, + charset="utf8mb4", + **kwargs, + ) diff --git a/diag.py b/diag.py new file mode 100644 index 0000000..9f2d78e --- /dev/null +++ b/diag.py @@ -0,0 +1,25 @@ +"""One-shot diagnostic for the truncated-import bug. +Run in the same shell where WIKI_DB_* are exported: python diag.py +""" +from pathlib import Path +import db + +src = Path("work-notes.md").read_text() +print(f"source file: {len(src)} chars") + +conn = db.connect() +with conn.cursor() as cur: + cur.execute("SHOW COLUMNS FROM notes LIKE 'body_md'") + col = cur.fetchone() + print(f"body_md column type: {col['Type']!r} <-- should be 'mediumtext'") + + cur.execute("SELECT slug, CHAR_LENGTH(body_md) AS n, body_md FROM notes WHERE slug='work-notes'") + row = cur.fetchone() + if not row: + print("no 'work-notes' note found") + else: + print(f"stored body: {row['n']} chars") + if row["n"] < len(src): + print(f"TRUNCATED by {len(src) - row['n']} chars") + print("...last 80 stored chars:", repr(row["body_md"][-80:])) +conn.close() diff --git a/editor.html b/editor.html new file mode 100644 index 0000000..ae9fc53 --- /dev/null +++ b/editor.html @@ -0,0 +1,602 @@ + + + + + Wiki + + + + + + + + + + + + + +
+
+ +

+ + saved + + + + + +
+
+
+ +
+
+ + + + + diff --git a/migrate.py b/migrate.py new file mode 100644 index 0000000..bf1bc1a --- /dev/null +++ b/migrate.py @@ -0,0 +1,77 @@ +"""Create the wiki database + schema, and import the existing work-notes.md. + +Idempotent: safe to run repeatedly. Connection comes from the WIKI_DB_* env +vars (see db.py). Run: python migrate.py +""" + +from __future__ import annotations + +import re +from pathlib import Path + +import db +from wikilib import slugify + +LEGACY_FILE = Path("work-notes.md") + + +def _statements(sql: str) -> list[str]: + # naive splitter: our schema has no semicolons inside statements + return [s.strip() for s in sql.split(";") if s.strip()] + + +def main() -> None: + name = db.db_name() + + # 1. create the database (connect without selecting one) + root = db.connect(select_db=False) + with root.cursor() as cur: + cur.execute( + f"CREATE DATABASE IF NOT EXISTS `{name}` " + "CHARACTER SET utf8mb4 COLLATE utf8mb4_unicode_ci" + ) + root.close() + print(f"database `{name}` ready") + + # 2. apply schema + conn = db.connect() + schema = Path("schema.sql").read_text() + with conn.cursor() as cur: + for stmt in _statements(schema): + cur.execute(stmt) + # additive migration for existing installs (MySQL lacks ADD COLUMN IF NOT EXISTS) + try: + cur.execute( + "ALTER TABLE notes ADD COLUMN archived TINYINT(1) NOT NULL DEFAULT 0" + ) + print("added notes.archived column") + except Exception: + pass # column already exists + print("schema applied") + + # 3. import the legacy single-file note, if present and not already imported + if LEGACY_FILE.exists(): + body = LEGACY_FILE.read_text() + slug = "work-notes" + # title = first markdown heading, else a sensible default + m = re.search(r"^#\s+(.+)$", body, re.MULTILINE) + title = m.group(1).strip() if m else "Work Notes" + with conn.cursor() as cur: + cur.execute("SELECT id FROM notes WHERE slug=%s", (slug,)) + if cur.fetchone(): + print(f"note '{slug}' already exists, skipping import") + else: + cur.execute( + "INSERT INTO notes (slug, title, body_md) VALUES (%s,%s,%s)", + (slug, title, body), + ) + print(f"imported {LEGACY_FILE} as note '{slug}' (title: {title!r})") + else: + print(f"{LEGACY_FILE} not found, nothing to import") + + conn.close() + print("done") + + +if __name__ == "__main__": + main() diff --git a/ocr.py b/ocr.py new file mode 100644 index 0000000..7ccb375 --- /dev/null +++ b/ocr.py @@ -0,0 +1,56 @@ +"""Pluggable OCR. + +The engine is chosen by the OCR_ENGINE env var: + + tesseract (default) -- pytesseract + Pillow + the system `tesseract` binary + none -- OCR disabled; images are marked 'skipped' + +Swapping in another engine later (e.g. a Claude-vision call) is just another +branch in `run_ocr` behind the same signature. `run_ocr` raises OCRUnavailable +when the configured engine's dependencies are missing (the worker then marks the +row 'skipped' rather than 'failed'), and lets real OCR errors propagate. +""" + +from __future__ import annotations + +import io +import os + + +class OCRUnavailable(RuntimeError): + """The configured OCR engine can't run (missing binary / library).""" + + +def engine() -> str: + return os.environ.get("OCR_ENGINE", "tesseract").lower() + + +def enabled() -> bool: + return engine() != "none" + + +def run_ocr(data: bytes, mime: str) -> str: + """Return extracted text for an image. Raises OCRUnavailable if the engine + is missing, or any other exception on a genuine OCR failure.""" + eng = engine() + if eng == "none": + raise OCRUnavailable("OCR_ENGINE=none") + if eng == "tesseract": + return _tesseract(data) + raise OCRUnavailable(f"unknown OCR_ENGINE={eng!r}") + + +def _tesseract(data: bytes) -> str: + try: + import pytesseract + from PIL import Image + except ImportError as e: + raise OCRUnavailable(f"pytesseract/Pillow not installed: {e}") from e + try: + img = Image.open(io.BytesIO(data)) + except Exception as e: # not a decodable image + raise OCRUnavailable(f"not a decodable image: {e}") from e + try: + return pytesseract.image_to_string(img) + except pytesseract.TesseractNotFoundError as e: + raise OCRUnavailable("tesseract binary not found on PATH") from e diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..3648208 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,8 @@ +Flask>=3.0 +PyMySQL>=1.1 +requests>=2.31 +PyYAML>=6.0 +flask-sock>=0.7 +simple-websocket>=1.0 +# Optional, only needed when OCR_ENGINE=tesseract: +# pip install pytesseract Pillow (and the system `tesseract-ocr` binary) diff --git a/schema.sql b/schema.sql new file mode 100644 index 0000000..92f5b6f --- /dev/null +++ b/schema.sql @@ -0,0 +1,57 @@ +-- Wiki schema. Run via migrate.py (which also CREATEs the database and imports +-- the existing work-notes.md). All statements are idempotent (IF NOT EXISTS). + +CREATE TABLE IF NOT EXISTS notes ( + id INT AUTO_INCREMENT PRIMARY KEY, + slug VARCHAR(255) NOT NULL UNIQUE, + title VARCHAR(255) NOT NULL, + body_md MEDIUMTEXT NOT NULL, + archived TINYINT(1) NOT NULL DEFAULT 0, + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP, + FULLTEXT KEY ft_notes (title, body_md) +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4; + +-- Snapshots written only when the user supplies a message on save +-- (mirrors the old "Create a new version?" git-commit behaviour). +CREATE TABLE IF NOT EXISTS note_versions ( + id INT AUTO_INCREMENT PRIMARY KEY, + note_id INT NOT NULL, + title VARCHAR(255) NOT NULL, + body_md MEDIUMTEXT NOT NULL, + message VARCHAR(500), + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + KEY idx_ver_note (note_id), + CONSTRAINT fk_ver_note FOREIGN KEY (note_id) REFERENCES notes(id) ON DELETE CASCADE +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4; + +CREATE TABLE IF NOT EXISTS attachments ( + id INT AUTO_INCREMENT PRIMARY KEY, + note_id INT NULL, + filename VARCHAR(255) NOT NULL, + mime_type VARCHAR(128) NOT NULL, + size INT NOT NULL, + sha256 CHAR(64) NOT NULL, + data LONGBLOB NOT NULL, + ocr_text MEDIUMTEXT NULL, + ocr_status ENUM('pending','done','failed','skipped') NOT NULL DEFAULT 'pending', + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + KEY idx_sha (sha256), + KEY idx_ocr_status (ocr_status), + CONSTRAINT fk_att_note FOREIGN KEY (note_id) REFERENCES notes(id) ON DELETE SET NULL, + FULLTEXT KEY ft_ocr (ocr_text) +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4; + +-- One row per [[wiki link]] found in a note body. to_note_id is NULL while the +-- target note does not yet exist (broken / forward link). Powers backlinks. +CREATE TABLE IF NOT EXISTS links ( + id INT AUTO_INCREMENT PRIMARY KEY, + from_note_id INT NOT NULL, + to_slug VARCHAR(255) NOT NULL, + to_note_id INT NULL, + KEY idx_link_from (from_note_id), + KEY idx_link_to_slug (to_slug), + KEY idx_link_to_note (to_note_id), + CONSTRAINT fk_link_from FOREIGN KEY (from_note_id) REFERENCES notes(id) ON DELETE CASCADE, + CONSTRAINT fk_link_to FOREIGN KEY (to_note_id) REFERENCES notes(id) ON DELETE SET NULL +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4; diff --git a/server.py b/server.py new file mode 100644 index 0000000..211a18c --- /dev/null +++ b/server.py @@ -0,0 +1,704 @@ +"""Wiki server: multiple MySQL-backed notes with [[wiki links]], file/screenshot +attachments (stored as BLOBs with metadata + pluggable OCR), and full-text search. + +Replaces the original 29-line single-file http.server. Run: python server.py +Listens on 0.0.0.0:33333 (same as before). DB settings come from WIKI_DB_* env +vars (see db.py); OCR engine from OCR_ENGINE (see ocr.py). +""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import threading +import time +from pathlib import Path + +import requests +import yaml +from flask import Flask, Response, abort, jsonify, request, send_from_directory +from flask_sock import Sock + +import db +import ocr +from wikilib import parse_links, slugify + +app = Flask(__name__) +sock = Sock(app) +HERE = __file__.rsplit("/", 1)[0] + + +# --------------------------------------------------------------------------- # +# websocket fan-out: notify all open instances when a note changes +# --------------------------------------------------------------------------- # +_ws_clients: set = set() +_ws_lock = threading.Lock() + + +def broadcast(payload: dict) -> None: + msg = json.dumps(payload) + with _ws_lock: + clients = list(_ws_clients) + for ws in clients: + try: + ws.send(msg) + except Exception: + with _ws_lock: + _ws_clients.discard(ws) + + +@sock.route("/ws") +def ws(ws): + with _ws_lock: + _ws_clients.add(ws) + try: + while True: + if ws.receive() is None: # blocks; None on disconnect + break + finally: + with _ws_lock: + _ws_clients.discard(ws) + + +# --------------------------------------------------------------------------- # +# Jira config + ticket lookup (read-only proxy with an in-memory TTL cache) +# --------------------------------------------------------------------------- # +JIRA_CONFIG_PATH = Path( + os.environ.get("WIKI_JIRA_CONFIG", "~/.config/metalsoft-mcp/environments.yaml") +).expanduser() +_JIRA_KEY_RE = re.compile(r"^[A-Z][A-Z0-9]+-\d+$") +_JIRA_SLUG_RE = re.compile(r"^(ms-\d+)(?:-|$)", re.I) # leading Jira id of a note slug + + +def _jira_key(slug: str | None) -> str | None: + """Uppercase Jira key if this slug starts with a Jira ticket id, else None. + Matches both "ms-7764" and legacy "ms-7764-bare-metal-..." slugs.""" + m = _JIRA_SLUG_RE.match(slug or "") + return m.group(1).upper() if m else None +_jira_cache: dict[str, tuple[float, int, dict]] = {} # key -> (expires, status, payload) +_JIRA_TTL_OK = 600.0 +_JIRA_TTL_ERR = 60.0 + + +def _load_jira_cfg() -> dict | None: + """Read the `jira` block from the MCP environments.yaml. Returns None if the + file is missing/unparseable or the block is absent (Jira features disabled).""" + try: + raw = yaml.safe_load(JIRA_CONFIG_PATH.read_text()) + j = (raw or {}).get("jira") + if j and j.get("base_url") and j.get("email") and j.get("api_token"): + return j + except Exception: + pass + return None + + +_JIRA_CFG = _load_jira_cfg() + + +def _flatten_adf(node) -> str: + """Best-effort flatten of Atlassian Document Format to plain text.""" + if node is None: + return "" + if isinstance(node, str): + return node + if isinstance(node, list): + return "".join(_flatten_adf(n) for n in node) + if not isinstance(node, dict): + return str(node) + t = node.get("type") + if t == "text": + return node.get("text", "") + inner = _flatten_adf(node.get("content", [])) + if t in ("paragraph", "heading"): + return inner + "\n" + if t == "hardBreak": + return "\n" + if t == "listItem": + return "- " + inner + return inner + + +_JIRA_BUCKET = {"new": "todo", "indeterminate": "inprogress", "done": "done"} + + +def _fetch_jira(key: str) -> tuple[int, dict]: + """(status_code, payload) for a Jira key, using the cache. Never raises.""" + now = time.time() + cached = _jira_cache.get(key) + if cached and cached[0] > now: + return cached[1], cached[2] + if _JIRA_CFG is None: + return 503, {"error": "jira_not_configured"} + base = _JIRA_CFG["base_url"].rstrip("/") + try: + r = requests.get( + f"{base}/rest/api/3/issue/{key}", + auth=(_JIRA_CFG["email"], _JIRA_CFG["api_token"]), + params={ + "fields": "summary,status,issuetype,assignee,priority,description", + "expand": "renderedFields", + }, + headers={"Accept": "application/json"}, + timeout=15, + ) + except requests.RequestException as e: + status, payload = 502, {"error": "jira_unreachable", "detail": str(e)} + _jira_cache[key] = (now + _JIRA_TTL_ERR, status, payload) + return status, payload + + if r.status_code == 404: + status, payload = 404, {"error": "not_found"} + _jira_cache[key] = (now + _JIRA_TTL_ERR, status, payload) + return status, payload + if r.status_code >= 400: + status, payload = 502, {"error": "jira_error", "status_code": r.status_code} + _jira_cache[key] = (now + _JIRA_TTL_ERR, status, payload) + return status, payload + + f = r.json().get("fields", {}) + st = f.get("status") or {} + cat = ((st.get("statusCategory") or {}).get("key")) or "new" + desc = _flatten_adf(f.get("description")).strip() + payload = { + "key": key, + "summary": f.get("summary"), + "status": st.get("name"), + "category": _JIRA_BUCKET.get(cat, "todo"), + "type": (f.get("issuetype") or {}).get("name"), + "assignee": (f.get("assignee") or {}).get("displayName"), + "priority": (f.get("priority") or {}).get("name"), + "url": f"{base}/browse/{key}", + "preview": desc[:280] + ("…" if len(desc) > 280 else ""), + } + _jira_cache[key] = (now + _JIRA_TTL_OK, 200, payload) + return 200, payload + + +_jira_assigned_cache: tuple[float, dict] | None = None +_JIRA_ASSIGNED_TTL = 120.0 + + +def _issue_to_info(issue: dict) -> dict: + base = _JIRA_CFG["base_url"].rstrip("/") + f = issue.get("fields", {}) + st = f.get("status") or {} + cat = ((st.get("statusCategory") or {}).get("key")) or "new" + return { + "key": issue["key"], + "summary": f.get("summary"), + "status": st.get("name"), + "category": _JIRA_BUCKET.get(cat, "todo"), + "url": f"{base}/browse/{issue['key']}", + } + + +def _jira_search(jql: str) -> list[dict]: + """Run a JQL search; returns a list of issue-info dicts (never raises).""" + if _JIRA_CFG is None: + return [] + base = _JIRA_CFG["base_url"].rstrip("/") + try: + r = requests.get( + f"{base}/rest/api/3/search/jql", + auth=(_JIRA_CFG["email"], _JIRA_CFG["api_token"]), + params={"jql": jql, "fields": "summary,status", "maxResults": 100}, + headers={"Accept": "application/json"}, + timeout=20, + ) + if r.status_code != 200: + return [] + out = [_issue_to_info(i) for i in r.json().get("issues", [])] + except requests.RequestException: + return [] + # seed the per-key cache so badges/headers render instantly + now = time.time() + for info in out: + _jira_cache[info["key"]] = ( + now + _JIRA_TTL_OK, + 200, + {**info, "type": None, "assignee": None, "priority": None, "preview": ""}, + ) + return out + + +def _jira_assigned() -> dict[str, dict]: + """{KEY: info} for tickets assigned to me that aren't Done (cached).""" + global _jira_assigned_cache + now = time.time() + if _jira_assigned_cache and _jira_assigned_cache[0] > now: + return _jira_assigned_cache[1] + issues = _jira_search( + "assignee = currentUser() AND statusCategory != Done ORDER BY updated DESC" + ) + out = {i["key"]: i for i in issues} + _jira_assigned_cache = (now + _JIRA_ASSIGNED_TTL, out) + return out + + +def _jira_status_many(keys: list[str]) -> dict[str, dict]: + """{KEY: info} for an explicit set of keys, in one search.""" + if not keys: + return {} + jql = "key in (" + ",".join(keys) + ")" + return {i["key"]: i for i in _jira_search(jql)} + + +# --------------------------------------------------------------------------- # +# helpers +# --------------------------------------------------------------------------- # +def _recompute_links(cur, note_id: int, body_md: str) -> None: + """Replace this note's outgoing links with the [[...]] refs in body_md, + resolving each target slug to a note id when that note exists.""" + cur.execute("DELETE FROM links WHERE from_note_id=%s", (note_id,)) + for slug in parse_links(body_md): + cur.execute("SELECT id FROM notes WHERE slug=%s", (slug,)) + row = cur.fetchone() + cur.execute( + "INSERT INTO links (from_note_id, to_slug, to_note_id) VALUES (%s,%s,%s)", + (note_id, slug, row["id"] if row else None), + ) + + +def _resolve_forward_links(cur, slug: str, note_id: int) -> None: + """A note with this slug now exists: point previously-unresolved links at it.""" + cur.execute( + "UPDATE links SET to_note_id=%s WHERE to_slug=%s AND to_note_id IS NULL", + (note_id, slug), + ) + + +def _note_payload(cur, note: dict) -> dict: + nid = note["id"] + # outgoing links (with existence flag) + cur.execute( + "SELECT to_slug, to_note_id IS NOT NULL AS exists_flag " + "FROM links WHERE from_note_id=%s", + (nid,), + ) + links = [{"slug": r["to_slug"], "exists": bool(r["exists_flag"])} for r in cur.fetchall()] + # backlinks: notes that link here + cur.execute( + "SELECT n.slug, n.title FROM links l JOIN notes n ON n.id=l.from_note_id " + "WHERE l.to_note_id=%s ORDER BY n.title", + (nid,), + ) + backlinks = [{"slug": r["slug"], "title": r["title"]} for r in cur.fetchall()] + # attachments + cur.execute( + "SELECT id, filename, mime_type, size, ocr_status FROM attachments " + "WHERE note_id=%s ORDER BY id", + (nid,), + ) + attachments = [ + { + "id": r["id"], + "filename": r["filename"], + "mime_type": r["mime_type"], + "size": r["size"], + "ocr_status": r["ocr_status"], + "url": f"/attachments/{r['id']}", + } + for r in cur.fetchall() + ] + return { + "id": nid, + "slug": note["slug"], + "title": note["title"], + "body_md": note["body_md"], + "archived": bool(note.get("archived")), + "updated_at": note["updated_at"].isoformat() if note.get("updated_at") else None, + "links": links, + "backlinks": backlinks, + "attachments": attachments, + } + + +# --------------------------------------------------------------------------- # +# pages (SPA shell) +# --------------------------------------------------------------------------- # +@app.get("/") +@app.get("/wiki/") +def index(slug: str | None = None): + return send_from_directory(HERE, "editor.html") + + +@app.after_request +def _no_cache_html(resp): + # keep the SPA shell fresh on every reload (avoids serving a stale editor) + if resp.mimetype == "text/html": + resp.headers["Cache-Control"] = "no-store" + return resp + + +# --------------------------------------------------------------------------- # +# notes API +# --------------------------------------------------------------------------- # +@app.get("/api/notes") +def list_notes(): + conn = db.connect() + with conn.cursor() as cur: + cur.execute("SELECT slug, title, updated_at FROM notes ORDER BY updated_at DESC") + rows = cur.fetchall() + conn.close() + return jsonify( + [ + { + "slug": r["slug"], + "title": r["title"], + "updated_at": r["updated_at"].isoformat() if r["updated_at"] else None, + } + for r in rows + ] + ) + + +@app.get("/api/sidebar") +def sidebar(): + """Four sections blending local notes with live Jira data: + Notes (plain) / In Progress / Backlog / Archive.""" + conn = db.connect() + with conn.cursor() as cur: + cur.execute("SELECT slug, title, archived FROM notes ORDER BY updated_at DESC") + notes = cur.fetchall() + conn.close() + + assigned = _jira_assigned() # {KEY: info}, one search (cached) + # statuses for local defect pages not already covered by the assigned search + local_keys = [] + for n in notes: + k = _jira_key(n["slug"]) + if k and not n["archived"] and k not in assigned: + local_keys.append(k) + extra = _jira_status_many(sorted(set(local_keys))) + + out = {"notes": [], "inprogress": [], "backlog": [], "archive": []} + seen_keys = set() + + for n in notes: + key = _jira_key(n["slug"]) + if n["archived"]: + info = assigned.get(key) or extra.get(key) if key else None + out["archive"].append({ + "slug": n["slug"], "title": (info or {}).get("summary") or n["title"], + "key": key, "has_page": True, "archived": True, + "url": (info or {}).get("url"), + }) + if key: + seen_keys.add(key) + continue + if key: + info = assigned.get(key) or extra.get(key) or {} + cat = info.get("category", "backlog") + bucket = {"inprogress": "inprogress", "todo": "backlog", + "done": "archive"}.get(cat, "backlog") + out[bucket].append({ + "slug": n["slug"], "title": info.get("summary") or n["title"], + "key": key, "category": cat, "status": info.get("status"), + "has_page": True, "archived": False, "url": info.get("url"), + }) + seen_keys.add(key) + else: + out["notes"].append({ + "slug": n["slug"], "title": n["title"], + "key": None, "has_page": True, "archived": False, + }) + + # assigned Jira tickets without a local page → add by category, has_page=false + for key, info in assigned.items(): + if key in seen_keys: + continue + bucket = "inprogress" if info["category"] == "inprogress" else "backlog" + out[bucket].append({ + "slug": key.lower(), "title": info.get("summary") or key, + "key": key, "category": info["category"], "status": info.get("status"), + "has_page": False, "archived": False, "url": info.get("url"), + }) + + for b in out.values(): + b.sort(key=lambda e: (e.get("title") or "").lower()) + return jsonify(out) + + +@app.post("/api/notes") +def create_note(): + data = request.get_json(force=True) + raw_title = (data.get("title") or "").strip() + body = data.get("body_md") or "" + slug = slugify(data.get("slug") or raw_title or "untitled") + # a Jira-backed note (slug == ticket id) takes its title from Jira, not the client + jkey = _jira_key(slug) + if jkey: + st, pj = _fetch_jira(jkey) + title = pj["summary"] if (st == 200 and pj.get("summary")) else jkey + else: + title = raw_title or "Untitled" + conn = db.connect() + with conn.cursor() as cur: + cur.execute("SELECT id FROM notes WHERE slug=%s", (slug,)) + if cur.fetchone(): + conn.close() + abort(409, f"a note with slug {slug!r} already exists") + cur.execute( + "INSERT INTO notes (slug, title, body_md) VALUES (%s,%s,%s)", + (slug, title, body), + ) + nid = cur.lastrowid + _recompute_links(cur, nid, body) + _resolve_forward_links(cur, slug, nid) + cur.execute("SELECT * FROM notes WHERE id=%s", (nid,)) + payload = _note_payload(cur, cur.fetchone()) + conn.close() + broadcast({"type": "note", "slug": slug, "origin": request.headers.get("X-Origin")}) + return jsonify(payload), 201 + + +@app.get("/api/notes/") +def get_note(slug: str): + conn = db.connect() + with conn.cursor() as cur: + cur.execute("SELECT * FROM notes WHERE slug=%s", (slug,)) + note = cur.fetchone() + if not note: + conn.close() + abort(404, f"no note {slug!r}") + # Jira-backed note: refresh the (locked) title from Jira and attach status + jira = None + jkey = _jira_key(slug) + if jkey: + st, pj = _fetch_jira(jkey) + if st == 200: + if pj.get("summary") and pj["summary"] != note["title"]: + cur.execute( + "UPDATE notes SET title=%s WHERE id=%s", (pj["summary"], note["id"]) + ) + note["title"] = pj["summary"] + jira = {"key": jkey, "status": pj["status"], + "category": pj["category"], "url": pj["url"]} + else: + jira = {"key": jkey} + payload = _note_payload(cur, note) + payload["jira"] = jira + conn.close() + return jsonify(payload) + + +@app.put("/api/notes/") +def update_note(slug: str): + data = request.get_json(force=True) + body = data.get("body_md") or "" + message = (data.get("message") or "").strip() + conn = db.connect() + with conn.cursor() as cur: + cur.execute("SELECT * FROM notes WHERE slug=%s", (slug,)) + note = cur.fetchone() + if not note: + conn.close() + abort(404, f"no note {slug!r}") + # Jira-backed notes have a locked, Jira-sourced title; ignore client title + jkey = _jira_key(slug) + if jkey: + st, pj = _fetch_jira(jkey) + title = pj["summary"] if (st == 200 and pj.get("summary")) else note["title"] + else: + title = (data.get("title") or note["title"]).strip() + cur.execute( + "UPDATE notes SET title=%s, body_md=%s WHERE id=%s", + (title, body, note["id"]), + ) + _recompute_links(cur, note["id"], body) + if message: # snapshot a version, like the old "Create a new version?" + cur.execute( + "INSERT INTO note_versions (note_id, title, body_md, message) " + "VALUES (%s,%s,%s,%s)", + (note["id"], title, body, message), + ) + cur.execute("SELECT * FROM notes WHERE id=%s", (note["id"],)) + payload = _note_payload(cur, cur.fetchone()) + conn.close() + broadcast({"type": "note", "slug": slug, "origin": request.headers.get("X-Origin")}) + return jsonify(payload) + + +@app.post("/api/notes//archive") +def archive_note(slug: str): + archived = bool((request.get_json(force=True) or {}).get("archived", True)) + conn = db.connect() + with conn.cursor() as cur: + cur.execute("UPDATE notes SET archived=%s WHERE slug=%s", (1 if archived else 0, slug)) + if cur.rowcount == 0: + conn.close() + abort(404, f"no note {slug!r}") + conn.close() + broadcast({"type": "note", "slug": slug, "origin": request.headers.get("X-Origin")}) + return jsonify({"slug": slug, "archived": archived}) + + +@app.get("/api/notes//backlinks") +def backlinks(slug: str): + conn = db.connect() + with conn.cursor() as cur: + cur.execute("SELECT id FROM notes WHERE slug=%s", (slug,)) + note = cur.fetchone() + if not note: + conn.close() + abort(404, f"no note {slug!r}") + cur.execute( + "SELECT n.slug, n.title FROM links l JOIN notes n ON n.id=l.from_note_id " + "WHERE l.to_note_id=%s ORDER BY n.title", + (note["id"],), + ) + rows = cur.fetchall() + conn.close() + return jsonify([{"slug": r["slug"], "title": r["title"]} for r in rows]) + + +# --------------------------------------------------------------------------- # +# search (LIKE-based: reliable for short tokens like "MS-7764" that FULLTEXT's +# min-token-length would drop; FULLTEXT indexes are kept in schema for future) +# --------------------------------------------------------------------------- # +@app.get("/api/search") +def search(): + q = (request.args.get("q") or "").strip() + if not q: + return jsonify([]) + like = f"%{q}%" + conn = db.connect() + with conn.cursor() as cur: + cur.execute( + "SELECT slug, title, " + " CASE WHEN title LIKE %s THEN 'title' ELSE 'body' END AS matched_in " + "FROM notes WHERE title LIKE %s OR body_md LIKE %s " + "ORDER BY updated_at DESC LIMIT 50", + (like, like, like), + ) + hits = {r["slug"]: {"slug": r["slug"], "title": r["title"], "matched_in": r["matched_in"]} + for r in cur.fetchall()} + # notes whose attachment OCR text matches + cur.execute( + "SELECT DISTINCT n.slug, n.title FROM attachments a JOIN notes n ON n.id=a.note_id " + "WHERE a.ocr_text LIKE %s LIMIT 50", + (like,), + ) + for r in cur.fetchall(): + hits.setdefault( + r["slug"], {"slug": r["slug"], "title": r["title"], "matched_in": "attachment"} + ) + conn.close() + return jsonify(list(hits.values())) + + +# --------------------------------------------------------------------------- # +# jira +# --------------------------------------------------------------------------- # +@app.get("/api/jira/") +def jira_ticket(key: str): + if not _JIRA_KEY_RE.match(key): + abort(400, "invalid jira key") + status, payload = _fetch_jira(key) + return jsonify(payload), status + + +# --------------------------------------------------------------------------- # +# attachments +# --------------------------------------------------------------------------- # +@app.post("/attachments") +def upload_attachment(): + if "file" not in request.files: + abort(400, "expected a multipart 'file' field") + f = request.files["file"] + data = f.read() + if not data: + abort(400, "empty file") + sha = hashlib.sha256(data).hexdigest() + mime = f.mimetype or "application/octet-stream" + filename = f.filename or "upload" + note_slug = request.form.get("note") + + conn = db.connect() + with conn.cursor() as cur: + note_id = None + if note_slug: + cur.execute("SELECT id FROM notes WHERE slug=%s", (note_slug,)) + row = cur.fetchone() + note_id = row["id"] if row else None + # dedupe by content hash: reuse an existing identical blob + cur.execute("SELECT id FROM attachments WHERE sha256=%s", (sha,)) + existing = cur.fetchone() + if existing: + aid = existing["id"] + else: + status = "pending" if mime.startswith("image/") and ocr.enabled() else "skipped" + cur.execute( + "INSERT INTO attachments (note_id, filename, mime_type, size, sha256, data, ocr_status) " + "VALUES (%s,%s,%s,%s,%s,%s,%s)", + (note_id, filename, mime, len(data), sha, data, status), + ) + aid = cur.lastrowid + conn.close() + return jsonify({"id": aid, "url": f"/attachments/{aid}", "filename": filename, "mime_type": mime}), 201 + + +@app.get("/attachments/") +def get_attachment(aid: int): + conn = db.connect() + with conn.cursor() as cur: + cur.execute( + "SELECT filename, mime_type, data FROM attachments WHERE id=%s", (aid,) + ) + row = cur.fetchone() + conn.close() + if not row: + abort(404) + disposition = "inline" if row["mime_type"].startswith("image/") else "attachment" + return Response( + row["data"], + mimetype=row["mime_type"], + headers={"Content-Disposition": f'{disposition}; filename="{row["filename"]}"'}, + ) + + +# --------------------------------------------------------------------------- # +# background OCR worker +# --------------------------------------------------------------------------- # +def _ocr_worker(poll_seconds: float = 5.0) -> None: + """Fill ocr_text for pending image attachments using the configured engine.""" + while True: + try: + conn = db.connect() + with conn.cursor() as cur: + cur.execute( + "SELECT id, mime_type, data FROM attachments " + "WHERE ocr_status='pending' LIMIT 5" + ) + rows = cur.fetchall() + for r in rows: + try: + text = ocr.run_ocr(r["data"], r["mime_type"]) + cur.execute( + "UPDATE attachments SET ocr_text=%s, ocr_status='done' WHERE id=%s", + (text, r["id"]), + ) + except ocr.OCRUnavailable: + cur.execute( + "UPDATE attachments SET ocr_status='skipped' WHERE id=%s", (r["id"],) + ) + except Exception: # genuine OCR failure on this image + cur.execute( + "UPDATE attachments SET ocr_status='failed' WHERE id=%s", (r["id"],) + ) + conn.close() + except Exception: + pass # DB hiccup: try again next tick + time.sleep(poll_seconds) + + +def start_ocr_worker() -> None: + threading.Thread(target=_ocr_worker, daemon=True).start() + + +if __name__ == "__main__": + start_ocr_worker() + app.run(host="0.0.0.0", port=33333, threaded=True) diff --git a/wikilib.py b/wikilib.py new file mode 100644 index 0000000..9575f46 --- /dev/null +++ b/wikilib.py @@ -0,0 +1,25 @@ +"""Small shared helpers: slugs and [[wiki link]] parsing.""" + +from __future__ import annotations + +import re + +_WIKILINK_RE = re.compile(r"\[\[([^\]]+)\]\]") +_SLUG_STRIP = re.compile(r"[^a-z0-9]+") + + +def slugify(text: str) -> str: + """Lowercase, collapse non-alphanumerics to single hyphens, trim. + + Mirrors the JS slugify() in editor.html — keep the two in sync. + """ + s = _SLUG_STRIP.sub("-", text.strip().lower()).strip("-") + return s or "untitled" + + +def parse_links(body_md: str) -> list[str]: + """Return the distinct target slugs referenced by [[...]] in a note body.""" + seen: dict[str, None] = {} + for m in _WIKILINK_RE.finditer(body_md or ""): + seen.setdefault(slugify(m.group(1)), None) + return list(seen) diff --git a/work-notes.md b/work-notes.md index 25290ba..8dd4253 100644 --- a/work-notes.md +++ b/work-notes.md @@ -20,14 +20,6 @@ admin 4xfC3B6LCMMlRiuQmA2kWWXo3L8= -### MS-8868 -The OS is not reinstalled when the templated id is changed on the server instance group - -***TODO:*** -Reproduce this and fix. I have two requests to look into this from Alex Bordei and https://chat.metalsoft.io/bigstep/pl/751ugq9xrtg8tkgaz7pz1xs1xc -Sergiu ? -There is a discussion on matermost that Mike will add in the defect - - @@ -168,3 +160,16 @@ Also, I was thinking that some checks/loops should be moved in the agent, and no 1. TBD: clarify what the fuck is the requirement here + + +## OLD Stuff + +### MS-8868 -The OS is not reinstalled when the templated id is changed on the server instance group +closed as WONT DO + +***TODO:*** +Reproduce this and fix. I have two requests to look into this from Alex Bordei and https://chat.metalsoft.io/bigstep/pl/751ugq9xrtg8tkgaz7pz1xs1xc +Sergiu ? +There is a discussion on matermost that Mike will add in the defect + +