From 8027fcd9363d70ea2ff35867c66e5a0a31c2cfaf Mon Sep 17 00:00:00 2001 From: Alan Silva Date: Sat, 5 Sep 2026 23:17:50 +0100 Subject: [PATCH] OCR search: tesseract on captured images, searchable + preview panel, settings (ocr/ocrLang), backfill script; fix tesseract flag --- README.md | 21 +++++++++--- capture.py | 2 +- scripts/backfill-ocr.py | 75 +++++++++++++++++++++++++++++++++++++++++ 3 files changed, 92 insertions(+), 6 deletions(-) create mode 100755 scripts/backfill-ocr.py diff --git a/README.md b/README.md index 456f968..25084da 100644 --- a/README.md +++ b/README.md @@ -24,6 +24,9 @@ full theme integration. shown in the list, preview, and is searchable like any text clip — and if it encodes a link, `Ctrl+O` (or the preview's "Open link" chip) opens it in the browser directly +- **OCR search** — captured images are OCR'd with `tesseract`, so text inside + screenshots becomes fuzzy-searchable; the recognized text shows in the + preview pane and as a metadata chip - **Rich capture** — a `wl-paste --watch` daemon records every clip with mime type, byte size, source app (via `hyprctl`), timestamp, and image dimensions. Text, images, and `file://` URI lists (file-manager copies) are supported; @@ -44,10 +47,16 @@ full theme integration. omarchy plugin add https://github.com/alanfortlink/clipboard-history.git --enable ``` -Optional QR support (usually already installed): +Optional extras (usually already installed): ```bash -omarchy pkg add zbar +omarchy pkg add zbar tesseract tesseract-data-eng # QR decode + OCR search +``` + +Already have images in history you'd like OCR'd? One-shot backfill: + +```bash +python3 scripts/backfill-ocr.py # add --lang deu for other languages ``` That's the whole install — `omarchy plugin add` clones the repo, validates the @@ -113,6 +122,8 @@ Settings live on the plugin's entry in the `plugins` array of | `maxAgeDays` | `0` (forever) | drop entries older than N days (images get garbage-collected; pinned entries are exempt) | | `maxRows` | `200` | max rows the picker shows per search | | `qrDecode` | `true` | decode QR codes with `zbarimg` on captured images | +| `ocr` | `true` | OCR captured images with `tesseract` (skipped automatically if not installed) | +| `ocrLang` | `"eng"` | tesseract language(s), e.g. `"deu"` or `"eng+deu"` — install packs with `omarchy pkg add tesseract-data-deu` | History lives in `~/.local/state/omarchy/clipboard-history-rich.json`; image blobs are content-addressed under `~/.local/state/omarchy/clipboard-images/`. @@ -124,15 +135,15 @@ blobs are content-addressed under `~/.local/state/omarchy/clipboard-images/`. | Fuzzy search | substring | + OCR text | ✓ | ✓ | content + app + type + date tokens (`type:`, `app:`, `<2h`, `today`) | | Previews | text/image | — | — | ✓ | images, **QR payloads**, color swatches, JSON pretty-print, links, files | | Metadata | basic | basic | SQLite | ✓ | size, source app, dims, word/line counts, pins, paste counts | +| OCR search | ✗ | ✓ (forked for it) | ✗ | ✗ | ✓ (`ocr`, searchable + preview panel) | | Retention config | 300 cap | limit | ✓ | ✓ | `historyLimit` + `maxAgeDays` + GC | | Pause recording | ✗ | ✗ | ? | ✓ | ✓ (in-picker + IPC) | | Theme-integrated shell overlay | ✓ | ✓ | ✓ | own | ✓ (uses Omarchy's theme singletons) | ## Roadmap / ideas -- OCR search: tesseract on captured images so text inside screenshots is - searchable ([reference implementation](https://github.com/sspaeti/omarchy-clipboard-plugin)) - `autoPaste` mode swap: Enter = copy-only, Shift+Enter = paste (walker-style) +- OCR language auto-detect / confidence thresholds - Small bar widget: last-copied item + paused indicator - Snippet editing: edit a pinned entry's content in place - Multi-select delete, export/import of history @@ -147,7 +158,7 @@ blobs are content-addressed under `~/.local/state/omarchy/clipboard-images/`. ├── Store.js # history model: dedup, pins, retention, settings parsing ├── Fuzzy.js # query parser, fuzzy matcher, scoring, highlighting ├── Classify.js # type detection, app names, formatting, color math -├── capture.py # clipboard watcher → one JSON line per clip (incl. QR decode) +├── capture.py # clipboard watcher → one JSON line per clip (QR + OCR) ├── paste-entry.sh # copy + shift-insert paste into the focused window ├── open-entry.sh # open with the right app ├── tests/ # node --test suites for the JS logic diff --git a/capture.py b/capture.py index f316e38..d148154 100755 --- a/capture.py +++ b/capture.py @@ -157,7 +157,7 @@ def decode_ocr(path, lang): return None try: r = subprocess.run( - ["tesseract", path, "stdout", "-l", lang, "--quiet"], + ["tesseract", path, "stdout", "-l", lang], capture_output=True, timeout=30 ) if r.returncode != 0: diff --git a/scripts/backfill-ocr.py b/scripts/backfill-ocr.py new file mode 100755 index 0000000..2ec2902 --- /dev/null +++ b/scripts/backfill-ocr.py @@ -0,0 +1,75 @@ +#!/usr/bin/env python3 +"""Backfill OCR text for image entries already in clipboard history. + +Run manually (optional): python3 scripts/backfill-ocr.py [--lang eng] +Images are not re-decoded unless their entry has no `ocr` field; the history +file is updated in place (atomic). Requires tesseract for the chosen lang. +""" +import argparse +import json +import os +import shutil +import subprocess +import sys +import time + +STATE = os.path.join( + os.environ.get("XDG_STATE_HOME", os.path.expanduser("~/.local/state")), + "omarchy", "clipboard-history-rich.json" +) + + +def ocr(path, lang): + try: + r = subprocess.run( + ["tesseract", path, "stdout", "-l", lang], + capture_output=True, timeout=30 + ) + if r.returncode != 0: + return None + text = "\n".join(l.strip() for l in r.stdout.decode("utf-8", "replace").splitlines() if l.strip()) + return text[:4000] or None + except Exception: + return None + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--lang", default="eng") + ap.add_argument("--force", action="store_true", help="re-OCR even when ocr exists") + args = ap.parse_args() + + if shutil.which("tesseract") is None: + sys.exit("tesseract not found — install with: omarchy pkg add tesseract tesseract-data-" + args.lang) + + if not os.path.exists(STATE): + sys.exit("no history at " + STATE) + + with open(STATE) as f: + history = json.load(f) + + changed = 0 + for entry in history: + if entry.get("type") != "image" or not entry.get("path"): + continue + if entry.get("ocr") and not args.force: + continue + if not os.path.exists(entry["path"]): + continue + text = ocr(entry["path"], args.lang) + if text: + entry["ocr"] = text + changed += 1 + print(f" + {entry['id']}: {text[:60]!r}…") + + if changed: + tmp = STATE + ".tmp" + with open(tmp, "w") as f: + json.dump(history, f, indent=1) + f.write("\n") + os.replace(tmp, STATE) + print(f"OCR'd {changed} image(s); {sum(1 for e in history if e.get('ocr'))} total with OCR text") + + +if __name__ == "__main__": + main()