feature fixes

This commit is contained in:
pj committed 2026-08-28 16:08:39 +05:30
1 parent 1b31e2f189
commit 586ee946d0
38 files changed
+1938 -260

No files matched your search

+115 -10
View File
@@ -20,7 +20,7 @@ use std::sync::atomic::{AtomicU64, Ordering as Memory};
use std::sync::{Arc, LazyLock, Mutex};
use std::time::{SystemTime, UNIX_EPOCH};
use ignore::WalkBuilder;
use ignore::{DirEntry, WalkBuilder};
use tauri::{AppHandle, State};
use tauri_plugin_opener::OpenerExt;
@@ -387,6 +387,27 @@ fn assemble(
Some(node)
}
/// True for everything a walk should keep, which is everything except one of the four always
/// skipped names turning up as a folder somewhere below the walk root. Returning false for a
/// directory prunes it, so nothing inside it is walked either.
///
/// The root itself is kept whatever it is called, because a user who opens a folder named `dist`
/// opened it deliberately and hiding its entire contents from them would be absurd. Files are kept
/// whatever they are called too: the four names describe folders, and a document called `target.md`
/// is a document.
///
/// Shared by every walk in the app rather than written out once per walk, so the sidebar, the index
/// and the link sweep cannot drift apart about which folders are never worth descending into.
fn not_always_skipped(entry: &DirEntry) -> bool {
if entry.depth() == 0 {
return true;
}
if !entry.file_type().map(|t| t.is_dir()).unwrap_or(false) {
return true;
}
!ALWAYS_SKIPPED.contains(&entry.file_name().to_string_lossy().as_ref())
}
/// One pass over a folder, returning the root node with everything under it already attached.
///
/// `show_ignored` turns off gitignore, the hidden file rule and the four always skipped folders in
@@ -407,15 +428,7 @@ pub fn scan_tree(root: &Path, show_ignored: bool) -> Result<FileNode, String> {
.require_git(false)
.standard_filters(!show_ignored);
if !show_ignored {
builder.filter_entry(|entry| {
if entry.depth() == 0 {
return true;
}
if !entry.file_type().map(|t| t.is_dir()).unwrap_or(false) {
return true;
}
!ALWAYS_SKIPPED.contains(&entry.file_name().to_string_lossy().as_ref())
});
builder.filter_entry(not_always_skipped);
}
let mut nodes: HashMap<PathBuf, FileNode> = HashMap::new();
@@ -445,6 +458,73 @@ pub fn scan_tree(root: &Path, show_ignored: bool) -> Result<FileNode, String> {
assemble(root, &mut nodes, &children).ok_or_else(|| format!("cannot read {}", root.display()))
}
/// Every markdown document under `root`, as flat paths, walked by the link sweep's rules rather
/// than the sidebar's.
///
/// This walk exists because a .gitignore is a statement about version control and not about whether
/// a file is a document. The tree and the index honour it, and they are right to: a sidebar full of
/// build output and a search box full of vendored READMEs are both worse than those files staying
/// out of sight, and neither of them changes anything by leaving a file alone. A link is the
/// opposite case. A relative link inside an ignored draft is still a link the user follows, and
/// leaving it pointing at a path that this app is the thing that moved is a break nobody finds
/// until the day they click it. Hiding that file costs the user a broken document rather than a
/// tidy sidebar, so the sweep walks by its own rules and the two are allowed to disagree.
///
/// So the three git sources come off and the four hardcoded folders stay on, which is what keeps a
/// checkout's node_modules out of the sweep whether or not git was ever asked about it. Hidden
/// files stay out for the same reason, since a dotted folder is where other languages keep their
/// tooling and none of `.venv`, `.next`, `.cache`, `.tox` or `.gradle` holds a link anybody wrote.
/// A `.ignore` or `.rgignore` is still honoured, because that file is written for tools that walk
/// rather than for git, which makes it the honest way to tell this walk to stay out of a folder.
///
/// `limit` bounds the work, because the sweep reads and rewrites every file this returns and an
/// unbounded one over a folder the size of somebody's home directory is not what they asked for
/// when they renamed a file. There is deliberately no depth cap to go with it: a document one level
/// past a depth cap is silently not swept and nothing anywhere says so, which is the same class of
/// bug this function exists to fix. A count is honest instead, because the caller can see it was
/// hit. Which is why one path past the limit comes back rather than exactly `limit` of them: a
/// folder holding exactly the budget and a folder holding ten thousand more look identical at
/// `limit` paths, and the caller has to be able to tell them apart to say that its coverage was
/// partial rather than reporting a complete sweep of a subset.
pub fn documents_for_sweep(root: &Path, limit: usize) -> Vec<String> {
let mut builder = WalkBuilder::new(root);
builder
// A symlinked folder pointing back at one of its own ancestors would otherwise walk for
// ever, exactly as it would for the tree.
.follow_links(false)
.hidden(true)
// No climbing above the walk root looking for ignore files. The sweep is about this one
// folder, and what some parent of it happens to say about it is not this folder's business.
.parents(false)
.ignore(true)
.git_ignore(false)
.git_global(false)
.git_exclude(false);
builder.filter_entry(not_always_skipped);
let mut out = Vec::new();
for entry in builder.build() {
// One unreadable entry is one document the sweep does not visit and not a failed sweep. The
// caller reports what it covered either way, so losing a row here is a smaller and more
// honest failure than refusing to rewrite anything at all.
let Ok(entry) = entry else { continue };
if entry.file_type().map(|t| t.is_dir()).unwrap_or(true) {
continue;
}
// Markdown only. Plain text is indexed and searchable, but nothing in a .txt is a markdown
// link this app knows how to rewrite, and opening every one of them to find that out would
// be work spent to change nothing.
if kind_for(entry.path(), false) != "markdown" {
continue;
}
out.push(path_string(entry.path()));
if out.len() > limit {
break;
}
}
out
}
pub fn read_document(path: &Path) -> Result<ReadResult, String> {
// The mtime is taken before the read rather than after. Read the other way round and a change
// landing between the two would be stamped onto older text, and the next save would overwrite
@@ -756,6 +836,31 @@ pub fn tree_read(roots: State<'_, Roots>, root_id: String) -> Result<FileNode, S
scan_tree(Path::new(&path), false)
}
/// The flat list of markdown documents the link rewrite sweep has to visit in one root, at most
/// `limit` of them plus one more if the folder holds more than that.
///
/// This is deliberately not `tree_read`, and the difference is the whole point of it. The tree
/// hides what the folder's gitignore hides, which is the right answer for a sidebar and for search
/// because both of them only ever read: the worst a hidden row costs is a file the user has to find
/// another way. The sweep writes. A stale link left inside a file the tree chose not to show is
/// bytes on disk that look correct and are not, and the user learns about it by clicking the link
/// long after the rename that broke it. Reusing the tree's walk here would mean the app quietly
/// breaks the documents it decided were not worth showing, which is worse than either not renaming
/// or renaming loudly.
///
/// The count that comes back is the caller's, not this command's, business: a list longer than
/// `limit` means the folder overflowed the budget, and the caller is expected to say the sweep was
/// partial rather than rewrite the first `limit` files and report a finished job.
#[tauri::command(async)]
pub fn sweep_documents(
roots: State<'_, Roots>,
root_id: String,
limit: u32,
) -> Result<Vec<String>, String> {
let path = roots.path_for(&root_id)?;
Ok(documents_for_sweep(Path::new(&path), limit as usize))
}
/// Opens Finder with the file selected, rather than opening the file.
#[tauri::command]
pub fn reveal_in_finder(
+150 -21
View File
@@ -20,6 +20,7 @@
// than the editor ever does, so the promise that opening a folder writes nothing into it matters
// more here than anywhere: no sidecar, no lock, no mtime bumped, nothing.
use std::collections::HashMap;
use std::fs;
use std::path::{Path, PathBuf};
use std::sync::atomic::{AtomicBool, Ordering as Memory};
@@ -53,6 +54,14 @@ const LAST_INDEXED_KEY: &str = "last_indexed";
/// enough that a pass is not one transaction per file.
const BATCH: usize = 64;
/// Largest file whose text is read into the full text table. Nothing anybody typed is this big: it
/// is an export, a dataset or a log that happens to end in .txt, and an appended-to log is rewritten
/// on every debounce window for as long as the app is open, so the whole of it would be read and
/// tokenised again every time. The row is still written, because the path is worth finding in quick
/// open and only the text is left out, which is the same answer this file already gives a document
/// that is not UTF-8.
const BODY_MAX: u64 = 8 * 1024 * 1024;
/// How often a pass says where it has got to. The event drives a status line, not a progress bar
/// anybody watches closely, and emitting per file would cost more than the indexing.
const PROGRESS_EVERY: u32 = 64;
@@ -199,8 +208,13 @@ pub fn open(app: &AppHandle) -> Result<(), String> {
fn connect(file: &Path) -> Result<Connection, String> {
let conn = Connection::open(file).map_err(|e| format!("{}: {e}", file.display()))?;
// WAL so a search reads while the indexer writes, and NORMAL because every byte in here is
// derived from a file on disk: the worst a power cut can cost is a rescan.
conn.execute_batch("PRAGMA journal_mode = WAL; PRAGMA synchronous = NORMAL;")
// derived from a file on disk: the worst a power cut can cost is a rescan. The size limit is
// what makes the write ahead log give its space back after a checkpoint rather than keeping the
// high water mark of the largest rebuild for the life of the database, which on a big folder is
// the whole of it left sitting in the app data directory until the file is deleted.
conn.execute_batch(
"PRAGMA journal_mode = WAL; PRAGMA synchronous = NORMAL; PRAGMA journal_size_limit = 33554432;",
)
.map_err(|e| format!("{}: {e}", file.display()))?;
migrate(&conn)?;
Ok(conn)
@@ -367,22 +381,92 @@ fn work(app: AppHandle, jobs: mpsc::Receiver<Job>) {
pass
};
for job in jobs {
let Ok(index) = state(&app) else { continue };
let outcome = match job {
Job::Rebuild(roots) => {
let done = rebuild_pass(&app, &index, &roots, next());
index.rebuilding.store(false, Memory::SeqCst);
done
}
Job::Scan(root) => scan_pass(&app, &index, std::slice::from_ref(&root), next()),
Job::Forget(root_id) => with_conn(&index, |conn| forget_root_rows(conn, &root_id)),
Job::Changed(path) => changed(&app, &index, &path, next()),
Job::Removed(path) => with_conn(&index, |conn| remove_under(conn, &path)),
};
if let Err(e) = outcome {
eprintln!("search index: {e}");
// One blocking wait for the first job, then everything else already sitting behind it, taken as
// a batch rather than one at a time. When the kernel drops filesystem events the watcher reports
// the root itself as modified, and that job is a walk and a sweep of every folder the user has
// open: a two minute build that keeps the kernel dropping queues hundreds of them, and doing
// each one in turn means repeating the same full walk hundreds of times while the index falls
// further behind the disk with every repeat. It has to be collapsed on this side of the channel
// and not by bounding it, because the sender is the debounce callback and holds up the next
// batch of events for as long as it is made to wait.
while let Ok(first) = jobs.recv() {
let mut batch = vec![first];
while let Ok(more) = jobs.try_recv() {
batch.push(more);
}
for job in coalesce(batch) {
let Ok(index) = state(&app) else { continue };
// A panic in here would take this thread with it and nothing above would notice. The
// sender lives in a OnceLock that is never replaced, so every later job would be dropped
// by `send` without a word, and search, quick open and backlinks would go on answering
// from the snapshot the index happened to be holding at that moment for the rest of the
// session. One document the indexer cannot handle is not worth that.
let caught = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| match job {
Job::Rebuild(roots) => {
let done = rebuild_pass(&app, &index, &roots, next());
index.rebuilding.store(false, Memory::SeqCst);
done
}
Job::Scan(root) => scan_pass(&app, &index, std::slice::from_ref(&root), next()),
Job::Forget(root_id) => with_conn(&index, |conn| forget_root_rows(conn, &root_id)),
Job::Changed(path) => changed(&app, &index, &path, next()),
Job::Removed(path) => with_conn(&index, |conn| remove_under(conn, &path)),
}));
let outcome = match caught {
Ok(outcome) => outcome,
Err(_) => {
// A rebuild that unwound never reached its own `store`, and the flag left set is
// the status line stuck on "indexing" and every later rebuild declining to run.
index.rebuilding.store(false, Memory::SeqCst);
// Clearing the poison is safe because there is no half written state to inherit:
// whatever transaction the panic happened inside was dropped on the way out, and
// dropping a transaction rolls it back, so the database is exactly where it was
// before the job started. Leaving the poison would fail every later lock instead,
// which is the same frozen index arrived at by a different route.
index.conn.clear_poison();
index.status.clear_poison();
let mut status = status_of(&index).unwrap_or_default();
status.phase = "error".to_string();
status.error =
Some("the indexer hit a document it could not handle".to_string());
publish(&app, &index, status);
Err("a job panicked and was abandoned".to_string())
}
};
if let Err(e) = outcome {
eprintln!("search index: {e}");
}
}
}
}
/// One drained batch with the jobs that have been overtaken taken out of it.
///
/// Nothing is reordered, because the order is what makes the queue correct in the first place. A job
/// is dropped only when a later job in the same drain speaks about the same path, and that later one
/// is the one the disk now agrees with, so a create that followed a delete still wins and a delete
/// that followed a create still wins. It is the rule `watch::merge` applies within one debounced
/// batch, applied again across the batches that piled up while the worker was busy. A rebuild, a
/// scan and a forget name no path at all and are always kept.
fn coalesce(batch: Vec<Job>) -> Vec<Job> {
let mut last: HashMap<PathBuf, usize> = HashMap::new();
for (at, job) in batch.iter().enumerate() {
if let Some(path) = job_path(job) {
last.insert(path.to_path_buf(), at);
}
}
batch
.into_iter()
.enumerate()
.filter(|(at, job)| job_path(job).is_none_or(|path| last.get(path) == Some(at)))
.map(|(_, job)| job)
.collect()
}
fn job_path(job: &Job) -> Option<&Path> {
match job {
Job::Changed(path) | Job::Removed(path) => Some(path),
_ => None,
}
}
@@ -397,7 +481,16 @@ fn rebuild_pass(
) -> Result<(), String> {
let ids: Vec<String> = roots.iter().map(|root| root.id.clone()).collect();
with_conn(index, |conn| forget_roots_except(conn, &ids))?;
scan_pass(app, index, roots, pass)
scan_pass(app, index, roots, pass)?;
// FTS5 leaves a segment behind for every rewrite of a row, and a document is rewritten on every
// save, so an index that is never merged is one a long session slowly makes worse at the one
// thing it is for. A full pass is the moment it is fair to do the merging: the user has already
// asked for a walk of every folder they have open, and this costs less than the walk did.
with_conn(index, |conn| {
conn.execute("INSERT INTO docs_fts(docs_fts) VALUES('optimize')", [])
.map(|_| ())
.map_err(|e| e.to_string())
})
}
/// One pass over a set of roots: walk them all first so the total is known before the first file is
@@ -457,6 +550,15 @@ fn changed(app: &AppHandle, index: &Index, path: &Path, pass: i64) -> Result<(),
// Gone again between the event and here, which a debounce window makes perfectly ordinary.
return with_conn(index, |conn| remove_under(conn, path));
};
if is_skipped_below(&root.path, path) {
// The walk hides these folders and so must the watcher, which otherwise reaches the indexer
// with everything the walk refused to look at. An npm install under an open root is a row
// and a body for every README in node_modules, thousands of them, and a sweep will not take
// them back out because they were written by the pass that is sweeping. Checked after the
// stat rather than before it so a deletion under one of these folders is still applied,
// which is what takes away rows an earlier build of this file put there.
return Ok(());
}
if !meta.is_dir() {
if !is_document(path) {
return Ok(());
@@ -571,6 +673,21 @@ fn is_document(path: &Path) -> bool {
matches!(crate::fs::kind_for(path, false), "markdown" | "text")
}
/// Whether a path sits inside one of the folders the tree never shows.
///
/// Only the four unconditional names, and deliberately not the folder's gitignore: these are the
/// ones the tree hides whatever a gitignore says, and building an ignore matcher for every event
/// that arrives would cost more than the indexing it saves. A path that is the root itself strips to
/// an empty relative path with no components at all, so the watcher's "the kernel dropped events,
/// here is the root" report is not caught by this and still rescans everything.
fn is_skipped_below(root: &str, path: &Path) -> bool {
let Ok(rel) = path.strip_prefix(root) else {
return false;
};
rel.components()
.any(|part| crate::fs::ALWAYS_SKIPPED.contains(&part.as_os_str().to_string_lossy().as_ref()))
}
/// One document into the three tables, or one stat if the file has not moved since the last pass.
///
/// The mtime shortcut is what makes a rescan of an unchanged folder cost a walk rather than a read
@@ -602,8 +719,14 @@ fn index_document(
}
// A file that is not UTF-8 is indexed with no text rather than skipped. Its path is still worth
// finding in quick open, and refusing the whole row would make it invisible instead.
let body = fs::read_to_string(path).unwrap_or_default();
// finding in quick open, and refusing the whole row would make it invisible instead. A file
// past `BODY_MAX` is given the same answer for the same reason: nothing an extension can tell
// us says how big a .txt is, and the stat that decided the mtime above already knows.
let body = if meta.len() > BODY_MAX {
String::new()
} else {
fs::read_to_string(path).unwrap_or_default()
};
let title = title_for(path, &body);
let name = path
.file_name()
@@ -1188,7 +1311,13 @@ fn snippet_of(line: &str) -> (String, Vec<MatchRange>) {
if tail.saturating_sub(from) < SNIPPET_MAX {
from = tail.saturating_sub(SNIPPET_MAX).max(lead);
}
let to = tail.min(from + SNIPPET_MAX);
// `from` can end up past `tail` when the line has no text left on it at all. A document is free
// to contain the control character the marks are made of, and every line carrying one is read
// as a line with a match on it here whether there is anything else on it or not, so a line that
// is one stray mark and some spaces reaches this. `tail` is at or after `from` in every ordinary
// case, so this only ever changes that one: what comes out is an empty snippet with no ranges
// rather than a slice that starts after it ends.
let to = tail.max(from).min(from + SNIPPET_MAX);
let mut out = String::new();
let mut shift = from;
+1
View File
@@ -230,6 +230,7 @@ pub fn run() {
fs::root_open,
fs::root_close,
fs::tree_read,
fs::sweep_documents,
fs::reveal_in_finder,
fs::open_external,
fs::file_read,
+68 -6
View File
@@ -9,7 +9,7 @@
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::{LazyLock, Mutex};
use std::sync::{Arc, LazyLock, Mutex};
use std::time::{Duration, Instant, SystemTime};
use notify::event::{ModifyKind, RenameMode};
@@ -192,7 +192,16 @@ where
}
let canonical = std::fs::canonicalize(&root).map_err(|e| e.to_string())?;
// Shared rather than owned by the handler, because the root's own disappearance is reported by
// the watchdog below and not by the debouncer, and both have to emit into the same place. Behind
// a lock because a sink is only `Send` and not `Sync`, which also has the two take turns rather
// than interleave two batches in whatever the frontend is doing with them.
let sink = Arc::new(Mutex::new(sink));
let watchdog_sink = Arc::downgrade(&sink);
let watched = canonical.clone();
let watchdog_id = root_id.clone();
let watchdog_path = root.clone();
let handler = move |result: DebounceEventResult| {
let batch = match result {
Ok(batch) => batch,
@@ -205,7 +214,9 @@ where
};
let events = watch_events(&batch, &root_id, &watched, &root);
if !events.is_empty() {
sink(events);
if let Ok(sink) = sink.lock() {
(*sink)(events);
}
}
};
@@ -228,6 +239,43 @@ where
debouncer
.watch(&canonical, RecursiveMode::Recursive)
.map_err(|e| e.to_string())?;
// The root's own removal is the one change this watcher cannot wait for, so it is asked about
// instead. An FSEvents stream is placed on a path and hears nothing that happens above that
// path, and notify does not ask for the flag that would change that, so a parent folder renamed
// or deleted takes the root with it in complete silence. Even the root's own deletion is a
// favour rather than a promise: a folder emptied and removed can come back as one coalesced
// event on the parent, which is not a path this stream matches, and then the whole batch is
// dropped before anything here sees it. Waiting for an event that may never be sent is what left
// a folder deleted out from under the app looking open, with a watcher still in the map holding
// a stream on a path that no longer exists.
//
// One stat per debounce tick settles it on any filesystem, and the answer is terminal: nothing
// further will ever arrive on a stream whose path is gone, so the thread reports the removal and
// stops. `watch_start` hears that removal like any other and drops the watcher.
//
// The thread ends with the watch. The sink is the only thing it holds and it holds it weakly, so
// once the debouncer is dropped and its own thread lets go of the handler there is nothing left
// to report into and nothing to report about.
std::thread::spawn(move || loop {
std::thread::sleep(DEBOUNCE);
let Some(sink) = watchdog_sink.upgrade() else {
return;
};
if !is_gone(&canonical) {
continue;
}
if let Ok(sink) = sink.lock() {
(*sink)(vec![WatchEvent {
root: watchdog_id.clone(),
path: watchdog_path.to_string_lossy().into_owned(),
kind: "removed".to_string(),
old_path: None,
}]);
}
return;
});
Ok(debouncer)
}
@@ -292,10 +340,12 @@ fn watch_events(
merge(&mut events, &mut index, next);
}
// The root itself going away is the one change nothing under it can describe. macOS does report
// it as an event on the watched path, but a folder moved rather than emptied is a single rename
// this side may never see, so the state of the folder is checked rather than waited for.
if !canonical_root.exists() {
// The root itself going away is the one change nothing under it can describe. macOS usually does
// report it as an event on the watched path, but a folder moved rather than emptied is a single
// rename this side may never see, so the state of the folder is checked rather than waited for.
// This is the fast path only: it reports the removal in the same batch as the changes that came
// with it, and the watchdog in `spawn_watcher` is what makes it certain to be reported at all.
if is_gone(canonical_root) {
merge(
&mut events,
&mut index,
@@ -311,6 +361,18 @@ fn watch_events(
events
}
/// Whether the path is not there any more, as against unreadable for some other reason.
///
/// Only a missing file is an answer. A stat that fails because permissions changed or because a
/// volume stopped answering says nothing about whether the folder still exists, and closing the
/// user's open folder on the strength of it would be worse than reporting nothing at all.
fn is_gone(path: &Path) -> bool {
match std::fs::symlink_metadata(path) {
Ok(_) => false,
Err(error) => error.kind() == std::io::ErrorKind::NotFound,
}
}
fn merge(events: &mut Vec<WatchEvent>, index: &mut HashMap<String, usize>, next: WatchEvent) {
match index.get(&next.path) {
// A later `modified` says nothing a create or a rename in the same batch has not already