search.rs
⎇
Raw
1//! `GET /api/search` — index-free name and content search, streamed as SSE.
2//!
3//! No index, by design: every search walks the selected root with
4//! [`ignore::WalkParallel`] (fd/ripgrep's parallel walker, with all its
5//! filters off: hidden and gitignored entries are searched too) and matches
6//! on the fly.
7//!
8//! * **Name** (scope `name`/`both`): every word of the query must occur
9//! (case-insensitive) in the entry's own name — the last path component,
10//! not the whole relative path — so "invoice 2025" matches
11//! `invoice_2025-11_final.pdf` but a file is not a hit merely for sitting
12//! inside a directory whose name matches.
13//! * **Content** (scope `content`/`both`): the whole query as a literal,
14//! case-insensitive, ripgrep-style scan via the `grep` crates (ripgrep's
15//! engine). Binary files are skipped by NUL detection, files over
16//! [`SEARCH_MAX_FILE_BYTES`] are skipped and counted.
17//!
18//! Both scopes run in one walk, so `both` reads the tree once and its file
19//! and match events interleave.
20//!
21//! Results are one JSON object per SSE event, in the order found; the stream
22//! always ends with a `done` event. The search stops when the client goes
23//! away: axum drops the response body stream, the channel receiver is
24//! dropped with it, and the walkers bail at the next `is_closed` check —
25//! no per-search budgets or timeouts.
26
27use std::path::{Path, PathBuf};
28use std::sync::Arc;
29use std::sync::atomic::{AtomicUsize, Ordering};
30use std::time::Instant;
31
32use api_types::SearchEvent;
33use axum::extract::{Query as AxumQuery, State};
34use axum::http::StatusCode;
35use axum::response::sse::{Event, Sse};
36use axum::response::{IntoResponse, Response};
37use futures_util::StreamExt;
38use grep::regex::RegexMatcherBuilder;
39use grep::searcher::{BinaryDetection, Searcher, SearcherBuilder, Sink, SinkMatch};
40use ignore::{DirEntry, WalkBuilder, WalkState};
41use serde::Deserialize;
42
43use crate::api::common::AuthUser;
44use crate::db::RootRow;
45use crate::error::{ApiError, AppState};
46
47/// Concurrent searches. A search owns a pool of walker threads, so the two
48/// caps together bound the threads a burst of searches can create.
49static SEARCH_SLOTS: std::sync::LazyLock<Arc<tokio::sync::Semaphore>> =
50 std::sync::LazyLock::new(|| Arc::new(tokio::sync::Semaphore::new(4)));
51
52/// Walker threads per search: the machine's parallelism, capped at 8. The
53/// walk is IO-bound, so more threads buy nothing past that.
54fn walker_threads() -> usize {
55 std::thread::available_parallelism()
56 .map(|n| n.get())
57 .unwrap_or(4)
58 .min(8)
59}
60
61/// Content search skips files larger than this. There is no search budget
62/// (the client can stop at any time), but a multi-gigabyte text file would
63/// stall its worker indefinitely — ripgrep has no such cap, we do.
64const SEARCH_MAX_FILE_BYTES: u64 = 10 * 1024 * 1024;
65/// A matched line is truncated to this many bytes on the wire.
66const SEARCH_MAX_LINE_BYTES: usize = 500;
67/// Total matched lines emitted per search. A one-character query can match
68/// tens of thousands of lines; unbounded streaming would flood the client.
69const SEARCH_MAX_MATCH_EVENTS: usize = 50_000;
70/// Matched lines emitted per file (mirrors the client's render cap).
71const SEARCH_MAX_LINES_PER_FILE: usize = 500;
72
73#[derive(Debug, Deserialize)]
74pub(super) struct SearchQuery {
75 q: Option<String>,
76 /// `name` (default), `content` or `both`.
77 #[serde(default)]
78 scope: Option<String>,
79 /// The root to search; omitted = the caller's first root.
80 #[serde(default)]
81 root: Option<i64>,
82 /// Folder inside the root to start in; omitted = the whole root.
83 #[serde(default)]
84 path: Option<String>,
85}
86
87pub(super) async fn search(
88 State(state): State<Arc<AppState>>,
89 auth: AuthUser,
90 AxumQuery(query): AxumQuery<SearchQuery>,
91) -> Result<Response, ApiError> {
92 // Search is a session-user feature: a share visitor has no search UI and
93 // must not probe files outside the shared item.
94 if auth.share.is_some() {
95 return Err(ApiError::localized(
96 StatusCode::FORBIDDEN,
97 "search requires a signed-in session",
98 "err_search_forbidden",
99 ));
100 }
101 let q = query
102 .q
103 .as_deref()
104 .map(str::trim)
105 .filter(|q| !q.is_empty())
106 .ok_or_else(|| {
107 ApiError::localized(
108 StatusCode::BAD_REQUEST,
109 "search requires a non-empty query",
110 "err_search_empty_query",
111 )
112 })?;
113
114 let scope = query.scope.as_deref().unwrap_or("name");
115 let want_name = matches!(scope, "name" | "both");
116 let want_content = matches!(scope, "content" | "both");
117 if !(want_name || want_content) {
118 return Err(ApiError::localized(
119 StatusCode::BAD_REQUEST,
120 "scope must be name, content or both",
121 "err_search_bad_scope",
122 ));
123 }
124
125 // Only a root the caller actually has.
126 let root = match query.root {
127 Some(id) => auth.roots.iter().find(|r| r.id == id),
128 None => auth.roots.first(),
129 }
130 .cloned()
131 .ok_or_else(|| {
132 ApiError::localized(
133 StatusCode::FORBIDDEN,
134 "root not accessible",
135 "err_root_forbidden",
136 )
137 })?;
138
139 // The start folder must exist inside the root. Only validated here: the
140 // walk joins the un-canonicalized paths itself, so every result keeps
141 // the root's prefix and stays root-relative.
142 let start_rel = query
143 .path
144 .as_deref()
145 .unwrap_or("")
146 .trim_matches('/')
147 .to_string();
148 let (server_root, root_path, start) =
149 (state.root.clone(), root.path.clone(), start_rel.clone());
150 tokio::task::spawn_blocking(move || crate::fs::resolve_dir(&server_root, &root_path, &start))
151 .await
152 .map_err(|_| {
153 ApiError::localized(
154 StatusCode::INTERNAL_SERVER_ERROR,
155 "internal error",
156 "err_internal",
157 )
158 })??;
159
160 // One slot per running search, held until the walk ends. Each search owns
161 // a pool of walker threads, so unbounded concurrency would swamp the box.
162 // No queueing: a waiting request would hang without any response.
163 let slot = SEARCH_SLOTS.clone().try_acquire_owned().map_err(|_| {
164 ApiError::new(
165 StatusCode::SERVICE_UNAVAILABLE,
166 "too many searches are running",
167 )
168 })?;
169
170 let events = search_stream(
171 q.to_string(),
172 want_name,
173 want_content,
174 root,
175 start_rel,
176 state.root.clone(),
177 slot,
178 );
179 // `Sse` does the `data: <json>\n\n` framing and the content-type and
180 // cache headers; nginx and friends still need telling not to buffer.
181 let sse = Sse::new(events.map(|ev| Event::default().json_data(&ev)));
182 Ok(([("x-accel-buffering", "no")], sse).into_response())
183}
184
185/// Shared state of the search, behind one `Arc` that every walk visitor
186/// clones a handle to. The counters are atomics so they can be updated from
187/// the walker threads lock-free.
188struct SearchState {
189 tx: tokio::sync::mpsc::Sender<SearchEvent>,
190 words: Vec<String>,
191 root: RootRow,
192 /// Where the walk starts, relative to the root ("" = the root itself).
193 start_rel: String,
194 server_root: PathBuf,
195 started: Instant,
196 /// Directory entries visited.
197 scanned: AtomicUsize,
198 /// Files skipped for content search (over the size cap).
199 skipped: AtomicUsize,
200 names: AtomicUsize,
201 matches: AtomicUsize,
202}
203
204/// The search itself: one walk over the root, matching names and grepping
205/// contents as it goes, sending events as they are found. Runs on a plain
206/// thread — `WalkParallel` manages its own worker pool — while the SSE body
207/// stream polls the receiver.
208///
209/// Stopping: when the client goes away, axum drops the body stream, which
210/// drops the receiver. Every `is_closed` check in the walkers then yields
211/// `WalkState::Quit` at the next entry, so the walk unwinds promptly.
212fn search_stream(
213 q: String,
214 want_name: bool,
215 want_content: bool,
216 root: RootRow,
217 start_rel: String,
218 server_root: PathBuf,
219 slot: tokio::sync::OwnedSemaphorePermit,
220) -> impl futures_util::Stream<Item = SearchEvent> + Send {
221 let (tx, rx) = tokio::sync::mpsc::channel::<SearchEvent>(256);
222 let state = Arc::new(SearchState {
223 tx,
224 words: q.split_whitespace().map(|w| w.to_lowercase()).collect(),
225 root,
226 start_rel,
227 server_root,
228 started: Instant::now(),
229 scanned: AtomicUsize::new(0),
230 skipped: AtomicUsize::new(0),
231 names: AtomicUsize::new(0),
232 matches: AtomicUsize::new(0),
233 });
234
235 std::thread::spawn(move || {
236 // Released when the walk is done, not when the client stops reading.
237 let _slot = slot;
238 walk(&state, &q, want_name, want_content);
239 // `stopped`: the receiver went away (client stopped or navigated)
240 // or the match cap was hit, before the walk finished.
241 let stopped = state.tx.is_closed()
242 || state.matches.load(Ordering::Relaxed) >= SEARCH_MAX_MATCH_EVENTS;
243 let _ = state.tx.blocking_send(SearchEvent::Done {
244 stopped,
245 files: state.names.load(Ordering::Relaxed),
246 matches: state.matches.load(Ordering::Relaxed),
247 scanned: state.scanned.load(Ordering::Relaxed),
248 skipped: state.skipped.load(Ordering::Relaxed),
249 elapsed_ms: state.started.elapsed().as_millis() as u64,
250 });
251 });
252
253 tokio_stream::wrappers::ReceiverStream::new(rx)
254}
255
256/// True when every query word occurs in `name`. Both are already lowercased.
257///
258/// `name` is the entry's own name, never its path — see the module docs.
259fn name_matches(words: &[String], name: &str) -> bool {
260 words.iter().all(|w| name.contains(w.as_str()))
261}
262
263/// Walks the root in parallel, matching each entry against the wanted
264/// scopes. The visitor is cloned once per worker thread, and returns
265/// [`WalkState::Quit`] to stop the whole walk (client went away).
266fn walk(st: &Arc<SearchState>, q: &str, want_name: bool, want_content: bool) {
267 let abs = st.server_root.join(&st.root.path);
268 let start = abs.join(&st.start_rel);
269 if !start.is_dir() {
270 // Root or start folder removed out from under us: nothing to search.
271 return;
272 }
273 WalkBuilder::new(&start)
274 // A file browser shows everything, so search must too: no hidden
275 // or `.gitignore`/`.ignore` filtering. Symlinks stay unfollowed.
276 .standard_filters(false)
277 .threads(walker_threads())
278 .build_parallel()
279 .run(|| {
280 let abs = abs.clone();
281 Box::new(move |result| match result {
282 Ok(entry) => visit(st, q, want_name, want_content, &entry, &abs),
283 Err(_) => WalkState::Continue, // unreadable entry: skip like fd
284 })
285 });
286}
287
288/// One walked entry: emit a name hit, grep it, or both.
289fn visit(
290 st: &Arc<SearchState>,
291 q: &str,
292 want_name: bool,
293 want_content: bool,
294 entry: &DirEntry,
295 abs: &Path,
296) -> WalkState {
297 if st.tx.is_closed() {
298 return WalkState::Quit;
299 }
300 st.scanned.fetch_add(1, Ordering::Relaxed);
301 // The start folder itself (depth 0) is not a result. Paths stay
302 // relative to the root, not to the start folder, so the client opens
303 // them the same way as any listing entry.
304 if entry.depth() == 0 {
305 return WalkState::Continue;
306 }
307 let Some(rel) = entry.path().strip_prefix(abs).ok() else {
308 return WalkState::Continue;
309 };
310 // Original case, unlike the name used for matching: this path is what
311 // the client opens the entry by.
312 let rel = rel.to_string_lossy().replace('\\', "/");
313 let is_dir = entry.file_type().is_some_and(|t| t.is_dir());
314
315 if want_name {
316 // Matched against the entry's own name, not its path: matching the
317 // path makes every descendant of a matching directory a hit too
318 // ("e" matching `search-test/` dragged in all 400 files under it),
319 // which buries the entries the user actually named.
320 let name = entry.file_name().to_string_lossy().to_lowercase();
321 if name_matches(&st.words, &name) {
322 st.names.fetch_add(1, Ordering::Relaxed);
323 let size = if is_dir {
324 0
325 } else {
326 entry.metadata().map(|m| m.len()).unwrap_or(0)
327 };
328 let ev = SearchEvent::File {
329 root_id: st.root.id,
330 path: rel.clone(),
331 size,
332 is_dir,
333 // Same sniff a directory listing does, so the client needs no
334 // extension table of its own. One open() per name hit, on the
335 // walker thread that already stat'ed the entry.
336 kind: crate::fs::detect_kind(entry.path(), is_dir),
337 };
338 // blocking_send: the channel cap provides backpressure against a
339 // slow client. Err means the receiver is gone: stop the search.
340 if st.tx.blocking_send(ev).is_err() {
341 return WalkState::Quit;
342 }
343 }
344 }
345
346 if want_content && entry.file_type().is_some_and(|t| t.is_file()) {
347 if entry.metadata().map(|m| m.len()).unwrap_or(0) > SEARCH_MAX_FILE_BYTES {
348 st.skipped.fetch_add(1, Ordering::Relaxed);
349 return WalkState::Continue;
350 }
351 search_one_file(q, entry.path(), rel, st);
352 if st.tx.is_closed() || st.matches.load(Ordering::Relaxed) >= SEARCH_MAX_MATCH_EVENTS {
353 return WalkState::Quit;
354 }
355 }
356 WalkState::Continue
357}
358
359/// Grep one file with a fresh literal matcher. The matcher is built per file
360/// rather than shared: `RegexMatcher` is not `Sync`, so it cannot cross the
361/// walk's thread boundary; a case-insensitive literal compiles in
362/// microseconds, negligible next to the file read it protects.
363fn search_one_file(q: &str, path: &Path, rel: String, st: &SearchState) {
364 let matcher = match RegexMatcherBuilder::new()
365 .case_insensitive(true)
366 .build_literals(&[q])
367 {
368 Ok(m) => m,
369 Err(_) => return,
370 };
371 let mut searcher = SearcherBuilder::new().line_number(true).build();
372 // ripgrep's default: a NUL byte means "binary, stop".
373 searcher.set_binary_detection(BinaryDetection::quit(0));
374
375 let mut sink = MatchSink { st, lines: 0, rel };
376 // I/O errors (permissions, vanished file): skip like fd does.
377 let _ = searcher.search_path(matcher, path, &mut sink);
378}
379
380/// Pushes each matched line onto the event channel. `blocking_send` is
381/// correct here: it runs on the walk's worker threads, and the channel cap
382/// provides backpressure against a slow client.
383struct MatchSink<'a> {
384 st: &'a SearchState,
385 /// Matched lines emitted for the current file.
386 lines: usize,
387 rel: String,
388}
389
390impl Sink for MatchSink<'_> {
391 type Error = std::io::Error;
392
393 fn matched(&mut self, _searcher: &Searcher, mat: &SinkMatch) -> Result<bool, Self::Error> {
394 if self.st.tx.is_closed() || self.lines >= SEARCH_MAX_LINES_PER_FILE {
395 return Ok(false);
396 }
397 let line = match mat.line_number() {
398 Some(n) => n,
399 None => return Ok(true),
400 };
401 // Single-line matcher: exactly one line per match.
402 let text = match mat.lines().next() {
403 Some(l) => truncate_line(l),
404 None => return Ok(true),
405 };
406 // Cap check before the increment, so the counted number of matches
407 // is exactly the number of matches delivered.
408 if self.st.matches.load(Ordering::Relaxed) >= SEARCH_MAX_MATCH_EVENTS {
409 return Ok(false);
410 }
411 self.lines += 1;
412 self.st.matches.fetch_add(1, Ordering::Relaxed);
413 match self.st.tx.blocking_send(SearchEvent::Match {
414 root_id: self.st.root.id,
415 path: self.rel.clone(),
416 line,
417 text,
418 }) {
419 Ok(()) => Ok(true),
420 // Receiver gone: stop this file; the walker quits on the next
421 // `is_closed` check.
422 Err(_) => Ok(false),
423 }
424 }
425}
426
427/// Strips the line terminator and truncates at [`SEARCH_MAX_LINE_BYTES`].
428///
429/// A character split by the cap decodes to one replacement character, as
430/// does any invalid byte; the cap is therefore a byte cap on the input, not
431/// exactly on the output.
432fn truncate_line(line: &[u8]) -> String {
433 let end = line
434 .iter()
435 .rposition(|b| *b != b'\n' && *b != b'\r')
436 .map_or(0, |i| i + 1);
437 String::from_utf8_lossy(&line[..end.min(SEARCH_MAX_LINE_BYTES)]).into_owned()
438}
439
440#[cfg(test)]
441mod tests {
442 use super::*;
443
444 #[test]
445 fn name_match_needs_every_word() {
446 let words = vec!["invoice".to_string(), "2025".to_string()];
447 assert!(name_matches(&words, "invoice_2025-11_final.pdf"));
448 assert!(!name_matches(&words, "invoice_2024.pdf"));
449 }
450
451 #[test]
452 fn name_match_ignores_the_parent_path() {
453 // The caller passes the entry's own name, so a file does not match
454 // just because an ancestor directory does.
455 let words = vec!["test".to_string()];
456 assert!(name_matches(&words, "test-notes.md"));
457 assert!(!name_matches(&words, "f12.txt"));
458 }
459
460 #[test]
461 fn truncation_strips_terminator() {
462 assert_eq!(truncate_line(b"hello\n"), "hello");
463 assert_eq!(truncate_line(b"hello\r\n"), "hello");
464 assert_eq!(truncate_line(b"no newline"), "no newline");
465 assert_eq!(truncate_line(b"\n"), "");
466 }
467
468 #[test]
469 fn truncation_caps_length() {
470 let long = vec![b'x'; 600];
471 let t = truncate_line(&long);
472 assert_eq!(t.len(), SEARCH_MAX_LINE_BYTES);
473 }
474
475 #[test]
476 fn truncation_keeps_utf8_boundary() {
477 // 3-byte chars; the cap must not split one.
478 let long: Vec<u8> = "ä".repeat(200).into_bytes();
479 let t = truncate_line(&long);
480 assert!(t.len() <= SEARCH_MAX_LINE_BYTES);
481 assert!(t.chars().all(|c| c == 'ä'));
482 }
483}
484