search.rs
⎇
Raw
1//! `GET /api/search` — index-free name and content search, streamed as SSE.
2//!
3//! No index, by design: every search walks the selected root with
4//! [`ignore::WalkParallel`] (fd/ripgrep's parallel walker, with all its
5//! filters off: hidden and gitignored entries are searched too) and matches
6//! on the fly.
7//!
8//! * **Name** (scope `name`/`both`): every word of the query must occur
9//! (case-insensitive) in the entry's own name — the last path component,
10//! not the whole relative path — so "invoice 2025" matches
11//! `invoice_2025-11_final.pdf` but a file is not a hit merely for sitting
12//! inside a directory whose name matches.
13//! * **Content** (scope `content`/`both`): the whole query as a literal,
14//! case-insensitive, ripgrep-style scan via the `grep` crates (ripgrep's
15//! engine). Binary files are skipped by NUL detection, files over
16//! [`SEARCH_MAX_FILE_BYTES`] are skipped and counted.
17//!
18//! Both scopes run in one walk, so `both` reads the tree once and its file
19//! and match events interleave.
20//!
21//! Results are one JSON object per SSE event, in the order found; the stream
22//! always ends with a `done` event. The search stops when the client goes
23//! away: axum drops the response body stream, the channel receiver is
24//! dropped with it, and the walkers bail at the next `is_closed` check —
25//! no per-search budgets or timeouts.
26
27use std::path::{Path, PathBuf};
28use std::sync::Arc;
29use std::sync::atomic::{AtomicUsize, Ordering};
30use std::time::Instant;
31
32use api_types::SearchEvent;
33use axum::extract::{Query as AxumQuery, State};
34use axum::http::StatusCode;
35use axum::response::sse::{Event, Sse};
36use axum::response::{IntoResponse, Response};
37use futures_util::StreamExt;
38use grep::regex::RegexMatcherBuilder;
39use grep::searcher::{BinaryDetection, Searcher, SearcherBuilder, Sink, SinkMatch};
40use ignore::{DirEntry, WalkBuilder, WalkState};
41use serde::Deserialize;
42
43use crate::api::common::AuthUser;
44use crate::db::RootRow;
45use crate::error::{ApiError, AppState};
46
47/// Concurrent searches. A search owns a pool of walker threads, so the two
48/// caps together bound the threads a burst of searches can create.
49static SEARCH_SLOTS: std::sync::LazyLock<Arc<tokio::sync::Semaphore>> =
50 std::sync::LazyLock::new(|| Arc::new(tokio::sync::Semaphore::new(4)));
51
52/// Walker threads per search: the machine's parallelism, capped at 8. The
53/// walk is IO-bound, so more threads buy nothing past that.
54fn walker_threads() -> usize {
55 std::thread::available_parallelism()
56 .map(|n| n.get())
57 .unwrap_or(4)
58 .min(8)
59}
60
61/// Content search skips files larger than this. There is no search budget
62/// (the client can stop at any time), but a multi-gigabyte text file would
63/// stall its worker indefinitely — ripgrep has no such cap, we do.
64const SEARCH_MAX_FILE_BYTES: u64 = 10 * 1024 * 1024;
65/// A matched line is truncated to this many bytes on the wire.
66const SEARCH_MAX_LINE_BYTES: usize = 500;
67/// Total matched lines emitted per search. A one-character query can match
68/// tens of thousands of lines; unbounded streaming would flood the client.
69const SEARCH_MAX_MATCH_EVENTS: usize = 50_000;
70/// Matched lines emitted per file (mirrors the client's render cap).
71const SEARCH_MAX_LINES_PER_FILE: usize = 500;
72
73/// The field names are the shared [`api_types::P_Q`] / [`api_types::P_SCOPE`]
74/// / [`api_types::P_ROOT`] / [`api_types::P_PATH`] constants. `#[serde(rename)]`
75/// only takes a literal, so that link cannot be written here;
76/// `tests::query_fields_are_the_shared_constants` pins it instead.
77#[derive(Debug, Deserialize)]
78pub(super) struct SearchQuery {
79 q: Option<String>,
80 /// `name` (default), `content` or `both`.
81 #[serde(default)]
82 scope: Option<String>,
83 /// The root to search; omitted = the caller's first root.
84 #[serde(default)]
85 root: Option<i64>,
86 /// Folder inside the root to start in; omitted = the whole root.
87 #[serde(default)]
88 path: Option<String>,
89}
90
91pub(super) async fn search(
92 State(state): State<Arc<AppState>>,
93 auth: AuthUser,
94 AxumQuery(query): AxumQuery<SearchQuery>,
95) -> Result<Response, ApiError> {
96 // Search is a session-user feature: a share visitor has no search UI and
97 // must not probe files outside the shared item.
98 if auth.share.is_some() {
99 return Err(ApiError::localized(
100 StatusCode::FORBIDDEN,
101 "search requires a signed-in session",
102 "err_search_forbidden",
103 ));
104 }
105 let q = query
106 .q
107 .as_deref()
108 .map(str::trim)
109 .filter(|q| !q.is_empty())
110 .ok_or_else(|| {
111 ApiError::localized(
112 StatusCode::BAD_REQUEST,
113 "search requires a non-empty query",
114 "err_search_empty_query",
115 )
116 })?;
117
118 let scope = query.scope.as_deref().unwrap_or("name");
119 let want_name = matches!(scope, "name" | "both");
120 let want_content = matches!(scope, "content" | "both");
121 if !(want_name || want_content) {
122 return Err(ApiError::localized(
123 StatusCode::BAD_REQUEST,
124 "scope must be name, content or both",
125 "err_search_bad_scope",
126 ));
127 }
128
129 // Only a root the caller actually has.
130 let root = match query.root {
131 Some(id) => auth.roots.iter().find(|r| r.id == id),
132 None => auth.roots.first(),
133 }
134 .cloned()
135 .ok_or_else(|| {
136 ApiError::localized(
137 StatusCode::FORBIDDEN,
138 "root not accessible",
139 "err_root_forbidden",
140 )
141 })?;
142
143 // The start folder must exist inside the root. Only validated here: the
144 // walk joins the un-canonicalized paths itself, so every result keeps
145 // the root's prefix and stays root-relative.
146 let start_rel = query
147 .path
148 .as_deref()
149 .unwrap_or("")
150 .trim_matches('/')
151 .to_string();
152 let (server_root, root_path, start) =
153 (state.root.clone(), root.path.clone(), start_rel.clone());
154 tokio::task::spawn_blocking(move || crate::fs::resolve_dir(&server_root, &root_path, &start))
155 .await
156 .map_err(|_| {
157 ApiError::localized(
158 StatusCode::INTERNAL_SERVER_ERROR,
159 "internal error",
160 "err_internal",
161 )
162 })??;
163
164 // One slot per running search, held until the walk ends. Each search owns
165 // a pool of walker threads, so unbounded concurrency would swamp the box.
166 // No queueing: a waiting request would hang without any response.
167 let slot = SEARCH_SLOTS.clone().try_acquire_owned().map_err(|_| {
168 ApiError::new(
169 StatusCode::SERVICE_UNAVAILABLE,
170 "too many searches are running",
171 )
172 })?;
173
174 let events = search_stream(
175 q.to_string(),
176 want_name,
177 want_content,
178 root,
179 start_rel,
180 state.root.clone(),
181 slot,
182 );
183 // `Sse` does the `data: <json>\n\n` framing and the content-type and
184 // cache headers; nginx and friends still need telling not to buffer.
185 let sse = Sse::new(events.map(|ev| Event::default().json_data(&ev)));
186 Ok(([("x-accel-buffering", "no")], sse).into_response())
187}
188
189/// Shared state of the search, behind one `Arc` that every walk visitor
190/// clones a handle to. The counters are atomics so they can be updated from
191/// the walker threads lock-free.
192struct SearchState {
193 tx: tokio::sync::mpsc::Sender<SearchEvent>,
194 words: Vec<String>,
195 root: RootRow,
196 /// Where the walk starts, relative to the root ("" = the root itself).
197 start_rel: String,
198 server_root: PathBuf,
199 started: Instant,
200 /// Directory entries visited.
201 scanned: AtomicUsize,
202 /// Files skipped for content search (over the size cap).
203 skipped: AtomicUsize,
204 names: AtomicUsize,
205 matches: AtomicUsize,
206}
207
208/// The search itself: one walk over the root, matching names and grepping
209/// contents as it goes, sending events as they are found. Runs on a plain
210/// thread — `WalkParallel` manages its own worker pool — while the SSE body
211/// stream polls the receiver.
212///
213/// Stopping: when the client goes away, axum drops the body stream, which
214/// drops the receiver. Every `is_closed` check in the walkers then yields
215/// `WalkState::Quit` at the next entry, so the walk unwinds promptly.
216fn search_stream(
217 q: String,
218 want_name: bool,
219 want_content: bool,
220 root: RootRow,
221 start_rel: String,
222 server_root: PathBuf,
223 slot: tokio::sync::OwnedSemaphorePermit,
224) -> impl futures_util::Stream<Item = SearchEvent> + Send {
225 let (tx, rx) = tokio::sync::mpsc::channel::<SearchEvent>(256);
226 let state = Arc::new(SearchState {
227 tx,
228 words: q.split_whitespace().map(|w| w.to_lowercase()).collect(),
229 root,
230 start_rel,
231 server_root,
232 started: Instant::now(),
233 scanned: AtomicUsize::new(0),
234 skipped: AtomicUsize::new(0),
235 names: AtomicUsize::new(0),
236 matches: AtomicUsize::new(0),
237 });
238
239 std::thread::spawn(move || {
240 // Released when the walk is done, not when the client stops reading.
241 let _slot = slot;
242 walk(&state, &q, want_name, want_content);
243 // `stopped`: the receiver went away (client stopped or navigated)
244 // or the match cap was hit, before the walk finished.
245 let stopped = state.tx.is_closed()
246 || state.matches.load(Ordering::Relaxed) >= SEARCH_MAX_MATCH_EVENTS;
247 let _ = state.tx.blocking_send(SearchEvent::Done {
248 stopped,
249 files: state.names.load(Ordering::Relaxed),
250 matches: state.matches.load(Ordering::Relaxed),
251 scanned: state.scanned.load(Ordering::Relaxed),
252 skipped: state.skipped.load(Ordering::Relaxed),
253 elapsed_ms: state.started.elapsed().as_millis() as u64,
254 });
255 });
256
257 tokio_stream::wrappers::ReceiverStream::new(rx)
258}
259
260/// True when every query word occurs in `name`. Both are already lowercased.
261///
262/// `name` is the entry's own name, never its path — see the module docs.
263fn name_matches(words: &[String], name: &str) -> bool {
264 words.iter().all(|w| name.contains(w.as_str()))
265}
266
267/// Walks the root in parallel, matching each entry against the wanted
268/// scopes. The visitor is cloned once per worker thread, and returns
269/// [`WalkState::Quit`] to stop the whole walk (client went away).
270fn walk(st: &Arc<SearchState>, q: &str, want_name: bool, want_content: bool) {
271 let abs = st.server_root.join(&st.root.path);
272 let start = abs.join(&st.start_rel);
273 if !start.is_dir() {
274 // Root or start folder removed out from under us: nothing to search.
275 return;
276 }
277 WalkBuilder::new(&start)
278 // A file browser shows everything, so search must too: no hidden
279 // or `.gitignore`/`.ignore` filtering. Symlinks stay unfollowed.
280 .standard_filters(false)
281 .threads(walker_threads())
282 .build_parallel()
283 .run(|| {
284 let abs = abs.clone();
285 Box::new(move |result| match result {
286 Ok(entry) => visit(st, q, want_name, want_content, &entry, &abs),
287 Err(_) => WalkState::Continue, // unreadable entry: skip like fd
288 })
289 });
290}
291
292/// One walked entry: emit a name hit, grep it, or both.
293fn visit(
294 st: &Arc<SearchState>,
295 q: &str,
296 want_name: bool,
297 want_content: bool,
298 entry: &DirEntry,
299 abs: &Path,
300) -> WalkState {
301 if st.tx.is_closed() {
302 return WalkState::Quit;
303 }
304 st.scanned.fetch_add(1, Ordering::Relaxed);
305 // The start folder itself (depth 0) is not a result. Paths stay
306 // relative to the root, not to the start folder, so the client opens
307 // them the same way as any listing entry.
308 if entry.depth() == 0 {
309 return WalkState::Continue;
310 }
311 let Some(rel) = entry.path().strip_prefix(abs).ok() else {
312 return WalkState::Continue;
313 };
314 // Original case, unlike the name used for matching: this path is what
315 // the client opens the entry by.
316 let rel = rel.to_string_lossy().replace('\\', "/");
317 let is_dir = entry.file_type().is_some_and(|t| t.is_dir());
318
319 if want_name {
320 // Matched against the entry's own name, not its path: matching the
321 // path makes every descendant of a matching directory a hit too
322 // ("e" matching `search-test/` dragged in all 400 files under it),
323 // which buries the entries the user actually named.
324 let name = entry.file_name().to_string_lossy().to_lowercase();
325 if name_matches(&st.words, &name) {
326 st.names.fetch_add(1, Ordering::Relaxed);
327 let size = if is_dir {
328 0
329 } else {
330 entry.metadata().map(|m| m.len()).unwrap_or(0)
331 };
332 let ev = SearchEvent::File {
333 root_id: st.root.id,
334 path: rel.clone(),
335 size,
336 is_dir,
337 // Same sniff a directory listing does, so the client needs no
338 // extension table of its own. One open() per name hit, on the
339 // walker thread that already stat'ed the entry.
340 kind: crate::fs::detect_kind(entry.path(), is_dir),
341 };
342 // blocking_send: the channel cap provides backpressure against a
343 // slow client. Err means the receiver is gone: stop the search.
344 if st.tx.blocking_send(ev).is_err() {
345 return WalkState::Quit;
346 }
347 }
348 }
349
350 if want_content && entry.file_type().is_some_and(|t| t.is_file()) {
351 if entry.metadata().map(|m| m.len()).unwrap_or(0) > SEARCH_MAX_FILE_BYTES {
352 st.skipped.fetch_add(1, Ordering::Relaxed);
353 return WalkState::Continue;
354 }
355 search_one_file(q, entry.path(), rel, st);
356 if st.tx.is_closed() || st.matches.load(Ordering::Relaxed) >= SEARCH_MAX_MATCH_EVENTS {
357 return WalkState::Quit;
358 }
359 }
360 WalkState::Continue
361}
362
363/// Grep one file with a fresh literal matcher. The matcher is built per file
364/// rather than shared: `RegexMatcher` is not `Sync`, so it cannot cross the
365/// walk's thread boundary; a case-insensitive literal compiles in
366/// microseconds, negligible next to the file read it protects.
367fn search_one_file(q: &str, path: &Path, rel: String, st: &SearchState) {
368 let matcher = match RegexMatcherBuilder::new()
369 .case_insensitive(true)
370 .build_literals(&[q])
371 {
372 Ok(m) => m,
373 Err(_) => return,
374 };
375 let mut searcher = SearcherBuilder::new().line_number(true).build();
376 // ripgrep's default: a NUL byte means "binary, stop".
377 searcher.set_binary_detection(BinaryDetection::quit(0));
378
379 let mut sink = MatchSink { st, lines: 0, rel };
380 // I/O errors (permissions, vanished file): skip like fd does.
381 let _ = searcher.search_path(matcher, path, &mut sink);
382}
383
384/// Pushes each matched line onto the event channel. `blocking_send` is
385/// correct here: it runs on the walk's worker threads, and the channel cap
386/// provides backpressure against a slow client.
387struct MatchSink<'a> {
388 st: &'a SearchState,
389 /// Matched lines emitted for the current file.
390 lines: usize,
391 rel: String,
392}
393
394impl Sink for MatchSink<'_> {
395 type Error = std::io::Error;
396
397 fn matched(&mut self, _searcher: &Searcher, mat: &SinkMatch) -> Result<bool, Self::Error> {
398 if self.st.tx.is_closed() || self.lines >= SEARCH_MAX_LINES_PER_FILE {
399 return Ok(false);
400 }
401 let line = match mat.line_number() {
402 Some(n) => n,
403 None => return Ok(true),
404 };
405 // Single-line matcher: exactly one line per match.
406 let text = match mat.lines().next() {
407 Some(l) => truncate_line(l),
408 None => return Ok(true),
409 };
410 // Cap check before the increment, so the counted number of matches
411 // is exactly the number of matches delivered.
412 if self.st.matches.load(Ordering::Relaxed) >= SEARCH_MAX_MATCH_EVENTS {
413 return Ok(false);
414 }
415 self.lines += 1;
416 self.st.matches.fetch_add(1, Ordering::Relaxed);
417 match self.st.tx.blocking_send(SearchEvent::Match {
418 root_id: self.st.root.id,
419 path: self.rel.clone(),
420 line,
421 text,
422 }) {
423 Ok(()) => Ok(true),
424 // Receiver gone: stop this file; the walker quits on the next
425 // `is_closed` check.
426 Err(_) => Ok(false),
427 }
428 }
429}
430
431/// Strips the line terminator and truncates at [`SEARCH_MAX_LINE_BYTES`].
432///
433/// A character split by the cap decodes to one replacement character, as
434/// does any invalid byte; the cap is therefore a byte cap on the input, not
435/// exactly on the output.
436fn truncate_line(line: &[u8]) -> String {
437 let end = line
438 .iter()
439 .rposition(|b| *b != b'\n' && *b != b'\r')
440 .map_or(0, |i| i + 1);
441 String::from_utf8_lossy(&line[..end.min(SEARCH_MAX_LINE_BYTES)]).into_owned()
442}
443
444#[cfg(test)]
445mod tests {
446 use super::*;
447
448 /// A rename of one of the `P_*` constants without the matching field
449 /// rename would silently stop the server from reading the parameter the
450 /// client sends. This builds the query string from the constants and
451 /// runs the real extractor over it.
452 #[test]
453 fn query_fields_are_the_shared_constants() {
454 use api_types::{P_PATH, P_Q, P_ROOT, P_SCOPE};
455 let uri: axum::http::Uri =
456 format!("/api/search?{P_Q}=invoice&{P_SCOPE}=both&{P_ROOT}=7&{P_PATH}=docs")
457 .parse()
458 .unwrap();
459 let q: SearchQuery = AxumQuery::try_from_uri(&uri).unwrap().0;
460 assert_eq!(q.q.as_deref(), Some("invoice"));
461 assert_eq!(q.scope.as_deref(), Some("both"));
462 assert_eq!(q.root, Some(7));
463 assert_eq!(q.path.as_deref(), Some("docs"));
464 }
465
466 #[test]
467 fn name_match_needs_every_word() {
468 let words = vec!["invoice".to_string(), "2025".to_string()];
469 assert!(name_matches(&words, "invoice_2025-11_final.pdf"));
470 assert!(!name_matches(&words, "invoice_2024.pdf"));
471 }
472
473 #[test]
474 fn name_match_ignores_the_parent_path() {
475 // The caller passes the entry's own name, so a file does not match
476 // just because an ancestor directory does.
477 let words = vec!["test".to_string()];
478 assert!(name_matches(&words, "test-notes.md"));
479 assert!(!name_matches(&words, "f12.txt"));
480 }
481
482 #[test]
483 fn truncation_strips_terminator() {
484 assert_eq!(truncate_line(b"hello\n"), "hello");
485 assert_eq!(truncate_line(b"hello\r\n"), "hello");
486 assert_eq!(truncate_line(b"no newline"), "no newline");
487 assert_eq!(truncate_line(b"\n"), "");
488 }
489
490 #[test]
491 fn truncation_caps_length() {
492 let long = vec![b'x'; 600];
493 let t = truncate_line(&long);
494 assert_eq!(t.len(), SEARCH_MAX_LINE_BYTES);
495 }
496
497 #[test]
498 fn truncation_keeps_utf8_boundary() {
499 // 3-byte chars; the cap must not split one.
500 let long: Vec<u8> = "ä".repeat(200).into_bytes();
501 let t = truncate_line(&long);
502 assert!(t.len() <= SEARCH_MAX_LINE_BYTES);
503 assert!(t.chars().all(|c| c == 'ä'));
504 }
505}
506