From 83d71a7041224e8b86e0c8d919584af2f785d26d Mon Sep 17 00:00:00 2001 From: rgdevment Date: Sun, 20 Sep 2026 22:25:44 -0300 Subject: [PATCH 1/7] feat(core): what the panel draws, paste as, and the sweep MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The mockup lists everything a card shows; the store only gave five of those fields. `Listed` now carries app, label, colour, thumbnail, paste count, last use, creation and breakage, and `list` returns a `Page` with an opaque cursor keyed on the order so paging never repeats or skips under ties. `facets` counts the classes within the current search and `distinct_apps` the source applications. `Filter` grows exclusion of classes and applications, application equality on the folded column, `since` on `modified_at` (a recopy is a copy), broken hidden by default and askable, a label-scoped query and three indexed orders. `parse` turns `token @code ~1h` into a `Filter` through one table of keys and symbols, treating a prefix as an operator only when its value is known. Searching marks the fragment that matched, cut from the text as it was copied: `search::excerpt` maps folded matches back to the original bytes and says whether it hit the text, the label, the application or what was read inside an image. Folding no longer allocates per character — 64 KB in 0.05 ms instead of 2.6 — which is what makes a snippet per row affordable. `sweep` applies a `Policy` in one pass — broken past their grace, expired, beyond a count, beyond a byte quota, blob files nobody references — and checkpoints once. Purging a broken item releases its blobs; `update_text` edits in place, drops the renderings that no longer match and checkpoints so the old text does not stay readable in the database or its log; `mark_present` undoes `mark_broken`. The blob sweep only touches digest-shaped names and reclaiming a blob refreshes its mtime. Schema version 2 rebuilds the ordering indexes that gained an `id` tiebreaker. `paste_as` takes a neutral `Content` and answers `forms_for` and `render`: JSON pretty, minified, keys and TSV table; colour as hex, rgb and hsl with alpha; link as Markdown, domain or with its title; code as one line, a fenced block or dedented; token as an Authorization header, a curl or the decoded JWT claims; image as JPEG on white or as its OCR text; files as paths; rich text as Markdown through a small HTML converter. `Kind::Token` is the fifteenth class: a JWT or a known prefix, no long-hex heuristic. `serde_json` joins the core. `watch::insist` retries a too-slow source five times with 150 ms between attempts and gives up as soon as the change count moves; `cp_mac::capture::capture_insisting` wires it and probe C6 checks a source that answers costs no pause. macOS builds the `Content` from its formats and restores the synthetic ones the store writes, the way Windows already did. Windows takes the rest in its own session. --- .cargo/mutants.toml | 7 +- Cargo.lock | 86 +- Cargo.toml | 1 + crates/cp-core/Cargo.toml | 1 + crates/cp-core/src/kind.rs | 59 + crates/cp-core/src/lib.rs | 2 + crates/cp-core/src/paste_as.rs | 1402 ++++++++++++++++++ crates/cp-core/src/search.rs | 391 ++++- crates/cp-core/src/token.rs | 267 ++++ crates/cp-core/src/watch.rs | 160 ++ crates/cp-mac-sys/src/frontmost.rs | 15 +- crates/cp-mac/examples/probe/battery.rs | 22 + crates/cp-mac/src/capture.rs | 93 ++ crates/cp-mac/src/content.rs | 163 ++ crates/cp-mac/src/lib.rs | 1 + crates/cp-mac/src/restore.rs | 105 +- crates/cp-store/src/blobs.rs | 181 ++- crates/cp-store/src/lib.rs | 11 +- crates/cp-store/src/query.rs | 394 +++++ crates/cp-store/src/schema.rs | 68 +- crates/cp-store/src/store.rs | 1806 +++++++++++++++++++++-- 21 files changed, 5037 insertions(+), 198 deletions(-) create mode 100644 crates/cp-core/src/paste_as.rs create mode 100644 crates/cp-core/src/token.rs create mode 100644 crates/cp-mac/src/content.rs create mode 100644 crates/cp-store/src/query.rs diff --git a/.cargo/mutants.toml b/.cargo/mutants.toml index 5917349..035c654 100644 --- a/.cargo/mutants.toml +++ b/.cargo/mutants.toml @@ -1,6 +1,11 @@ -exclude_re = ["replace > with >= in of_image"] +exclude_re = [ + "replace > with >= in of_image", + "replace - with / in Store::evict_until_under", + "replace \\| with \\^ in base64url", + "replace < with <= in rgb_of_hsl", +] exclude_globs = [ "crates/cp-win/src/capture.rs", diff --git a/Cargo.lock b/Cargo.lock index a38cc36..f60fade 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -109,6 +109,7 @@ version = "3.0.0" dependencies = [ "image", "proptest", + "serde_json", "thiserror", "unicode-normalization", "xxhash-rust", @@ -205,6 +206,12 @@ dependencies = [ "objc2", ] +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + [[package]] name = "errno" version = "0.3.14" @@ -330,13 +337,19 @@ dependencies = [ "foldhash", ] +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" + [[package]] name = "hashlink" version = "0.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7382cf6263419f2d8df38c55d7da83da5c18aef87fc7a7fc1fb1e344edfe14c1" dependencies = [ - "hashbrown", + "hashbrown 0.15.5", ] [[package]] @@ -368,6 +381,22 @@ dependencies = [ "quick-error 2.0.1", ] +[[package]] +name = "indexmap" +version = "2.14.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" +dependencies = [ + "equivalent", + "hashbrown 0.17.1", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + [[package]] name = "libc" version = "0.2.189" @@ -391,6 +420,12 @@ version = "0.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + [[package]] name = "miniz_oxide" version = "0.8.9" @@ -942,6 +977,49 @@ dependencies = [ "wait-timeout", ] +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.5", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "indexmap", + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + [[package]] name = "shlex" version = "2.0.1" @@ -1252,6 +1330,12 @@ version = "0.6.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" + [[package]] name = "zune-core" version = "0.5.3" diff --git a/Cargo.toml b/Cargo.toml index cecba3b..3896f8f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -26,6 +26,7 @@ rusqlite = { version = "0.37", features = ["bundled"] } blake3 = "1" xxhash-rust = { version = "0.8", features = ["xxh3"] } unicode-normalization = "0.1" +serde_json = { version = "1", features = ["preserve_order"] } image = { version = "0.25", default-features = false, features = ["png", "jpeg", "tiff", "gif", "bmp", "webp"] } proptest = "1" tempfile = "3" diff --git a/crates/cp-core/Cargo.toml b/crates/cp-core/Cargo.toml index 4f3bcbd..e0ad5f6 100644 --- a/crates/cp-core/Cargo.toml +++ b/crates/cp-core/Cargo.toml @@ -10,6 +10,7 @@ publish = false [dependencies] thiserror.workspace = true unicode-normalization.workspace = true +serde_json.workspace = true xxhash-rust.workspace = true image.workspace = true diff --git a/crates/cp-core/src/kind.rs b/crates/cp-core/src/kind.rs index 3dacfea..c033c1c 100644 --- a/crates/cp-core/src/kind.rs +++ b/crates/cp-core/src/kind.rs @@ -14,9 +14,32 @@ pub enum Kind { Folder, Audio, Video, + Token, } impl Kind { + pub const ALL: [Kind; 15] = [ + Kind::Text, + Kind::Code, + Kind::Json, + Kind::Link, + Kind::Email, + Kind::Phone, + Kind::Color, + Kind::Ip, + Kind::Uuid, + Kind::Image, + Kind::File, + Kind::Folder, + Kind::Audio, + Kind::Video, + Kind::Token, + ]; + + pub fn from_name(name: &str) -> Option { + Kind::ALL.into_iter().find(|kind| kind.as_str() == name) + } + pub fn as_str(self) -> &'static str { match self { Kind::Text => "text", @@ -33,6 +56,7 @@ impl Kind { Kind::Folder => "folder", Kind::Audio => "audio", Kind::Video => "video", + Kind::Token => "token", } } } @@ -44,6 +68,9 @@ pub fn classify_text(content: &str) -> Kind { } if !text.contains('\n') { + if crate::token::looks_like(text) { + return Kind::Token; + } if is_email(text) { return Kind::Email; } @@ -276,6 +303,24 @@ fn looks_like_code(text: &str) -> bool { mod tests { use super::*; + #[test] + fn every_class_goes_to_its_name_and_back() { + for kind in Kind::ALL { + assert_eq!(Kind::from_name(kind.as_str()), Some(kind)); + } + let mut names: Vec<&str> = Kind::ALL.iter().map(|kind| kind.as_str()).collect(); + names.sort_unstable(); + names.dedup(); + assert_eq!(names.len(), 15, "quince clases, quince nombres distintos"); + } + + #[test] + fn a_name_that_is_not_a_class_is_nobody() { + assert_eq!(Kind::from_name("Text"), None, "el nombre es exacto"); + assert_eq!(Kind::from_name("imagen"), None); + assert_eq!(Kind::from_name(""), None); + } + #[test] fn the_classes_2x_already_recognised() { assert_eq!(classify_text("alguien@ejemplo.test"), Kind::Email); @@ -291,6 +336,20 @@ mod tests { assert_eq!(classify_text(r#"{"a": 1}"#), Kind::Json); } + #[test] + fn a_token_is_its_own_class_and_wins_over_the_rest() { + assert_eq!( + classify_text("ghp_not_a_real_token_for_tests_0000000000"), + Kind::Token + ); + assert_eq!( + classify_text(" sk-not-a-real-key-for-tests-0123456789\n"), + Kind::Token, + "recortado, como todo lo demás" + ); + assert_eq!(classify_text("sk-corto"), Kind::Text); + } + #[test] fn the_classes_2x_had_a_type_for_but_never_assigned() { assert_eq!(classify_text("https://ejemplo.test/ruta?q=1"), Kind::Link); diff --git a/crates/cp-core/src/lib.rs b/crates/cp-core/src/lib.rs index d65efb4..3c546a1 100644 --- a/crates/cp-core/src/lib.rs +++ b/crates/cp-core/src/lib.rs @@ -5,8 +5,10 @@ pub mod hash; pub mod item; pub mod kind; pub mod paste; +pub mod paste_as; pub mod search; pub mod thumbnail; +pub mod token; pub mod watch; pub use formats::Family; diff --git a/crates/cp-core/src/paste_as.rs b/crates/cp-core/src/paste_as.rs new file mode 100644 index 0000000..70aada5 --- /dev/null +++ b/crates/cp-core/src/paste_as.rs @@ -0,0 +1,1402 @@ +use crate::kind::Kind; +use serde_json::Value; + +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct Content<'a> { + pub kind: Option, + pub text: Option<&'a str>, + pub html: Option<&'a str>, + pub rich: bool, + pub png: Option<&'a [u8]>, + pub paths: Vec, + pub title: Option<&'a str>, + pub ocr: Option<&'a str>, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Form { + AsIs, + PlainText, + Markdown, + JsonPretty, + JsonMinified, + JsonKeys, + JsonTable, + ColorHex, + ColorRgb, + ColorHsl, + LinkMarkdown, + LinkDomain, + LinkTitled, + CodeOneLine, + CodeBlock, + CodeDedented, + TokenHeader, + TokenClaims, + TokenCurl, + ImageJpeg, + ImageOcr, + Path, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Rendered { + Text(String), + Png(Vec), + Jpeg(Vec), +} + +pub const JPEG_QUALITY: u8 = 85; + +pub fn forms_for(content: &Content) -> Vec
{ + let mut forms = vec![Form::AsIs]; + if content.rich && content.text.is_some() { + forms.push(Form::PlainText); + } + if content.html.is_some() { + forms.push(Form::Markdown); + } + let text = content.text.map(str::trim).unwrap_or_default(); + match content.kind { + Some(Kind::Json) => { + if let Ok(value) = serde_json::from_str::(text) { + forms.extend([Form::JsonPretty, Form::JsonMinified]); + if json_keys(&value).is_some() { + forms.push(Form::JsonKeys); + } + if json_table(&value).is_some() { + forms.push(Form::JsonTable); + } + } + } + Some(Kind::Color) => { + if parse_color(text).is_some() { + forms.extend([Form::ColorHex, Form::ColorRgb, Form::ColorHsl]); + } + } + Some(Kind::Link) => { + forms.extend([Form::LinkMarkdown, Form::LinkDomain]); + if content.title.is_some_and(|title| !title.trim().is_empty()) { + forms.push(Form::LinkTitled); + } + } + Some(Kind::Code) => { + forms.extend([Form::CodeOneLine, Form::CodeBlock, Form::CodeDedented]); + } + Some(Kind::Token) => { + forms.extend([Form::TokenHeader, Form::TokenCurl]); + if crate::token::claims_of(text).is_some() { + forms.push(Form::TokenClaims); + } + } + _ => {} + } + if content.png.is_some() { + forms.push(Form::ImageJpeg); + if content.ocr.is_some_and(|ocr| !ocr.trim().is_empty()) { + forms.push(Form::ImageOcr); + } + } + if !content.paths.is_empty() { + forms.push(Form::Path); + } + forms +} + +pub fn render(form: Form, content: &Content) -> Option { + let text = content.text.map(str::trim).unwrap_or_default(); + let rendered = match form { + Form::AsIs => { + return match (content.text, content.png) { + (Some(text), _) => Some(Rendered::Text(text.to_owned())), + (None, Some(png)) => Some(Rendered::Png(png.to_vec())), + (None, None) => None, + }; + } + Form::PlainText => content.text?.to_owned(), + Form::Markdown => markdown_of_html(content.html?), + Form::JsonPretty => serde_json::to_string_pretty(&json_of(text)?).ok()?, + Form::JsonMinified => serde_json::to_string(&json_of(text)?).ok()?, + Form::JsonKeys => json_keys(&json_of(text)?)?, + Form::JsonTable => json_table(&json_of(text)?)?, + Form::ColorHex => hex_of(parse_color(text)?), + Form::ColorRgb => rgb_of(parse_color(text)?), + Form::ColorHsl => hsl_of(parse_color(text)?), + Form::LinkMarkdown => format!("[{}]({text})", content.title.map(str::trim).unwrap_or(text)), + Form::LinkDomain => domain_of(text)?.to_owned(), + Form::LinkTitled => format!("{} — {text}", content.title?.trim()), + Form::CodeOneLine => one_line(text), + Form::CodeBlock => format!("```\n{text}\n```"), + Form::CodeDedented => dedent(content.text?), + Form::TokenHeader => format!("Authorization: Bearer {text}"), + Form::TokenClaims => { + let claims = crate::token::claims_of(text)?; + serde_json::to_string_pretty(&Value::Object(claims.payload)).ok()? + } + Form::TokenCurl => format!("curl -H 'Authorization: Bearer {text}' \"$URL\""), + Form::ImageJpeg => return jpeg_of(content.png?).map(Rendered::Jpeg), + Form::ImageOcr => content.ocr?.trim().to_owned(), + Form::Path => content.paths.join("\n"), + }; + Some(Rendered::Text(rendered)) +} + +fn json_of(text: &str) -> Option { + serde_json::from_str(text).ok() +} + +fn json_keys(value: &Value) -> Option { + let keys: Vec<&str> = match value { + Value::Object(map) => map.keys().map(String::as_str).collect(), + Value::Array(rows) => { + let mut seen = Vec::new(); + for row in rows { + let Value::Object(map) = row else { + return None; + }; + for key in map.keys() { + if !seen.contains(&key.as_str()) { + seen.push(key.as_str()); + } + } + } + seen + } + _ => return None, + }; + (!keys.is_empty()).then(|| keys.join("\n")) +} + +fn json_table(value: &Value) -> Option { + match value { + Value::Object(map) => { + let rows: Vec = map + .iter() + .map(|(key, value)| format!("{key}\t{}", cell(value))) + .collect(); + (!rows.is_empty()).then(|| rows.join("\n")) + } + Value::Array(rows) => { + let header = json_keys(value)?; + let columns: Vec<&str> = header.lines().collect(); + let mut lines = vec![columns.join("\t")]; + for row in rows { + let Value::Object(map) = row else { + return None; + }; + let cells: Vec = columns + .iter() + .map(|column| map.get(*column).map(cell).unwrap_or_default()) + .collect(); + lines.push(cells.join("\t")); + } + Some(lines.join("\n")) + } + _ => None, + } +} + +fn cell(value: &Value) -> String { + match value { + Value::String(text) => text.replace(['\t', '\n'], " "), + Value::Null => String::new(), + other => other.to_string(), + } +} + +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct Rgba { + pub r: u8, + pub g: u8, + pub b: u8, + pub a: f64, +} + +pub fn parse_color(text: &str) -> Option { + if let Some(hex) = text.strip_prefix('#') { + let digits: Vec = hex + .chars() + .map(|c| c.to_digit(16).map(|d| d as u8)) + .collect::>()?; + let (r, g, b, a) = match digits.as_slice() { + [r, g, b] => (r * 17, g * 17, b * 17, 255), + [r, r2, g, g2, b, b2] => (r * 16 + r2, g * 16 + g2, b * 16 + b2, 255), + [r, r2, g, g2, b, b2, a, a2] => (r * 16 + r2, g * 16 + g2, b * 16 + b2, a * 16 + a2), + _ => return None, + }; + return Some(Rgba { + r, + g, + b, + a: f64::from(a) / 255.0, + }); + } + let lower = text.to_ascii_lowercase(); + let (kind, inner) = lower + .split_once('(') + .and_then(|(kind, rest)| Some((kind, rest.strip_suffix(')')?)))?; + let parts: Vec = inner + .split(',') + .map(|part| part.trim().trim_end_matches('%').parse::().ok()) + .collect::>()?; + let alpha = |value: Option<&f64>| value.copied().unwrap_or(1.0).clamp(0.0, 1.0); + match (kind, parts.as_slice()) { + ("rgb", [r, g, b]) | ("rgba", [r, g, b, _]) => Some(Rgba { + r: channel(*r), + g: channel(*g), + b: channel(*b), + a: alpha(parts.get(3)), + }), + ("hsl", [h, s, l]) | ("hsla", [h, s, l, _]) => { + let (r, g, b) = rgb_of_hsl(*h, s / 100.0, l / 100.0); + Some(Rgba { + r, + g, + b, + a: alpha(parts.get(3)), + }) + } + _ => None, + } +} + +fn channel(value: f64) -> u8 { + value.round().clamp(0.0, 255.0) as u8 +} + +fn rgb_of_hsl(h: f64, s: f64, l: f64) -> (u8, u8, u8) { + let h = h.rem_euclid(360.0) / 360.0; + let (s, l) = (s.clamp(0.0, 1.0), l.clamp(0.0, 1.0)); + let q = if l < 0.5 { + l * (1.0 + s) + } else { + l + s - l * s + }; + let p = 2.0 * l - q; + let hue = |mut t: f64| { + t = t.rem_euclid(1.0); + let value = if t < 1.0 / 6.0 { + p + (q - p) * 6.0 * t + } else if t < 0.5 { + q + } else if t < 2.0 / 3.0 { + p + (q - p) * (2.0 / 3.0 - t) * 6.0 + } else { + p + }; + channel(value * 255.0) + }; + (hue(h + 1.0 / 3.0), hue(h), hue(h - 1.0 / 3.0)) +} + +fn hex_of(color: Rgba) -> String { + if color.a < 1.0 { + format!( + "#{:02X}{:02X}{:02X}{:02X}", + color.r, + color.g, + color.b, + channel(color.a * 255.0) + ) + } else { + format!("#{:02X}{:02X}{:02X}", color.r, color.g, color.b) + } +} + +fn rgb_of(color: Rgba) -> String { + if color.a < 1.0 { + format!( + "rgba({}, {}, {}, {})", + color.r, + color.g, + color.b, + trim_float(color.a) + ) + } else { + format!("rgb({}, {}, {})", color.r, color.g, color.b) + } +} + +fn hsl_of(color: Rgba) -> String { + let (r, g, b) = ( + f64::from(color.r) / 255.0, + f64::from(color.g) / 255.0, + f64::from(color.b) / 255.0, + ); + let max = r.max(g).max(b); + let min = r.min(g).min(b); + let l = (max + min) / 2.0; + let delta = max - min; + let (h, s) = if delta == 0.0 { + (0.0, 0.0) + } else { + let s = delta / (1.0 - (2.0 * l - 1.0).abs()); + let h = if max == r { + 60.0 * (((g - b) / delta).rem_euclid(6.0)) + } else if max == g { + 60.0 * ((b - r) / delta + 2.0) + } else { + 60.0 * ((r - g) / delta + 4.0) + }; + (h, s) + }; + let (h, s, l) = ( + h.round() as i64, + (s * 100.0).round() as i64, + (l * 100.0).round() as i64, + ); + if color.a < 1.0 { + format!("hsla({h}, {s}%, {l}%, {})", trim_float(color.a)) + } else { + format!("hsl({h}, {s}%, {l}%)") + } +} + +fn trim_float(value: f64) -> String { + let text = format!("{value:.2}"); + text.trim_end_matches('0').trim_end_matches('.').to_owned() +} + +pub fn domain_of(url: &str) -> Option<&str> { + let rest = match url.trim().split_once(':') { + Some((scheme, rest)) + if scheme.starts_with(|c: char| c.is_ascii_alphabetic()) + && scheme + .chars() + .all(|c| c.is_ascii_alphanumeric() || "+-.".contains(c)) => + { + rest.trim_start_matches('/') + } + _ => url.trim(), + }; + if rest.is_empty() { + return None; + } + let authority = rest.split(['/', '?', '#']).next()?; + let host = authority.rsplit('@').next()?; + let host = match host.rsplit_once(':') { + Some((name, port)) if !port.is_empty() && port.chars().all(|c| c.is_ascii_digit()) => name, + _ => host, + }; + let host = host.strip_prefix("www.").unwrap_or(host); + (!host.is_empty() && host.contains('.')).then_some(host) +} + +fn one_line(text: &str) -> String { + text.lines() + .map(str::trim) + .filter(|line| !line.is_empty()) + .collect::>() + .join(" ") +} + +fn dedent(text: &str) -> String { + let indent = text + .lines() + .filter(|line| !line.trim().is_empty()) + .map(|line| line.len() - line.trim_start().len()) + .min() + .unwrap_or(0); + let mut out = text + .lines() + .map(|line| { + if line.len() >= indent { + &line[indent..] + } else { + line.trim_start() + } + }) + .collect::>() + .join("\n"); + if text.ends_with('\n') { + out.push('\n'); + } + out +} + +fn jpeg_of(png: &[u8]) -> Option> { + let decoded = image::load_from_memory(png).ok()?.to_rgba8(); + let (width, height) = decoded.dimensions(); + let mut flat = image::RgbImage::new(width, height); + for (x, y, pixel) in decoded.enumerate_pixels() { + let alpha = u32::from(pixel[3]); + let over_white = |c: u8| ((u32::from(c) * alpha + 255 * (255 - alpha)) / 255) as u8; + flat.put_pixel( + x, + y, + image::Rgb([ + over_white(pixel[0]), + over_white(pixel[1]), + over_white(pixel[2]), + ]), + ); + } + let mut out = Vec::new(); + let encoder = image::codecs::jpeg::JpegEncoder::new_with_quality(&mut out, JPEG_QUALITY); + flat.write_with_encoder(encoder).ok()?; + Some(out) +} + +pub fn markdown_of_html(html: &str) -> String { + let fragment = fragment_of(html); + let mut writer = MarkdownWriter::default(); + let mut rest = fragment; + while !rest.is_empty() { + if let Some(after) = rest.strip_prefix('<') + && after + .chars() + .next() + .is_some_and(|c| c.is_ascii_alphabetic() || "/!?".contains(c)) + && let Some(end) = after.find('>') + { + writer.tag(&after[..end]); + rest = &after[end + 1..]; + continue; + } + let next = rest[1..].find('<').map_or(rest.len(), |at| at + 1); + writer.text(&decode_entities(&rest[..next])); + rest = &rest[next..]; + } + writer.finish() +} + +fn fragment_of(html: &str) -> &str { + let start = html + .find("") + .map_or(0, |at| at + "".len()); + let end = html.find("").unwrap_or(html.len()); + if start <= end { + &html[start..end] + } else { + html + } +} + +const SKIPPED: [&str; 4] = ["script", "style", "head", "title"]; + +struct MarkdownWriter { + out: String, + lists: Vec>, + link: Option<(String, usize)>, + skipping: Option<&'static str>, + preformatted: bool, + quoting: bool, + opening: String, + fresh: bool, +} + +impl Default for MarkdownWriter { + fn default() -> Self { + Self { + out: String::new(), + lists: Vec::new(), + link: None, + skipping: None, + preformatted: false, + quoting: false, + opening: String::new(), + fresh: true, + } + } +} + +impl MarkdownWriter { + fn tag(&mut self, raw: &str) { + let raw = raw.trim(); + if raw.starts_with('!') { + return; + } + let closing = raw.starts_with('/'); + let body = raw.trim_start_matches('/').trim_end_matches('/').trim(); + let name = body + .split(|c: char| c.is_whitespace()) + .next() + .unwrap_or_default() + .to_ascii_lowercase(); + if let Some(skipped) = self.skipping { + if closing && name == skipped { + self.skipping = None; + } + return; + } + match (name.as_str(), closing) { + ("script" | "style" | "head" | "title", false) => { + self.skipping = SKIPPED.iter().copied().find(|tag| *tag == name); + } + ("br", _) => { + if !self.at_line_start() { + self.out.push('\n'); + self.fresh = true; + } + } + ("p" | "div" | "tr" | "table" | "section" | "article", _) => self.blank_line(), + ("h1" | "h2" | "h3" | "h4" | "h5" | "h6", false) => { + self.blank_line(); + let level = name[1..].parse::().unwrap_or(1); + self.out.push_str(&"#".repeat(level)); + self.out.push(' '); + } + ("h1" | "h2" | "h3" | "h4" | "h5" | "h6", true) => self.blank_line(), + ("b" | "strong", false) => self.open("**"), + ("i" | "em", false) => self.open("*"), + ("code", false) if !self.preformatted => self.open("`"), + ("b" | "strong", true) => self.close("**"), + ("i" | "em", true) => self.close("*"), + ("code", true) if !self.preformatted => self.close("`"), + ("pre", false) => { + self.blank_line(); + self.out.push_str("```\n"); + self.preformatted = true; + } + ("pre", true) => { + self.preformatted = false; + self.newline(); + self.out.push_str("```"); + self.blank_line(); + } + ("ul", false) => { + self.blank_line_if_top_level(); + self.lists.push(None); + } + ("ol", false) => { + self.blank_line_if_top_level(); + self.lists.push(Some(0)); + } + ("ul" | "ol", true) => { + self.lists.pop(); + if self.lists.is_empty() { + self.blank_line(); + } + } + ("li", false) => { + self.newline(); + let depth = self.lists.len().saturating_sub(1); + self.out.push_str(&" ".repeat(depth)); + match self.lists.last_mut() { + Some(Some(count)) => { + *count += 1; + self.out.push_str(&format!("{count}. ")); + } + _ => self.out.push_str("- "), + } + } + ("blockquote", false) => { + self.blank_line(); + self.quoting = true; + self.out.push_str("> "); + } + ("blockquote", true) => { + self.quoting = false; + self.blank_line(); + } + ("a", false) => { + let href = attribute(body, "href").unwrap_or_default(); + self.link = Some((href, self.out.len())); + } + ("a", true) => { + if let Some((href, from)) = self.link.take() { + let label = self.out[from..].trim().to_owned(); + self.out.truncate(from); + if href.is_empty() { + self.out.push_str(&label); + } else { + self.out.push_str(&format!("[{label}]({href})")); + } + } + } + ("img", false) => { + let alt = attribute(body, "alt").unwrap_or_default(); + let src = attribute(body, "src").unwrap_or_default(); + self.out.push_str(&format!("![{alt}]({src})")); + self.fresh = false; + } + ("td" | "th", true) => self.out.push('\t'), + ("hr", _) => { + self.blank_line(); + self.out.push_str("---"); + self.fresh = false; + self.blank_line(); + } + _ => {} + } + } + + fn open(&mut self, marker: &str) { + self.opening.push_str(marker); + } + + fn close(&mut self, marker: &str) { + if let Some(unopened) = self.opening.strip_suffix(marker) { + self.opening = unopened.to_owned(); + return; + } + let kept = self.out.trim_end_matches(' ').len(); + let had_space = kept < self.out.len(); + self.out.truncate(kept); + self.out.push_str(marker); + if had_space { + self.out.push(' '); + } + } + + fn text(&mut self, text: &str) { + if self.skipping.is_some() { + return; + } + if self.preformatted { + let opening = std::mem::take(&mut self.opening); + self.out.push_str(&opening); + self.out.push_str(text); + self.fresh = text.ends_with('\n'); + return; + } + let leading = text.starts_with(char::is_whitespace); + let trailing = text.ends_with(char::is_whitespace); + let words: Vec<&str> = text.split_whitespace().collect(); + if leading && !self.at_line_start() && !self.out.ends_with(' ') { + self.out.push(' '); + } + if words.is_empty() { + return; + } + let opening = std::mem::take(&mut self.opening); + self.out.push_str(&opening); + self.out.push_str(&words.join(" ")); + self.fresh = false; + if trailing { + self.out.push(' '); + } + } + + fn at_line_start(&self) -> bool { + self.fresh + } + + fn newline(&mut self) { + let trimmed = self.out.trim_end_matches(' ').len(); + self.out.truncate(trimmed); + if !self.out.is_empty() && !self.out.ends_with('\n') { + self.out.push('\n'); + } + self.fresh = true; + } + + fn blank_line(&mut self) { + self.newline(); + if !self.out.is_empty() && !self.out.ends_with("\n\n") { + self.out.push('\n'); + } + } + + fn blank_line_if_top_level(&mut self) { + if self.lists.is_empty() { + self.blank_line(); + } + } + + fn finish(mut self) -> String { + self.newline(); + let mut lines: Vec<&str> = self.out.lines().map(str::trim_end).collect(); + lines.dedup_by(|a, b| a.is_empty() && b.is_empty()); + lines.join("\n").trim_end().to_owned() + } +} + +fn attribute(tag: &str, name: &str) -> Option { + let lower = tag.to_ascii_lowercase(); + let at = lower + .find(&format!("{name}=")) + .map(|at| at + name.len() + 1)?; + let rest = &tag[at..]; + let value = match rest.chars().next()? { + quote @ ('"' | '\'') => rest[1..].split(quote).next()?, + _ => rest.split(char::is_whitespace).next()?, + }; + Some(decode_entities(value)) +} + +fn decode_entities(text: &str) -> String { + let mut out = String::with_capacity(text.len()); + let mut rest = text; + while let Some(at) = rest.find('&') { + out.push_str(&rest[..at]); + let after = &rest[at + 1..]; + let Some(end) = after.find(';').filter(|end| *end <= 10) else { + out.push('&'); + rest = after; + continue; + }; + let entity = &after[..end]; + let decoded = match entity { + "amp" => Some('&'), + "lt" => Some('<'), + "gt" => Some('>'), + "quot" => Some('"'), + "apos" => Some('\''), + "nbsp" => Some(' '), + _ => entity + .strip_prefix('#') + .and_then(|number| match number.strip_prefix(['x', 'X']) { + Some(hex) => u32::from_str_radix(hex, 16).ok(), + None => number.parse().ok(), + }) + .and_then(char::from_u32), + }; + match decoded { + Some(c) => { + out.push(c); + rest = &after[end + 1..]; + } + None => { + out.push('&'); + rest = after; + } + } + } + out.push_str(rest); + out +} + +#[cfg(test)] +mod tests { + use super::*; + + fn text_of(kind: Kind, text: &str) -> Content<'_> { + Content { + kind: Some(kind), + text: Some(text), + ..Default::default() + } + } + + fn rendered(form: Form, content: &Content) -> String { + match render(form, content).expect("se puede") { + Rendered::Text(text) => text, + other => panic!("no era texto: {other:?}"), + } + } + + #[test] + fn every_item_can_at_least_be_pasted_as_it_is() { + let content = text_of(Kind::Text, "hola"); + assert_eq!(forms_for(&content), vec![Form::AsIs]); + assert_eq!(rendered(Form::AsIs, &content), "hola"); + assert_eq!(render(Form::AsIs, &Content::default()), None); + } + + #[test] + fn a_rich_text_offers_plain_and_markdown_only_with_html() { + let with_html = Content { + kind: Some(Kind::Text), + text: Some("hola"), + html: Some("hola"), + rich: true, + ..Default::default() + }; + assert_eq!( + forms_for(&with_html), + vec![Form::AsIs, Form::PlainText, Form::Markdown] + ); + assert_eq!(rendered(Form::Markdown, &with_html), "**hola**"); + let rtf_only = Content { + rich: true, + ..text_of(Kind::Text, "hola") + }; + assert_eq!(forms_for(&rtf_only), vec![Form::AsIs, Form::PlainText]); + } + + #[test] + fn json_is_offered_formatted_minified_by_keys_and_as_a_table() { + let content = text_of( + Kind::Json, + r#"{"b": 1, "a": {"x": [1, 2]}, "c": "hola\tque tal"}"#, + ); + assert_eq!( + forms_for(&content), + vec![ + Form::AsIs, + Form::JsonPretty, + Form::JsonMinified, + Form::JsonKeys, + Form::JsonTable + ] + ); + assert_eq!( + rendered(Form::JsonPretty, &content), + "{\n \"b\": 1,\n \"a\": {\n \"x\": [\n 1,\n 2\n ]\n },\n \"c\": \"hola\\tque tal\"\n}", + "en el orden en que estaban las claves" + ); + assert_eq!( + rendered(Form::JsonMinified, &content), + r#"{"b":1,"a":{"x":[1,2]},"c":"hola\tque tal"}"# + ); + assert_eq!(rendered(Form::JsonKeys, &content), "b\na\nc"); + assert_eq!( + rendered(Form::JsonTable, &content), + "b\t1\na\t{\"x\":[1,2]}\nc\thola que tal", + "una tabulación entre clave y valor; los anidados van minificados" + ); + } + + #[test] + fn a_list_of_objects_becomes_a_table_with_a_header() { + let content = text_of( + Kind::Json, + r#"[{"name": "ana", "age": 3}, {"name": "bo", "city": "Lima"}]"#, + ); + assert_eq!(rendered(Form::JsonKeys, &content), "name\nage\ncity"); + assert_eq!( + rendered(Form::JsonTable, &content), + "name\tage\tcity\nana\t3\t\nbo\t\tLima", + "las columnas son la unión de claves; lo que falta queda vacío" + ); + } + + #[test] + fn json_that_is_not_an_object_has_no_keys_and_no_table() { + let content = text_of(Kind::Json, "[1, 2, 3]"); + assert_eq!( + forms_for(&content), + vec![Form::AsIs, Form::JsonPretty, Form::JsonMinified] + ); + assert_eq!(render(Form::JsonKeys, &content), None); + let broken = text_of(Kind::Json, "{no es json}"); + assert_eq!(forms_for(&broken), vec![Form::AsIs]); + } + + #[test] + fn a_colour_goes_round_the_three_notations() { + let content = text_of(Kind::Color, "#FF8800"); + assert_eq!( + forms_for(&content), + vec![Form::AsIs, Form::ColorHex, Form::ColorRgb, Form::ColorHsl] + ); + assert_eq!(rendered(Form::ColorHex, &content), "#FF8800"); + assert_eq!(rendered(Form::ColorRgb, &content), "rgb(255, 136, 0)"); + assert_eq!(rendered(Form::ColorHsl, &content), "hsl(32, 100%, 50%)"); + assert_eq!( + rendered(Form::ColorHex, &text_of(Kind::Color, "rgb(255, 136, 0)")), + "#FF8800" + ); + assert_eq!( + rendered(Form::ColorRgb, &text_of(Kind::Color, "hsl(32, 100%, 50%)")), + "rgb(255, 136, 0)" + ); + assert_eq!( + rendered(Form::ColorHex, &text_of(Kind::Color, "#f80")), + "#FF8800" + ); + } + + #[test] + fn transparency_survives_every_notation() { + let content = text_of(Kind::Color, "rgba(255, 136, 0, 0.5)"); + assert_eq!(rendered(Form::ColorHex, &content), "#FF880080"); + assert_eq!(rendered(Form::ColorRgb, &content), "rgba(255, 136, 0, 0.5)"); + assert_eq!( + rendered(Form::ColorHsl, &content), + "hsla(32, 100%, 50%, 0.5)" + ); + assert_eq!( + rendered(Form::ColorRgb, &text_of(Kind::Color, "#FF880080")), + "rgba(255, 136, 0, 0.5)" + ); + } + + #[test] + fn grey_has_no_hue_and_the_extremes_do_not_divide_by_zero() { + assert_eq!( + rendered(Form::ColorHsl, &text_of(Kind::Color, "#808080")), + "hsl(0, 0%, 50%)" + ); + assert_eq!( + rendered(Form::ColorHsl, &text_of(Kind::Color, "#000000")), + "hsl(0, 0%, 0%)" + ); + assert_eq!( + rendered(Form::ColorHsl, &text_of(Kind::Color, "#FFFFFF")), + "hsl(0, 0%, 100%)" + ); + assert_eq!( + rendered( + Form::ColorRgb, + &text_of(Kind::Color, "hsl(400, 150%, -10%)") + ), + "rgb(0, 0, 0)", + "fuera de rango se recorta, no se rompe" + ); + } + + #[test] + fn every_sextant_of_the_hue_wheel_round_trips_against_a_reference_table() { + for (hex, hsl) in [ + ("#FF0000", "hsl(0, 100%, 50%)"), + ("#FFFF00", "hsl(60, 100%, 50%)"), + ("#00FF00", "hsl(120, 100%, 50%)"), + ("#00FFFF", "hsl(180, 100%, 50%)"), + ("#0000FF", "hsl(240, 100%, 50%)"), + ("#FF00FF", "hsl(300, 100%, 50%)"), + ("#FF0080", "hsl(330, 100%, 50%)"), + ("#80FF00", "hsl(90, 100%, 50%)"), + ("#00FF80", "hsl(150, 100%, 50%)"), + ("#0080FF", "hsl(210, 100%, 50%)"), + ("#8000FF", "hsl(270, 100%, 50%)"), + ("#3498DB", "hsl(204, 70%, 53%)"), + ("#2ECC71", "hsl(145, 63%, 49%)"), + ("#9B59B6", "hsl(283, 39%, 53%)"), + ("#E67E22", "hsl(28, 80%, 52%)"), + ("#1A0B2E", "hsl(266, 61%, 11%)"), + ("#F0E68C", "hsl(54, 77%, 75%)"), + ] { + assert_eq!( + rendered(Form::ColorHsl, &text_of(Kind::Color, hex)), + hsl, + "{hex}" + ); + } + for (hsl, hex) in [ + ("hsl(0, 100%, 50%)", "#FF0000"), + ("hsl(60, 100%, 50%)", "#FFFF00"), + ("hsl(120, 100%, 50%)", "#00FF00"), + ("hsl(180, 100%, 50%)", "#00FFFF"), + ("hsl(240, 100%, 50%)", "#0000FF"), + ("hsl(300, 100%, 50%)", "#FF00FF"), + ("hsl(330, 100%, 50%)", "#FF0080"), + ("hsl(32, 50%, 50%)", "#BF8440"), + ("hsl(210, 40%, 70%)", "#94B2D1"), + ("hsl(90, 60%, 30%)", "#4D7A1F"), + ("hsl(270, 25%, 80%)", "#CCBFD9"), + ("hsl(15, 80%, 20%)", "#5C1F0A"), + ("hsl(400, 50%, 50%)", "#BF9540"), + ("hsl(-60, 100%, 50%)", "#FF00FF"), + ] { + assert_eq!( + rendered(Form::ColorHex, &text_of(Kind::Color, hsl)), + hex, + "{hsl}" + ); + } + } + + #[test] + fn short_and_eight_digit_hex_expand_digit_by_digit() { + assert_eq!( + rendered(Form::ColorRgb, &text_of(Kind::Color, "#0af")), + "rgb(0, 170, 255)" + ); + assert_eq!( + rendered(Form::ColorRgb, &text_of(Kind::Color, "#FF880099")), + "rgba(255, 136, 0, 0.6)" + ); + assert_eq!( + rendered(Form::ColorRgb, &text_of(Kind::Color, "#12345678")), + "rgba(18, 52, 86, 0.47)" + ); + assert_eq!( + rendered( + Form::ColorHex, + &text_of(Kind::Color, "hsla(200, 50%, 40%, 0.25)") + ), + "#33779940" + ); + } + + #[test] + fn a_colour_that_does_not_parse_offers_nothing_extra() { + assert_eq!( + forms_for(&text_of(Kind::Color, "#GGGGGG")), + vec![Form::AsIs] + ); + assert_eq!(parse_color("rgb(1, 2)"), None); + assert_eq!(parse_color("hsl(1, 2, 3, 4, 5)"), None); + assert_eq!(parse_color("#12345"), None); + } + + #[test] + fn a_link_is_offered_as_markdown_domain_and_with_its_title() { + let bare = text_of(Kind::Link, "https://www.ejemplo.test/ruta?x=1#f"); + assert_eq!( + forms_for(&bare), + vec![Form::AsIs, Form::LinkMarkdown, Form::LinkDomain] + ); + assert_eq!( + rendered(Form::LinkMarkdown, &bare), + "[https://www.ejemplo.test/ruta?x=1#f](https://www.ejemplo.test/ruta?x=1#f)" + ); + assert_eq!(rendered(Form::LinkDomain, &bare), "ejemplo.test"); + let titled = Content { + title: Some(" Ejemplo, la página "), + ..bare.clone() + }; + assert_eq!(forms_for(&titled).last(), Some(&Form::LinkTitled)); + assert_eq!( + rendered(Form::LinkMarkdown, &titled), + "[Ejemplo, la página](https://www.ejemplo.test/ruta?x=1#f)" + ); + assert_eq!( + rendered(Form::LinkTitled, &titled), + "Ejemplo, la página — https://www.ejemplo.test/ruta?x=1#f" + ); + } + + #[test] + fn the_domain_drops_credentials_port_and_path_but_not_a_subdomain() { + assert_eq!( + domain_of("https://user:pw@api.ejemplo.test:8443/v1"), + Some("api.ejemplo.test") + ); + assert_eq!( + domain_of("ftp://files.ejemplo.test"), + Some("files.ejemplo.test") + ); + assert_eq!( + domain_of("mailto:alguien@ejemplo.test"), + Some("ejemplo.test") + ); + assert_eq!( + domain_of("https://localhost:3000/"), + None, + "sin punto no es dominio" + ); + assert_eq!(domain_of("https://"), None); + assert_eq!(domain_of(""), None); + assert_eq!( + domain_of("192.168.0.1:8080"), + Some("192.168.0.1"), + "sin esquema, lo de antes del puerto es el host" + ); + assert_eq!( + domain_of("https://ejemplo.test:abc/"), + Some("ejemplo.test:abc"), + "un puerto que no es número no se recorta" + ); + assert_eq!(domain_of("https://ejemplo.test:/"), Some("ejemplo.test:")); + assert_eq!( + domain_of(":ejemplo.test"), + Some(":ejemplo.test"), + "un esquema vacío no es esquema" + ); + } + + #[test] + fn code_can_be_one_line_a_block_or_dedented() { + let source = " fn main() {\n println!(\"hola\");\n }\n"; + let content = text_of(Kind::Code, source); + assert_eq!( + forms_for(&content), + vec![ + Form::AsIs, + Form::CodeOneLine, + Form::CodeBlock, + Form::CodeDedented + ] + ); + assert_eq!( + rendered(Form::CodeOneLine, &content), + "fn main() { println!(\"hola\"); }" + ); + assert_eq!( + rendered(Form::CodeBlock, &content), + "```\nfn main() {\n println!(\"hola\");\n }\n```" + ); + assert_eq!( + rendered(Form::CodeDedented, &content), + "fn main() {\n println!(\"hola\");\n}\n", + "se quita lo que todas las líneas comparten y nada más" + ); + assert_eq!( + dedent("\n a\n\n b\n"), + "\na\n\n b\n", + "las líneas vacías no cuentan" + ); + } + + #[test] + fn a_token_becomes_a_header_or_a_curl_and_a_jwt_shows_its_claims() { + let opaque = text_of(Kind::Token, "ghp_not_a_real_token_for_tests_0000000000"); + assert_eq!( + forms_for(&opaque), + vec![Form::AsIs, Form::TokenHeader, Form::TokenCurl] + ); + assert_eq!( + rendered(Form::TokenHeader, &opaque), + "Authorization: Bearer ghp_not_a_real_token_for_tests_0000000000" + ); + assert_eq!( + rendered(Form::TokenCurl, &opaque), + "curl -H 'Authorization: Bearer ghp_not_a_real_token_for_tests_0000000000' \"$URL\"" + ); + let jwt = text_of( + Kind::Token, + "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiJjcC0zIiwiZW52Ijoic3RhZ2luZyJ9.firma", + ); + assert_eq!(forms_for(&jwt).last(), Some(&Form::TokenClaims)); + assert_eq!( + rendered(Form::TokenClaims, &jwt), + "{\n \"sub\": \"cp-3\",\n \"env\": \"staging\"\n}" + ); + } + + fn tiny_png() -> Vec { + let mut out = Vec::new(); + let mut image = image::RgbaImage::new(2, 2); + image.put_pixel(0, 0, image::Rgba([255, 0, 0, 255])); + image.put_pixel(1, 0, image::Rgba([0, 0, 0, 0])); + image.put_pixel(0, 1, image::Rgba([0, 255, 0, 255])); + image.put_pixel(1, 1, image::Rgba([0, 0, 255, 128])); + image + .write_with_encoder(image::codecs::png::PngEncoder::new(&mut out)) + .expect("png"); + out + } + + fn solid_png(pixel: [u8; 4]) -> Vec { + let mut out = Vec::new(); + let image = image::RgbaImage::from_pixel(16, 16, image::Rgba(pixel)); + image + .write_with_encoder(image::codecs::png::PngEncoder::new(&mut out)) + .expect("png"); + out + } + + fn middle_of_jpeg(bytes: &[u8]) -> [u8; 3] { + assert_eq!(&bytes[..2], &[0xFF, 0xD8], "cabecera JPEG"); + let back = image::load_from_memory(bytes).expect("se lee").to_rgb8(); + back.get_pixel(8, 8).0 + } + + #[test] + fn an_image_is_offered_as_jpeg_and_as_what_was_read_in_it() { + let png = tiny_png(); + let silent = Content { + kind: Some(Kind::Image), + png: Some(&png), + ..Default::default() + }; + assert_eq!(forms_for(&silent), vec![Form::AsIs, Form::ImageJpeg]); + assert_eq!( + render(Form::AsIs, &silent), + Some(Rendered::Png(png.clone())) + ); + let read = Content { + ocr: Some(" Pedido 4417 "), + ..silent.clone() + }; + assert_eq!( + forms_for(&read), + vec![Form::AsIs, Form::ImageJpeg, Form::ImageOcr] + ); + assert_eq!(rendered(Form::ImageOcr, &read), "Pedido 4417"); + let blank = Content { + ocr: Some(" "), + ..silent.clone() + }; + assert_eq!(forms_for(&blank), vec![Form::AsIs, Form::ImageJpeg]); + } + + #[test] + fn the_jpeg_is_a_real_jpeg_with_transparency_laid_on_white() { + let near = |got: [u8; 3], wanted: [u8; 3]| { + got.iter() + .zip(wanted) + .all(|(got, wanted)| got.abs_diff(wanted) <= 6) + }; + let jpeg_of_solid = |pixel: [u8; 4]| { + let png = solid_png(pixel); + let content = Content { + png: Some(&png), + ..Default::default() + }; + match render(Form::ImageJpeg, &content) { + Some(Rendered::Jpeg(bytes)) => middle_of_jpeg(&bytes), + other => panic!("no salió un JPEG: {other:?}"), + } + }; + assert!( + near(jpeg_of_solid([200, 30, 30, 255]), [200, 30, 30]), + "opaco intacto" + ); + assert!( + near(jpeg_of_solid([0, 0, 0, 0]), [255, 255, 255]), + "lo transparente se vuelve blanco, no negro" + ); + assert!( + near(jpeg_of_solid([0, 0, 255, 128]), [127, 127, 255]), + "azul al 50 % sobre blanco" + ); + assert!( + near(jpeg_of_solid([0, 0, 255, 64]), [191, 191, 255]), + "azul al 25 % sobre blanco" + ); + assert!( + near(jpeg_of_solid([200, 30, 30, 128]), [227, 142, 142]), + "un rojo apagado al 50 % se aclara canal a canal" + ); + assert_eq!( + render(Form::ImageJpeg, &text_of(Kind::Image, "no bytes")), + None + ); + let broken = Content { + png: Some(b"no es png"), + ..Default::default() + }; + assert_eq!(render(Form::ImageJpeg, &broken), None); + } + + #[test] + fn files_are_offered_by_their_paths() { + let content = Content { + kind: Some(Kind::File), + paths: vec!["/tmp/uno.txt".into(), "/tmp/dos.txt".into()], + ..Default::default() + }; + assert_eq!(forms_for(&content), vec![Form::AsIs, Form::Path]); + assert_eq!(rendered(Form::Path, &content), "/tmp/uno.txt\n/tmp/dos.txt"); + } + + #[test] + fn a_form_that_does_not_apply_renders_nothing() { + let plain = text_of(Kind::Text, "hola"); + for form in [ + Form::Markdown, + Form::JsonPretty, + Form::ColorHex, + Form::LinkDomain, + Form::TokenClaims, + Form::ImageJpeg, + Form::ImageOcr, + ] { + assert_eq!(render(form, &plain), None, "{form:?}"); + } + } + + #[test] + fn html_from_a_word_processor_becomes_readable_markdown() { + let html = "x\ +

Informe

\ +

Un párrafo con negrita, cursiva y un \ + enlace.

\ +
  • uno
  • dos fuerte
\ +
  1. primero
  2. segundo
\ +

Fin & código x < y

\ + "; + assert_eq!( + markdown_of_html(html), + "## Informe\n\n\ + Un párrafo con **negrita**, *cursiva* y un [enlace](https://ejemplo.test).\n\n\ + - uno\n- dos **fuerte**\n\n\ + 1. primero\n2. segundo\n\n\ + Fin & código `x < y`" + ); + } + + #[test] + fn the_windows_clipboard_header_is_left_out_and_only_the_fragment_counts() { + let html = "Version:0.9\r\nStartHTML:0000000105\r\nEndHTML:0000000250\r\n\ +

solo esto

"; + assert_eq!(markdown_of_html(html), "solo **esto**"); + } + + #[test] + fn preformatted_blocks_keep_their_whitespace_and_nested_lists_indent() { + let html = "
fn main() {\n    hola\n}
  • a
    • b
"; + assert_eq!( + markdown_of_html(html), + "```\nfn main() {\n hola\n}\n```\n\n- a\n - b" + ); + } + + #[test] + fn quotes_images_rules_and_line_breaks() { + let html = "
dicho
\"foto\"
uno
dos"; + assert_eq!( + markdown_of_html(html), + "> dicho\n\n![foto](a.png)\n\n---\n\nuno\ndos" + ); + } + + #[test] + fn whitespace_between_tags_collapses_like_a_browser_would() { + let html = "

\n varias\n palabras juntas \n

\n

otro

"; + assert_eq!(markdown_of_html(html), "varias palabras **juntas**\n\notro"); + } + + #[test] + fn what_is_not_html_at_all_comes_out_as_text() { + assert_eq!(markdown_of_html("2 < 3 y 4 > 1"), "2 < 3 y 4 > 1"); + assert_eq!(markdown_of_html(""), ""); + assert_eq!(markdown_of_html("<"), "<"); + assert_eq!(markdown_of_html("

"), ""); + } + + #[test] + fn entities_of_every_shape_are_decoded_and_the_rest_left_alone() { + assert_eq!( + decode_entities("&<>"' "), + "&<>\"' " + ); + assert_eq!(decode_entities("ABC"), "ABC"); + assert_eq!( + decode_entities("&desconocida; & suelto &;"), + "&desconocida; & suelto &;" + ); + assert_eq!( + decode_entities("�"), + "�", + "fuera de Unicode" + ); + assert_eq!( + decode_entities("á"), + "á", + "las nombradas raras se dejan" + ); + } + + #[test] + fn a_script_or_style_is_skipped_until_its_own_closing_tag() { + assert_eq!(markdown_of_html("c"), "c"); + assert_eq!(markdown_of_html("z"), "z"); + assert_eq!( + markdown_of_html("tcuerpo"), + "cuerpo" + ); + } + + #[test] + fn headings_tables_and_lists_keep_their_distance_from_what_follows() { + assert_eq!(markdown_of_html("

t

x"), "# t\n\nx"); + assert_eq!( + markdown_of_html("
ab
"), + "a\tb" + ); + assert_eq!(markdown_of_html("a
  • b
"), "a\n\n- b"); + assert_eq!(markdown_of_html("
x
"), "```\nx\n```"); + assert_eq!(markdown_of_html("a b"), "**a** b"); + assert_eq!(markdown_of_html("vacío"), "vacío"); + assert_eq!(markdown_of_html("

x

"), "x"); + assert_eq!(markdown_of_html("
q
"), "> q"); + assert_eq!( + markdown_of_html("
x"), + "x", + "un salto al principio no deja hueco" + ); + assert_eq!(markdown_of_html("
  • a
"), "- a"); + assert_eq!(markdown_of_html("
  1. a
"), "1. a"); + assert_eq!(markdown_of_html("uno

dos"), "uno\ndos"); + assert_eq!(markdown_of_html(" x"), "![](a.png) x"); + assert_eq!(markdown_of_html("
a\n
b"), "```\na\n```\n\nb"); + } + + #[test] + fn a_link_with_no_target_is_just_its_text() { + assert_eq!(markdown_of_html("sin destino"), "sin destino"); + assert_eq!( + markdown_of_html("comillas simples"), + "[comillas simples](x.html)" + ); + assert_eq!( + markdown_of_html("sin comillas"), + "[sin comillas](x.html)" + ); + } +} diff --git a/crates/cp-core/src/search.rs b/crates/cp-core/src/search.rs index 2920d32..39a9fce 100644 --- a/crates/cp-core/src/search.rs +++ b/crates/cp-core/src/search.rs @@ -1,35 +1,390 @@ use unicode_normalization::UnicodeNormalization; pub fn fold(input: &str) -> String { - input - .nfkd() - .filter(|c| !is_combining_mark(*c)) - .flat_map(expand_ligature) - .flat_map(char::to_lowercase) + let mut out = String::with_capacity(input.len()); + fold_each(input, |c, _| out.push(c)); + out +} + +pub struct Folded { + pub chars: Vec, + pub origin: Vec, +} + +pub fn fold_mapped(input: &str) -> Folded { + let mut chars = Vec::with_capacity(input.len()); + let mut origin = Vec::with_capacity(input.len()); + fold_each(input, |c, at| { + chars.push(c); + origin.push(at); + }); + Folded { chars, origin } +} + +fn fold_each(input: &str, mut push: impl FnMut(char, usize)) { + for (at, c) in input.char_indices() { + if c.is_ascii() { + push(c.to_ascii_lowercase(), at); + continue; + } + for decomposed in std::iter::once(c).nfkd() { + if is_combining_mark(decomposed) { + continue; + } + match expand_ligature(decomposed) { + Some(expanded) => expanded.chars().for_each(|c| push(c, at)), + None => decomposed.to_lowercase().for_each(|c| push(c, at)), + } + } + } +} + +pub fn terms_of(query: &str) -> Vec { + fold(query) + .split_whitespace() + .filter(|word| word.chars().any(char::is_alphanumeric)) + .map(str::to_owned) .collect() } +pub const EXCERPT_CHARS: usize = 160; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Segment { + pub text: String, + pub matched: bool, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Excerpt { + pub segments: Vec, +} + +impl Excerpt { + pub fn plain(&self) -> String { + self.segments.iter().map(|one| one.text.as_str()).collect() + } +} + +pub fn excerpt(text: &str, terms: &[String], width: usize) -> Option { + let folded = fold_mapped(text); + let tokens = tokens_of(&folded.chars); + let mut hits: Vec<(usize, usize)> = terms + .iter() + .flat_map(|term| matches_of(&folded.chars, &tokens, term)) + .map(|(from, to)| { + let last = folded.origin[to - 1]; + (folded.origin[from], last + char_at(text, last).len_utf8()) + }) + .collect(); + hits.sort_unstable(); + let hits = merged(hits); + let first = *hits.first()?; + + let start = text[..first.0] + .char_indices() + .rev() + .nth(width / 3) + .map_or(0, |(at, c)| at + c.len_utf8()); + let end = text[start..] + .char_indices() + .nth(width.max(1)) + .map_or(text.len(), |(at, _)| start + at); + + let mut segments = Vec::new(); + if start > 0 { + segments.push(ellipsis()); + } + let mut cursor = start; + for (from, to) in hits { + if from >= end { + break; + } + let (from, to) = (from.max(start), to.min(end)); + segments.extend(segment(text, cursor, from, false)); + segments.extend(segment(text, from, to, true)); + cursor = to; + } + segments.extend(segment(text, cursor, end, false)); + if end < text.len() { + segments.push(ellipsis()); + } + Some(Excerpt { segments }) +} + +fn char_at(text: &str, at: usize) -> char { + text[at..] + .chars() + .next() + .expect("un origen apunta a un carácter") +} + +fn segment(text: &str, from: usize, to: usize, matched: bool) -> Option { + (from < to).then(|| Segment { + text: text[from..to].into(), + matched, + }) +} + +fn ellipsis() -> Segment { + Segment { + text: "…".into(), + matched: false, + } +} + +fn tokens_of(chars: &[char]) -> Vec<(usize, usize)> { + let mut tokens = Vec::new(); + let mut open: Option = None; + for (at, c) in chars.iter().enumerate() { + match (c.is_alphanumeric(), open) { + (true, None) => open = Some(at), + (false, Some(from)) => { + tokens.push((from, at)); + open = None; + } + _ => {} + } + } + if let Some(from) = open { + tokens.push((from, chars.len())); + } + tokens +} + +fn matches_of(chars: &[char], tokens: &[(usize, usize)], term: &str) -> Vec<(usize, usize)> { + let wanted: Vec> = term + .split(|c: char| !c.is_alphanumeric()) + .filter(|part| !part.is_empty()) + .map(|part| part.chars().collect()) + .collect(); + let Some((last, head)) = wanted.split_last() else { + return Vec::new(); + }; + let mut found = Vec::new(); + for window in tokens.windows(wanted.len()) { + let whole = head + .iter() + .zip(window) + .all(|(part, (from, to))| chars[*from..*to] == part[..]); + let (from, to) = window[wanted.len() - 1]; + if whole && chars[from..to].starts_with(last) { + found.push((window[0].0, from + last.len())); + } + } + found +} + +fn merged(sorted: Vec<(usize, usize)>) -> Vec<(usize, usize)> { + let mut out: Vec<(usize, usize)> = Vec::new(); + for (from, to) in sorted { + match out.last_mut() { + Some(last) if from <= last.1 => last.1 = last.1.max(to), + _ => out.push((from, to)), + } + } + out +} + fn is_combining_mark(c: char) -> bool { matches!(c as u32, 0x0300..=0x036F | 0x1AB0..=0x1AFF | 0x20D0..=0x20FF) } -fn expand_ligature(c: char) -> std::vec::IntoIter { - let expanded: Vec = match c { - 'ß' => vec!['s', 's'], - 'æ' | 'Æ' => vec!['a', 'e'], - 'œ' | 'Œ' => vec!['o', 'e'], - 'ø' | 'Ø' => vec!['o'], - 'ł' | 'Ł' => vec!['l'], - 'đ' | 'Đ' => vec!['d'], - 'þ' | 'Þ' => vec!['t', 'h'], - other => vec![other], - }; - expanded.into_iter() +fn expand_ligature(c: char) -> Option<&'static str> { + match c { + 'ß' => Some("ss"), + 'æ' | 'Æ' => Some("ae"), + 'œ' | 'Œ' => Some("oe"), + 'ø' | 'Ø' => Some("o"), + 'ł' | 'Ł' => Some("l"), + 'đ' | 'Đ' => Some("d"), + 'þ' | 'Þ' => Some("th"), + _ => None, + } } #[cfg(test)] mod tests { - use super::fold; + use super::{EXCERPT_CHARS, Segment, excerpt, fold, fold_mapped, terms_of}; + + fn plain(text: &str) -> Segment { + Segment { + text: text.into(), + matched: false, + } + } + + fn hit(text: &str) -> Segment { + Segment { + text: text.into(), + matched: true, + } + } + + fn terms(words: &[&str]) -> Vec { + words.iter().map(|word| (*word).to_owned()).collect() + } + + #[test] + fn each_folded_char_remembers_where_it_came_from() { + let folded = fold_mapped("Straße"); + assert_eq!(folded.chars.iter().collect::(), "strasse"); + assert_eq!( + folded.origin, + vec![0, 1, 2, 3, 4, 4, 6], + "la ß se abre en dos y ocupa dos bytes: el origen es el byte" + ); + let folded = fold_mapped("café"); + assert_eq!(folded.origin, vec![0, 1, 2, 3]); + let folded = fold_mapped("ñu"); + assert_eq!(folded.origin, vec![0, 2]); + } + + #[test] + fn the_excerpt_shows_the_text_as_copied_with_the_match_marked() { + let found = excerpt("el café de la esquina", &terms(&["cafe"]), EXCERPT_CHARS) + .expect("hay coincidencia"); + assert_eq!( + found.segments, + vec![plain("el "), hit("café"), plain(" de la esquina")], + "se busca sin tilde y se enseña con ella" + ); + } + + #[test] + fn a_prefix_marks_only_what_was_typed() { + let found = excerpt("la esquina", &terms(&["esq"]), 100).expect("hay"); + assert_eq!( + found.segments, + vec![plain("la "), hit("esq"), plain("uina")] + ); + } + + #[test] + fn a_word_that_only_appears_inside_another_is_not_a_match() { + assert!( + excerpt("encafetado", &terms(&["cafe"]), 100).is_none(), + "el índice tampoco lo encontraría: los términos van por prefijo de palabra" + ); + } + + #[test] + fn a_term_with_punctuation_matches_the_same_phrase_the_index_does() { + let found = + excerpt("Pedido AB-4417 entrega 12 marzo", &terms(&["ab-4417"]), 100).expect("hay"); + assert_eq!( + found.segments, + vec![plain("Pedido "), hit("AB-4417"), plain(" entrega 12 marzo")] + ); + } + + #[test] + fn an_expanded_letter_is_marked_whole() { + let found = excerpt("Straße Hauptbahnhof", &terms(&["strasse"]), 100).expect("hay"); + assert_eq!(found.segments, vec![hit("Straße"), plain(" Hauptbahnhof")]); + } + + #[test] + fn overlapping_terms_become_one_mark() { + let found = excerpt("un café", &terms(&["caf", "cafe"]), 100).expect("hay"); + assert_eq!(found.segments, vec![plain("un "), hit("café")]); + } + + #[test] + fn every_term_that_matches_is_marked() { + let found = excerpt("rojo verde azul", &terms(&["rojo", "azul"]), 100).expect("hay"); + assert_eq!( + found.segments, + vec![hit("rojo"), plain(" verde "), hit("azul")] + ); + } + + #[test] + fn a_long_text_is_cut_around_the_first_match() { + let text = format!("{}aguja{}", "paja ".repeat(100), " heno".repeat(100)); + let found = excerpt(&text, &terms(&["aguja"]), 60).expect("hay"); + let shown = found.plain(); + assert!(shown.starts_with('…') && shown.ends_with('…'), "{shown}"); + assert_eq!( + shown.chars().count(), + 62, + "sesenta más los dos puntos suspensivos" + ); + assert!( + found + .segments + .iter() + .any(|one| one.matched && one.text == "aguja") + ); + let before = shown.find("aguja").expect("está"); + assert!( + before < shown.len() / 2, + "la coincidencia queda en el primer tercio, no al final de la ventana" + ); + } + + #[test] + fn exactly_a_third_of_the_window_comes_before_the_match() { + let found = excerpt("abcdefghij aguja", &terms(&["aguja"]), 9).expect("hay"); + assert_eq!( + found.plain(), + "…ij aguja", + "tres caracteres delante, ni uno más" + ); + let found = excerpt("ñññññ aguja", &terms(&["aguja"]), 9).expect("hay"); + assert_eq!( + found.plain(), + "…ññ aguja", + "y el corte cae en un límite de carácter aunque el anterior ocupe dos bytes" + ); + } + + #[test] + fn a_match_at_the_very_end_only_needs_the_leading_ellipsis() { + let text = format!("{}fin", "x ".repeat(200)); + let found = excerpt(&text, &terms(&["fin"]), 40).expect("hay"); + let shown = found.plain(); + assert!(shown.starts_with('…')); + assert!(shown.ends_with("fin"), "{shown}"); + } + + #[test] + fn a_match_at_the_start_only_needs_the_trailing_ellipsis() { + let text = format!("inicio{}", " x".repeat(200)); + let found = excerpt(&text, &terms(&["inicio"]), 40).expect("hay"); + let shown = found.plain(); + assert!(shown.starts_with("inicio"), "{shown}"); + assert!(shown.ends_with('…')); + } + + #[test] + fn a_short_text_comes_back_whole() { + let found = excerpt("nota corta", &terms(&["nota"]), EXCERPT_CHARS).expect("hay"); + assert_eq!(found.plain(), "nota corta"); + } + + #[test] + fn a_second_match_past_the_window_does_not_stretch_it() { + let text = format!("aguja {}aguja", "paja ".repeat(100)); + let found = excerpt(&text, &terms(&["aguja"]), 30).expect("hay"); + assert_eq!(found.segments.iter().filter(|one| one.matched).count(), 1); + assert!(found.plain().chars().count() <= 31); + } + + #[test] + fn nothing_to_look_for_is_nothing_found() { + assert!(excerpt("algo", &[], 100).is_none()); + assert!(excerpt("", &terms(&["algo"]), 100).is_none()); + assert!(excerpt("otra cosa", &terms(&["algo"]), 100).is_none()); + assert!(excerpt("otra cosa", &terms(&["!!!"]), 100).is_none()); + } + + #[test] + fn the_terms_are_the_words_the_index_would_look_for() { + assert_eq!(terms_of("Café AB-4417 ... !!"), vec!["cafe", "ab-4417"]); + assert!(terms_of(" ").is_empty()); + assert!(terms_of("!!! ...").is_empty()); + } #[test] fn every_ligature_in_the_table_is_expanded() { diff --git a/crates/cp-core/src/token.rs b/crates/cp-core/src/token.rs new file mode 100644 index 0000000..758a781 --- /dev/null +++ b/crates/cp-core/src/token.rs @@ -0,0 +1,267 @@ +const PREFIXES: &[(&str, usize)] = &[ + ("sk-ant-", 40), + ("sk_live_", 24), + ("sk_test_", 24), + ("rk_live_", 24), + ("sk-", 32), + ("ghp_", 36), + ("gho_", 36), + ("ghu_", 36), + ("ghs_", 36), + ("ghr_", 36), + ("github_pat_", 40), + ("glpat-", 20), + ("xoxb-", 20), + ("xoxp-", 20), + ("xoxa-", 20), + ("xapp-", 20), + ("AKIA", 20), + ("AIza", 39), + ("ya29.", 40), + ("npm_", 36), + ("pypi-", 40), + ("shpat_", 38), + ("SG.", 60), +]; + +pub fn looks_like(text: &str) -> bool { + let text = text.trim(); + if text.is_empty() || text.contains(char::is_whitespace) { + return false; + } + if claims_of(text).is_some() { + return true; + } + PREFIXES.iter().any(|(prefix, at_least)| { + text.starts_with(prefix) + && text.len() >= *at_least + && text[prefix.len()..] + .chars() + .all(|c| c.is_ascii_alphanumeric() || "-_.".contains(c)) + }) +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Claims { + pub header: serde_json::Map, + pub payload: serde_json::Map, +} + +impl Claims { + pub fn expires_at(&self) -> Option { + self.payload.get("exp").and_then(serde_json::Value::as_i64) + } + + pub fn expired_by(&self, now_seconds: i64) -> Option { + self.expires_at().map(|exp| exp <= now_seconds) + } + + pub fn subject(&self) -> Option<&str> { + self.payload.get("sub").and_then(serde_json::Value::as_str) + } + + pub fn issuer(&self) -> Option<&str> { + self.payload.get("iss").and_then(serde_json::Value::as_str) + } +} + +pub fn claims_of(text: &str) -> Option { + let mut parts = text.trim().split('.'); + let (header, payload, signature) = (parts.next()?, parts.next()?, parts.next()?); + if parts.next().is_some() || signature.is_empty() { + return None; + } + let header = object_of(header)?; + if !header.contains_key("alg") { + return None; + } + Some(Claims { + header, + payload: object_of(payload)?, + }) +} + +fn object_of(segment: &str) -> Option> { + let bytes = base64url(segment)?; + match serde_json::from_slice(&bytes).ok()? { + serde_json::Value::Object(map) => Some(map), + _ => None, + } +} + +pub fn base64url(segment: &str) -> Option> { + let mut out = Vec::with_capacity(segment.len() * 3 / 4); + let mut buffer: u32 = 0; + let mut bits = 0; + for c in segment.bytes() { + let value = match c { + b'A'..=b'Z' => c - b'A', + b'a'..=b'z' => c - b'a' + 26, + b'0'..=b'9' => c - b'0' + 52, + b'-' | b'+' => 62, + b'_' | b'/' => 63, + b'=' => break, + _ => return None, + }; + buffer = (buffer << 6) | u32::from(value); + bits += 6; + if bits >= 8 { + bits -= 8; + out.push((buffer >> bits) as u8); + buffer &= (1 << bits) - 1; + } + } + Some(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn jwt(payload: &str) -> String { + let encode = |raw: &str| { + const TABLE: &[u8] = + b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_"; + let bytes = raw.as_bytes(); + let mut out = String::new(); + for chunk in bytes.chunks(3) { + let mut buffer = 0u32; + for (i, b) in chunk.iter().enumerate() { + buffer |= u32::from(*b) << (16 - 8 * i); + } + for i in 0..=chunk.len() { + let index = (buffer >> (18 - 6 * i)) & 63; + out.push(TABLE[index as usize] as char); + } + } + out + }; + format!( + "{}.{}.firma", + encode(r#"{"alg":"HS256","typ":"JWT"}"#), + encode(payload) + ) + } + + #[test] + fn a_jwt_is_a_token_and_says_what_it_carries() { + let token = jwt(r#"{"sub":"cp-3","env":"staging","exp":1700000000,"iss":"cp"}"#); + assert!(looks_like(&token)); + let claims = claims_of(&token).expect("se lee"); + assert_eq!(claims.subject(), Some("cp-3")); + assert_eq!(claims.issuer(), Some("cp")); + assert_eq!(claims.expires_at(), Some(1_700_000_000)); + assert_eq!(claims.expired_by(1_700_000_001), Some(true)); + assert_eq!(claims.expired_by(1_600_000_000), Some(false)); + assert_eq!( + claims.payload.get("env").and_then(|v| v.as_str()), + Some("staging"), + "el problema real no es que se vea: es saber cuál de los doce es el de staging" + ); + } + + #[test] + fn a_jwt_without_expiry_does_not_pretend_to_know() { + let claims = claims_of(&jwt(r#"{"sub":"x"}"#)).expect("se lee"); + assert_eq!(claims.expired_by(0), None); + } + + #[test] + fn three_dotted_words_are_not_a_jwt() { + assert!( + claims_of("uno.dos.tres").is_none(), + "no es base64 de un JSON" + ); + assert!(claims_of("a.b").is_none(), "faltan partes"); + assert!( + claims_of(&format!("{}.extra", jwt("{}"))).is_none(), + "sobran" + ); + assert!(!looks_like("uno.dos.tres")); + assert!(!looks_like("archivo.tar.gz")); + } + + #[test] + fn a_header_without_alg_is_just_json_in_base64() { + let no_alg = format!("{}.firma", { + let t = jwt("{}"); + t.replace("eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9", "e30") + }); + assert!(claims_of(&no_alg).is_none()); + } + + #[test] + fn the_known_prefixes_are_tokens_when_long_enough() { + for token in [ + "sk-not-a-real-key-for-tests-0123456789", + "sk-ant-not-a-real-key-for-tests-0000000000000000000", + "ghp_not_a_real_token_for_tests_0000000000", + "github_pat_not_a_real_token_for_tests_0000000000", + "xoxb-not-a-real-slack-token-for-tests", + "AKIAIOSFODNN7EXAMPLE", + "AIzaNotARealGoogleKeyForTests00000000000000", + "glpat-not-a-real-token-for-tests", + ] { + assert!(looks_like(token), "{token}"); + } + } + + #[test] + fn a_prefix_alone_or_with_spaces_is_not_a_token() { + assert!(!looks_like("sk-"), "vacío detrás"); + assert!(!looks_like("sk-corto"), "demasiado corto"); + assert!(!looks_like( + "ghp_ tiene espacios dentro 0123456789012345678901" + )); + assert!(!looks_like("AKIA con espacios y todo")); + assert!(!looks_like( + "sk-not-a-real-key-for-tests-0123456789 y luego" + )); + assert!(!looks_like("")); + assert!( + !looks_like(&format!("{} con espacio", jwt("{}"))), + "un JWT seguido de palabras no es un token" + ); + assert!( + !looks_like("skeleton-key-of-the-castle-0123456789"), + "sk- exacto" + ); + } + + #[test] + fn what_a_person_copies_every_day_is_not_a_token() { + for text in [ + "https://ejemplo.test/ruta", + "alguien@ejemplo.test", + "7ab3f6de-1c4b-4f5e-8a2d-9f0e1b2c3d4e", + "d41d8cd98f00b204e9800998ecf8427e", + "ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789abcdefghijklmnop", + "AKIA", + "skater", + ] { + assert!(!looks_like(text), "{text}"); + } + } + + #[test] + fn base64url_decodes_with_or_without_padding() { + assert_eq!(base64url("aGVsbG8").as_deref(), Some(&b"hello"[..])); + assert_eq!(base64url("aGVsbG8=").as_deref(), Some(&b"hello"[..])); + assert_eq!(base64url("aGk").as_deref(), Some(&b"hi"[..])); + assert_eq!( + base64url("_w").as_deref(), + Some(&[0xff][..]), + "alfabeto url" + ); + assert_eq!( + base64url("/w").as_deref(), + Some(&[0xff][..]), + "y el clásico" + ); + assert_eq!(base64url("-w").as_deref(), Some(&[0xfb][..])); + assert_eq!(base64url("+w").as_deref(), Some(&[0xfb][..])); + assert_eq!(base64url(""), Some(Vec::new())); + assert_eq!(base64url("a b"), None); + assert_eq!(base64url("ñ"), None); + } +} diff --git a/crates/cp-core/src/watch.rs b/crates/cp-core/src/watch.rs index 8a1d8f3..e1485e3 100644 --- a/crates/cp-core/src/watch.rs +++ b/crates/cp-core/src/watch.rs @@ -11,6 +11,47 @@ pub enum Cadence { Opaque, } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Retry { + pub attempts: u8, + pub pause: std::time::Duration, +} + +pub const RETRY: Retry = Retry { + attempts: 5, + pause: std::time::Duration::from_millis(150), +}; + +const _: () = assert!(RETRY.attempts >= 2); + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Retried { + Done(T), + Superseded, + Exhausted, +} + +pub fn insist( + retry: Retry, + count: impl Fn() -> i64, + mut attempt: impl FnMut() -> Option, + pause: impl Fn(std::time::Duration), +) -> Retried { + let started_at = count(); + for done in 0..retry.attempts.max(1) { + if done > 0 { + pause(retry.pause); + if count() != started_at { + return Retried::Superseded; + } + } + if let Some(got) = attempt() { + return Retried::Done(got); + } + } + Retried::Exhausted +} + #[derive(Debug)] pub struct Watcher { last: Option, @@ -69,6 +110,125 @@ impl Watcher { } } +#[cfg(test)] +mod insisting { + use super::*; + use std::cell::Cell; + use std::time::Duration; + + fn quick() -> Retry { + Retry { + attempts: 4, + pause: Duration::from_millis(7), + } + } + + #[test] + fn a_source_that_answers_the_first_time_is_not_asked_twice() { + let asked = Cell::new(0); + let paused = Cell::new(0); + let got = insist( + quick(), + || 1, + || { + asked.set(asked.get() + 1); + Some("hola") + }, + |_| paused.set(paused.get() + 1), + ); + assert_eq!(got, Retried::Done("hola")); + assert_eq!((asked.get(), paused.get()), (1, 0)); + } + + #[test] + fn a_source_that_takes_a_while_is_asked_again_after_a_pause() { + let asked = Cell::new(0); + let pauses = std::cell::RefCell::new(Vec::new()); + let got = insist( + quick(), + || 1, + || { + asked.set(asked.get() + 1); + (asked.get() == 3).then_some("al fin") + }, + |how_long| pauses.borrow_mut().push(how_long), + ); + assert_eq!(got, Retried::Done("al fin")); + assert_eq!(asked.get(), 3); + assert_eq!( + *pauses.borrow(), + vec![Duration::from_millis(7); 2], + "una pausa entre cada dos intentos, ninguna antes del primero" + ); + } + + #[test] + fn a_source_that_never_answers_is_given_up_on_after_the_attempts() { + let asked = Cell::new(0); + let got: Retried<()> = insist( + quick(), + || 1, + || { + asked.set(asked.get() + 1); + None + }, + |_| {}, + ); + assert_eq!(got, Retried::Exhausted); + assert_eq!(asked.get(), 4, "exactamente los intentos de la política"); + } + + #[test] + fn a_newer_copy_while_waiting_wins_and_the_old_one_is_dropped() { + let asked = Cell::new(0); + let counter = Cell::new(10); + let got: Retried<()> = insist( + quick(), + || counter.get(), + || { + asked.set(asked.get() + 1); + None + }, + |_| counter.set(11), + ); + assert_eq!(got, Retried::Superseded); + assert_eq!( + asked.get(), + 1, + "lo copiado después lo verá el vigilante; no se lee dos veces" + ); + } + + #[test] + fn zero_attempts_still_asks_once() { + let asked = Cell::new(0); + let policy = Retry { + attempts: 0, + pause: Duration::ZERO, + }; + let got = insist( + policy, + || 1, + || { + asked.set(asked.get() + 1); + Some(()) + }, + |_| {}, + ); + assert_eq!(got, Retried::Done(())); + assert_eq!(asked.get(), 1); + } + + #[test] + fn the_shipped_policy_waits_less_than_a_person_notices() { + let worst = RETRY.pause.as_millis() * u128::from(RETRY.attempts - 1); + assert!( + worst <= 1_000, + "{worst} ms de pausas acumuladas es demasiado" + ); + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/cp-mac-sys/src/frontmost.rs b/crates/cp-mac-sys/src/frontmost.rs index d512718..90840af 100644 --- a/crates/cp-mac-sys/src/frontmost.rs +++ b/crates/cp-mac-sys/src/frontmost.rs @@ -16,17 +16,18 @@ pub fn missing_paths(file_urls: &str) -> Vec { file_urls .lines() .filter(|line| !line.trim().is_empty()) - .filter(|url| { - let path = url - .strip_prefix("file://") - .map(percent_decoded) - .unwrap_or_else(|| (*url).to_string()); - !std::path::Path::new(&path).exists() - }) + .filter(|url| !std::path::Path::new(&path_of(url)).exists()) .map(str::to_owned) .collect() } +pub fn path_of(file_url: &str) -> String { + file_url + .strip_prefix("file://") + .map(percent_decoded) + .unwrap_or_else(|| file_url.to_owned()) +} + fn percent_decoded(text: &str) -> String { let bytes = text.as_bytes(); let mut out: Vec = Vec::with_capacity(bytes.len()); diff --git a/crates/cp-mac/examples/probe/battery.rs b/crates/cp-mac/examples/probe/battery.rs index 84c128d..0f7a49b 100644 --- a/crates/cp-mac/examples/probe/battery.rs +++ b/crates/cp-mac/examples/probe/battery.rs @@ -802,6 +802,28 @@ fn main() -> std::process::ExitCode { }, ); + b.case( + "C6", + "insistir ante una fuente que responde no cuesta un segundo intento", + || { + pb.write_text("cp-c6"); + let direct = capture(&pb).ok_or("no se capturó")?; + let started = std::time::Instant::now(); + let got = cp_mac::capture::capture_insisting(PATIENCE, cp_core::watch::RETRY); + let took = started.elapsed(); + match got { + Captured::Kept(item) if item == direct => { + if took < cp_core::watch::RETRY.pause { + Ok(()) + } else { + Err(format!("tardó {took:?}: hubo pausa sin motivo")) + } + } + other => Err(format!("llegó {other:?}")), + } + }, + ); + b.group("D · Teclado"); b.case("D1", "el layout activo resuelve la «v»", || { diff --git a/crates/cp-mac/src/capture.rs b/crates/cp-mac/src/capture.rs index 51a4a93..acd1eac 100644 --- a/crates/cp-mac/src/capture.rs +++ b/crates/cp-mac/src/capture.rs @@ -10,6 +10,7 @@ pub enum Captured { Kept(Item), Nothing, TooSlow, + Superseded, } pub const PATIENCE: std::time::Duration = std::time::Duration::from_millis(400); @@ -23,6 +24,29 @@ pub fn capture_within(patience: std::time::Duration) -> Captured { .unwrap_or(Captured::TooSlow) } +pub fn capture_insisting(patience: std::time::Duration, retry: cp_core::watch::Retry) -> Captured { + insisting(retry, pasteboard::change_count_from_any_thread, || { + capture_within(patience) + }) +} + +fn insisting( + retry: cp_core::watch::Retry, + count: impl Fn() -> i64, + mut once: impl FnMut() -> Captured, +) -> Captured { + use cp_core::watch::{Retried, insist}; + let attempt = || match once() { + Captured::TooSlow => None, + other => Some(other), + }; + match insist(retry, count, attempt, std::thread::sleep) { + Retried::Done(captured) => captured, + Retried::Superseded => Captured::Superseded, + Retried::Exhausted => Captured::TooSlow, + } +} + pub fn capture(pb: &Pasteboard) -> Option { let offered = pb.types(); let ids: Vec<&str> = offered.iter().map(String::as_str).collect(); @@ -140,6 +164,75 @@ fn refine(family: Option, formats: &[Format]) -> Option { mod tests { use super::*; + fn quick() -> cp_core::watch::Retry { + cp_core::watch::Retry { + attempts: 3, + pause: std::time::Duration::from_millis(1), + } + } + + #[test] + fn a_slow_source_that_answers_on_the_second_try_is_kept() { + let mut tries = 0; + let got = insisting( + quick(), + || 7, + || { + tries += 1; + if tries < 2 { + Captured::TooSlow + } else { + Captured::Kept(Item::plain("tarde pero llega")) + } + }, + ); + assert_eq!(got, Captured::Kept(Item::plain("tarde pero llega"))); + } + + #[test] + fn a_source_that_stays_slow_is_too_slow_in_the_end() { + let mut tries = 0; + let got = insisting( + quick(), + || 7, + || { + tries += 1; + Captured::TooSlow + }, + ); + assert_eq!(got, Captured::TooSlow); + assert_eq!(tries, 3); + } + + #[test] + fn an_empty_pasteboard_is_an_answer_not_a_delay() { + let mut tries = 0; + let got = insisting( + quick(), + || 7, + || { + tries += 1; + Captured::Nothing + }, + ); + assert_eq!(got, Captured::Nothing); + assert_eq!(tries, 1); + } + + #[test] + fn a_copy_made_while_insisting_supersedes_the_slow_one() { + let count = std::cell::Cell::new(7); + let got = insisting( + quick(), + || count.get(), + || { + count.set(8); + Captured::TooSlow + }, + ); + assert_eq!(got, Captured::Superseded); + } + #[test] fn the_patience_sits_between_the_two_measured_worlds() { let delivered = std::time::Duration::from_millis(9); diff --git a/crates/cp-mac/src/content.rs b/crates/cp-mac/src/content.rs new file mode 100644 index 0000000..e6f47b0 --- /dev/null +++ b/crates/cp-mac/src/content.rs @@ -0,0 +1,163 @@ +use cp_core::item::{Item, Payload, SYNTHETIC_IMAGE, SYNTHETIC_TEXT}; +use cp_core::paste_as::Content; + +pub fn content_of<'a>(item: &'a Item, ocr: Option<&'a str>) -> Content<'a> { + let bytes = |id: &str| { + item.format(id).and_then(|format| match &format.payload { + Payload::Inline(bytes) | Payload::Blob(bytes) => Some(bytes.as_slice()), + _ => None, + }) + }; + let text = |id: &str| bytes(id).and_then(|bytes| std::str::from_utf8(bytes).ok()); + let html = text("public.html"); + Content { + kind: item.kind, + text: text("public.utf8-plain-text").or_else(|| text(SYNTHETIC_TEXT)), + html, + rich: html.is_some() + || bytes("public.rtf").is_some() + || bytes("com.apple.flat-rtfd").is_some(), + png: bytes("public.png").or_else(|| bytes(SYNTHETIC_IMAGE)), + paths: text("public.file-url") + .map(|urls| { + urls.lines() + .filter(|line| !line.trim().is_empty()) + .map(cp_mac_sys::frontmost::path_of) + .collect() + }) + .unwrap_or_default(), + title: text("public.url-name"), + ocr, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use cp_core::item::Format; + use cp_core::kind::Kind; + use cp_core::paste_as::{Form, forms_for}; + + fn inline(id: &str, bytes: &[u8]) -> Format { + Format { + id: id.into(), + payload: Payload::Inline(bytes.to_vec()), + } + } + + #[test] + fn a_word_paragraph_is_rich_with_html_and_plain_text() { + let item = Item { + kind: Some(Kind::Text), + formats: vec![ + inline("public.rtf", b"{\\rtf1 hola}"), + inline("public.html", b"

hola

"), + inline("public.utf8-plain-text", b"hola"), + Format { + id: "public.tiff".into(), + payload: Payload::Announced { size: None }, + }, + ], + }; + let content = content_of(&item, None); + assert_eq!(content.text, Some("hola")); + assert_eq!(content.html, Some("

hola

")); + assert!(content.rich); + assert_eq!(content.png, None); + assert_eq!( + forms_for(&content), + vec![Form::AsIs, Form::PlainText, Form::Markdown] + ); + } + + #[test] + fn html_alone_is_enough_to_be_rich() { + let item = Item { + kind: Some(Kind::Text), + formats: vec![ + inline("public.html", b"hola"), + inline("public.utf8-plain-text", b"hola"), + ], + }; + let content = content_of(&item, None); + assert!(content.rich); + assert_eq!(content.html, Some("hola")); + } + + #[test] + fn an_rtf_only_copy_is_rich_but_has_no_markdown() { + let item = Item { + kind: Some(Kind::Text), + formats: vec![ + inline("com.apple.flat-rtfd", b"rtfd"), + inline("public.utf8-plain-text", b"hola"), + ], + }; + let content = content_of(&item, None); + assert!(content.rich); + assert_eq!(content.html, None); + assert_eq!(forms_for(&content), vec![Form::AsIs, Form::PlainText]); + } + + #[test] + fn finder_urls_become_paths_and_a_link_brings_its_title() { + let files = Item { + kind: Some(Kind::File), + formats: vec![inline( + "public.file-url", + b"file:///tmp/cp%20a9.png\nfile:///tmp/dos.txt\n", + )], + }; + let content = content_of(&files, None); + assert_eq!(content.paths, vec!["/tmp/cp a9.png", "/tmp/dos.txt"]); + assert_eq!(content.text, None); + + let link = Item { + kind: Some(Kind::Link), + formats: vec![ + inline("public.utf8-plain-text", b"https://ejemplo.test"), + inline("public.url", b"https://ejemplo.test"), + inline("public.url-name", "Ejemplo — inicio".as_bytes()), + ], + }; + let content = content_of(&link, None); + assert_eq!(content.title, Some("Ejemplo — inicio")); + assert!(forms_for(&content).contains(&Form::LinkTitled)); + } + + #[test] + fn a_stored_image_and_what_was_read_in_it() { + let item = Item { + kind: Some(Kind::Image), + formats: vec![inline("public.png", &[137, 80, 78, 71])], + }; + let content = content_of(&item, Some("Pedido 4417")); + assert_eq!(content.png, Some(&[137u8, 80, 78, 71][..])); + assert_eq!(content.ocr, Some("Pedido 4417")); + assert!(!content.rich); + assert_eq!( + forms_for(&content), + vec![Form::AsIs, Form::ImageJpeg, Form::ImageOcr] + ); + } + + #[test] + fn what_the_store_wrote_itself_is_read_back_the_same_way() { + let edited = Item::plain("editado a mano"); + assert_eq!(content_of(&edited, None).text, Some("editado a mano")); + let synthetic = Item { + kind: Some(Kind::Image), + formats: vec![inline(SYNTHETIC_IMAGE, &[1, 2, 3])], + }; + assert_eq!(content_of(&synthetic, None).png, Some(&[1u8, 2, 3][..])); + } + + #[test] + fn bytes_that_are_not_text_do_not_pretend_to_be() { + let item = Item { + kind: Some(Kind::Text), + formats: vec![inline("public.utf8-plain-text", &[0xff, 0xfe, 0x00])], + }; + assert_eq!(content_of(&item, None).text, None); + } +} diff --git a/crates/cp-mac/src/lib.rs b/crates/cp-mac/src/lib.rs index 7be00e1..0ee38bf 100644 --- a/crates/cp-mac/src/lib.rs +++ b/crates/cp-mac/src/lib.rs @@ -1,6 +1,7 @@ #![cfg(target_os = "macos")] pub mod capture; +pub mod content; pub mod formats; pub mod paste; pub mod restore; diff --git a/crates/cp-mac/src/restore.rs b/crates/cp-mac/src/restore.rs index d0b6ebb..1008631 100644 --- a/crates/cp-mac/src/restore.rs +++ b/crates/cp-mac/src/restore.rs @@ -1,17 +1,28 @@ -use cp_core::item::{Item, Payload}; +use cp_core::item::{Item, Payload, SYNTHETIC_IMAGE, SYNTHETIC_TEXT}; use cp_mac_sys::pasteboard::Pasteboard; -pub fn to_pasteboard(pb: &Pasteboard, item: &Item) -> Restored { - let writable: Vec<(&str, &[u8])> = item - .formats +fn type_of(id: &str) -> &str { + match id { + SYNTHETIC_TEXT => PLAIN_TEXT, + SYNTHETIC_IMAGE => PNG, + other => other, + } +} + +fn writable_of(item: &Item) -> Vec<(&str, &[u8])> { + item.formats .iter() .filter_map(|format| match &format.payload { Payload::Inline(bytes) | Payload::Blob(bytes) => { - Some((format.id.as_str(), bytes.as_slice())) + Some((type_of(&format.id), bytes.as_slice())) } _ => None, }) - .collect(); + .collect() +} + +pub fn to_pasteboard(pb: &Pasteboard, item: &Item) -> Restored { + let writable = writable_of(item); if writable.is_empty() { return Restored::NothingToWrite; @@ -28,17 +39,14 @@ pub fn to_pasteboard(pb: &Pasteboard, item: &Item) -> Restored { } } +fn plain_text_of(item: &Item) -> Option<&[u8]> { + writable_of(item) + .into_iter() + .find_map(|(kind, bytes)| (kind == PLAIN_TEXT).then_some(bytes)) +} + pub fn to_pasteboard_as_plain_text(pb: &Pasteboard, item: &Item) -> Restored { - let Some(text) = item - .formats - .iter() - .find_map(|format| match &format.payload { - Payload::Inline(bytes) | Payload::Blob(bytes) if format.id == PLAIN_TEXT => { - Some(bytes.as_slice()) - } - _ => None, - }) - else { + let Some(text) = plain_text_of(item) else { return Restored::NothingToWrite; }; @@ -53,6 +61,7 @@ pub fn to_pasteboard_as_plain_text(pb: &Pasteboard, item: &Item) -> Restored { } const PLAIN_TEXT: &str = "public.utf8-plain-text"; +const PNG: &str = "public.png"; #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Restored { @@ -60,3 +69,67 @@ pub enum Restored { NothingToWrite, Failed, } + +#[cfg(test)] +mod tests { + use super::*; + use cp_core::item::Format; + + fn inline(id: &str, bytes: &[u8]) -> Format { + Format { + id: id.into(), + payload: Payload::Inline(bytes.to_vec()), + } + } + + #[test] + fn a_synthetic_text_goes_back_as_the_type_the_system_reads() { + let item = Item { + kind: None, + formats: vec![inline(SYNTHETIC_TEXT, b"hola")], + }; + assert_eq!(writable_of(&item), vec![(PLAIN_TEXT, &b"hola"[..])]); + assert_eq!(plain_text_of(&item), Some(&b"hola"[..])); + } + + #[test] + fn a_synthetic_image_goes_back_as_a_png() { + let item = Item { + kind: None, + formats: vec![inline(SYNTHETIC_IMAGE, &[137, 80, 78, 71])], + }; + assert_eq!(writable_of(&item), vec![(PNG, &[137u8, 80, 78, 71][..])]); + } + + #[test] + fn a_captured_type_goes_back_under_its_own_name() { + let item = Item { + kind: None, + formats: vec![ + inline("public.rtf", b"{\\rtf1 hola}"), + inline(PLAIN_TEXT, b"hola"), + Format { + id: "public.tiff".into(), + payload: Payload::Announced { size: None }, + }, + ], + }; + let written = writable_of(&item); + assert_eq!(written.len(), 2, "lo anunciado sin bytes no se escribe"); + assert_eq!(written[0].0, "public.rtf"); + assert_eq!(plain_text_of(&item), Some(&b"hola"[..])); + } + + #[test] + fn nothing_readable_is_nothing_to_write() { + let item = Item { + kind: None, + formats: vec![Format { + id: PLAIN_TEXT.into(), + payload: Payload::Absent, + }], + }; + assert!(writable_of(&item).is_empty()); + assert_eq!(plain_text_of(&item), None); + } +} diff --git a/crates/cp-store/src/blobs.rs b/crates/cp-store/src/blobs.rs index db8926d..b6e99e6 100644 --- a/crates/cp-store/src/blobs.rs +++ b/crates/cp-store/src/blobs.rs @@ -24,7 +24,9 @@ impl Blobs { pub fn put(&self, bytes: &[u8]) -> Result { let digest = blake3::hash(bytes).to_hex().to_string(); let path = self.path_for(&digest); - if path.exists() { + if let Ok(file) = std::fs::File::options().write(true).open(&path) { + file.set_modified(std::time::SystemTime::now()) + .map_err(Error::Io)?; return Ok(digest); } if let Some(parent) = path.parent() { @@ -48,19 +50,71 @@ impl Blobs { } pub fn remove(&self, digest: &str) -> Result<()> { - let path = self.path_for(digest); - let Ok(metadata) = std::fs::metadata(&path) else { - return Ok(()); - }; - let zeros = vec![0u8; metadata.len() as usize]; - std::fs::write(&path, &zeros).map_err(Error::Io)?; - std::fs::remove_file(&path).map_err(Error::Io)?; - Ok(()) + remove_at(&self.path_for(digest)) } pub fn exists(&self, digest: &str) -> bool { self.path_for(digest).exists() } + + pub const GRACE: std::time::Duration = std::time::Duration::from_secs(60); + + pub fn sweep(&self, referenced: &dyn Fn(&str) -> bool) -> Result { + let settled = std::time::SystemTime::now() - Self::GRACE; + let mut removed = 0; + for path in files_under(&self.root) { + let Some(name) = path.file_name().and_then(|name| name.to_str()) else { + continue; + }; + let Some((digest, partial)) = digest_of(name) else { + continue; + }; + let fresh = std::fs::metadata(&path) + .and_then(|meta| meta.modified()) + .is_ok_and(|modified| modified >= settled); + if fresh || (!partial && referenced(digest)) { + continue; + } + remove_at(&path)?; + removed += 1; + } + Ok(removed) + } +} + +fn digest_of(name: &str) -> Option<(&str, bool)> { + let (digest, partial) = match name.strip_suffix(".partial") { + Some(digest) => (digest, true), + None => (name, false), + }; + let shaped = digest.len() == 64 && digest.bytes().all(|b| b.is_ascii_hexdigit()); + shaped.then_some((digest, partial)) +} + +fn remove_at(path: &Path) -> Result<()> { + let Ok(metadata) = std::fs::metadata(path) else { + return Ok(()); + }; + let zeros = vec![0u8; metadata.len() as usize]; + std::fs::write(path, &zeros).map_err(Error::Io)?; + std::fs::remove_file(path).map_err(Error::Io)?; + Ok(()) +} + +pub(crate) fn files_under(root: &Path) -> Vec { + let mut found = Vec::new(); + let Ok(entries) = std::fs::read_dir(root) else { + return found; + }; + for entry in entries.flatten() { + let path = entry.path(); + if path.is_dir() { + found.extend(files_under(&path)); + } else { + found.push(path); + } + } + found } #[cfg(test)] @@ -143,26 +197,107 @@ mod tests { fn nothing_is_left_behind_when_writing_succeeds() { let (dir, blobs) = temporary(); blobs.put(b"algo").expect("guarda"); - let partials = walk(&dir.path().join("blobs")) + let partials = files_under(&dir.path().join("blobs")) .into_iter() .filter(|p| p.extension().is_some_and(|e| e == "partial")) .count(); assert_eq!(partials, 0, "el temporal se renombra, no se queda"); } - fn walk(root: &Path) -> Vec { - let mut found = Vec::new(); - let Ok(entries) = std::fs::read_dir(root) else { - return found; - }; - for entry in entries.flatten() { - let path = entry.path(); - if path.is_dir() { - found.extend(walk(&path)); - } else { - found.push(path); - } + fn aged(path: &Path) { + let file = std::fs::File::options() + .write(true) + .open(path) + .expect("abre"); + file.set_modified(std::time::UNIX_EPOCH).expect("envejece"); + } + + #[test] + fn a_blob_nobody_references_is_removed_once_it_is_old_enough() { + let (_dir, blobs) = temporary(); + let digest = blobs.put(b"huerfano").expect("guarda"); + assert_eq!( + blobs.sweep(&|_| false).expect("barre"), + 0, + "recién escrito: puede ser de un ítem a medio guardar" + ); + aged(&blobs.path_for(&digest)); + assert_eq!(blobs.sweep(&|_| false).expect("barre"), 1); + assert!(!blobs.exists(&digest)); + } + + #[test] + fn a_referenced_blob_survives_the_sweep_however_old() { + let (_dir, blobs) = temporary(); + let digest = blobs.put(b"con dueno").expect("guarda"); + aged(&blobs.path_for(&digest)); + let keep = digest.clone(); + assert_eq!(blobs.sweep(&|name| name == keep).expect("barre"), 0); + assert!(blobs.exists(&digest)); + } + + #[test] + fn a_partial_file_left_by_a_crash_goes_too() { + let (_dir, blobs) = temporary(); + let digest = "b".repeat(64); + let leftover = blobs.path_for(&digest).with_extension("partial"); + std::fs::create_dir_all(leftover.parent().expect("padre")).expect("carpeta"); + std::fs::write(&leftover, b"a medias").expect("escribe"); + aged(&leftover); + assert_eq!(blobs.sweep(&|_| true).expect("barre"), 1); + assert!(!leftover.exists(), "aunque su digest esté referenciado"); + } + + #[test] + fn sweeping_an_empty_store_is_nothing() { + let (_dir, blobs) = temporary(); + assert_eq!(blobs.sweep(&|_| false).expect("barre"), 0); + } + + #[test] + fn a_file_that_is_not_a_blob_is_neither_touched_nor_counted() { + let (dir, blobs) = temporary(); + let root = dir.path().join("blobs"); + let strays = [ + root.join(".DS_Store"), + root.join("x"), + root.join("ñ.txt"), + root.join("ab").join("cd").join("notes.partial"), + ]; + for stray in &strays { + std::fs::create_dir_all(stray.parent().expect("padre")).expect("carpeta"); + std::fs::write(stray, b"ajeno").expect("escribe"); + aged(stray); + } + assert_eq!(blobs.sweep(&|_| false).expect("barre"), 0); + for stray in &strays { + assert!(stray.exists(), "{} no era nuestro", stray.display()); } - found + } + + #[test] + fn only_a_digest_shaped_name_is_a_blob() { + let digest = "0123456789abcdef".repeat(4); + assert_eq!(digest_of(&digest), Some((digest.as_str(), false))); + let partial = format!("{digest}.partial"); + assert_eq!(digest_of(&partial), Some((digest.as_str(), true))); + assert_eq!(digest_of(".DS_Store"), None); + assert_eq!(digest_of(&"g".repeat(64)), None, "no es hexadecimal"); + assert_eq!(digest_of(&"a".repeat(63)), None, "le falta uno"); + assert_eq!(digest_of("x.partial"), None); + } + + #[test] + fn reclaiming_a_blob_that_already_exists_makes_it_fresh_again() { + let (_dir, blobs) = temporary(); + let digest = blobs.put(b"reclamado").expect("guarda"); + aged(&blobs.path_for(&digest)); + blobs.put(b"reclamado").expect("otra vez"); + assert_eq!( + blobs.sweep(&|_| false).expect("barre"), + 0, + "quien lo acaba de reclamar aún no ha escrito su fila" + ); + assert!(blobs.exists(&digest)); } } diff --git a/crates/cp-store/src/lib.rs b/crates/cp-store/src/lib.rs index e0dfc7d..c5c3788 100644 --- a/crates/cp-store/src/lib.rs +++ b/crates/cp-store/src/lib.rs @@ -1,10 +1,15 @@ pub mod blobs; +pub mod query; pub mod schema; pub mod store; pub use blobs::Blobs; +pub use query::{Clock, parse}; pub use schema::SCHEMA_VERSION; -pub use store::{Restricted, Store}; +pub use store::{ + AppCount, Broken, Cursor, Facet, Filter, Listed, Order, Page, Policy, Restricted, Snippet, + Store, Swept, Usage, Where, +}; #[derive(Debug, thiserror::Error)] pub enum Error { @@ -16,6 +21,10 @@ pub enum Error { Io(#[from] std::io::Error), #[error("la base es de la versión {found} y esta copia entiende hasta la {supported}")] FromTheFuture { found: u32, supported: u32 }, + #[error("{size} bytes es más de lo que un ítem puede guardar")] + TooBig { size: usize }, + #[error("no hay ningún ítem {id}")] + NoSuchItem { id: i64 }, } pub type Result = std::result::Result; diff --git a/crates/cp-store/src/query.rs b/crates/cp-store/src/query.rs new file mode 100644 index 0000000..908ff0a --- /dev/null +++ b/crates/cp-store/src/query.rs @@ -0,0 +1,394 @@ +use crate::store::{Broken, Filter, Order}; +use cp_core::kind::Kind; + +pub const SECOND: i64 = 1_000; +pub const MINUTE: i64 = 60 * SECOND; +pub const HOUR: i64 = 60 * MINUTE; +pub const DAY: i64 = 24 * HOUR; +pub const WEEK: i64 = 7 * DAY; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Clock { + pub now: i64, + pub day_start: i64, +} + +pub const COLORS: [(&str, i64); 7] = [ + ("none", 0), + ("red", 1), + ("green", 2), + ("purple", 3), + ("yellow", 4), + ("blue", 5), + ("orange", 6), +]; + +pub fn parse(input: &str, clock: &Clock) -> Filter { + let mut filter = Filter::default(); + let mut words: Vec = Vec::new(); + let mut label: Vec = Vec::new(); + for token in tokens(input) { + match operator(&token, clock) { + Some(Op::Kinds(kinds, false)) => filter.kinds.extend(kinds), + Some(Op::Kinds(kinds, true)) => filter.exclude_kinds.extend(kinds), + Some(Op::Apps(apps, false)) => filter.apps.extend(apps), + Some(Op::Apps(apps, true)) => filter.exclude_apps.extend(apps), + Some(Op::Colors(colors)) => filter.colors.extend(colors), + Some(Op::Since(at)) => filter.since = Some(filter.since.map_or(at, |had| had.max(at))), + Some(Op::Pinned) => filter.pinned_only = true, + Some(Op::OnlyBroken) => filter.broken = Broken::Only, + Some(Op::Label(text)) => label.push(text), + Some(Op::Order(order)) => filter.order = order, + None => words.push(token), + } + } + filter.query = (!words.is_empty()).then(|| words.join(" ")); + filter.label = (!label.is_empty()).then(|| label.join(" ")); + filter +} + +enum Op { + Kinds(Vec, bool), + Apps(Vec, bool), + Colors(Vec), + Since(i64), + Pinned, + OnlyBroken, + Label(String), + Order(Order), +} + +fn operator(token: &str, clock: &Clock) -> Option { + let (negated, body) = match token.strip_prefix('-') { + Some(rest) => (true, rest), + None => (false, token), + }; + let (key, value) = split(body)?; + if value.is_empty() { + return None; + } + let values: Vec<&str> = value.split(',').filter(|one| !one.is_empty()).collect(); + if values.is_empty() { + return None; + } + match key { + "k" | "kind" | "type" | "t" => values + .iter() + .map(|one| Kind::from_name(&one.to_ascii_lowercase())) + .collect::>>() + .map(|kinds| Op::Kinds(kinds, negated)), + "a" | "app" => Some(Op::Apps( + values.iter().map(|one| (*one).to_owned()).collect(), + negated, + )), + "c" | "color" if !negated => values + .iter() + .map(|one| color(one)) + .collect::>>() + .map(Op::Colors), + "d" | "date" | "since" if !negated => since(value, clock).map(Op::Since), + "is" if !negated => match value { + "pinned" => Some(Op::Pinned), + "broken" => Some(Op::OnlyBroken), + _ => None, + }, + "l" | "label" if !negated => Some(Op::Label(value.to_owned())), + "sort" | "order" if !negated => match value { + "recent" => Some(Op::Order(Order::Recent)), + "pasted" => Some(Op::Order(Order::MostPasted)), + "used" => Some(Op::Order(Order::LastUsed)), + _ => None, + }, + _ => None, + } +} + +fn split(body: &str) -> Option<(&str, &str)> { + let symbol = match body.chars().next()? { + '/' => Some("k"), + '@' => Some("a"), + '#' => Some("c"), + '~' => Some("d"), + _ => None, + }; + if let Some(key) = symbol { + return Some((key, &body[1..])); + } + let (key, value) = body.split_once(':')?; + let known = key.chars().all(|c| c.is_ascii_alphabetic()); + known.then_some((key, value)) +} + +fn color(name: &str) -> Option { + let lower = name.to_ascii_lowercase(); + if let Ok(index) = lower.parse::() { + return COLORS + .iter() + .any(|(_, value)| *value == index) + .then_some(index); + } + COLORS + .iter() + .find(|(known, _)| *known == lower) + .map(|(_, value)| *value) +} + +fn since(value: &str, clock: &Clock) -> Option { + let lower = value.to_ascii_lowercase(); + if lower == "today" { + return Some(clock.day_start); + } + let unit = match lower.chars().last()? { + 'm' => MINUTE, + 'h' => HOUR, + 'd' => DAY, + 'w' => WEEK, + _ => return None, + }; + let amount: i64 = lower[..lower.len() - 1].parse().ok()?; + (amount > 0).then(|| clock.now.saturating_sub(amount.saturating_mul(unit))) +} + +fn tokens(input: &str) -> Vec { + let mut out = Vec::new(); + let mut current = String::new(); + let mut quoted = false; + for c in input.chars() { + match c { + '"' => quoted = !quoted, + c if c.is_whitespace() && !quoted => { + if !current.is_empty() { + out.push(std::mem::take(&mut current)); + } + } + c => current.push(c), + } + } + if !current.is_empty() { + out.push(current); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + const CLOCK: Clock = Clock { + now: 1_000 * DAY + 5 * HOUR, + day_start: 1_000 * DAY, + }; + + fn parsed(input: &str) -> Filter { + parse(input, &CLOCK) + } + + #[test] + fn plain_words_are_the_search() { + let filter = parsed("el café de la esquina"); + assert_eq!(filter.query.as_deref(), Some("el café de la esquina")); + assert_eq!( + filter, + Filter { + query: Some("el café de la esquina".into()), + ..Default::default() + } + ); + } + + #[test] + fn nothing_typed_is_the_whole_history() { + assert_eq!(parsed(""), Filter::default()); + assert_eq!(parsed(" "), Filter::default()); + } + + #[test] + fn a_class_by_key_or_by_symbol() { + assert_eq!(parsed("k:json").kinds, vec![Kind::Json]); + assert_eq!(parsed("kind:Link").kinds, vec![Kind::Link]); + assert_eq!(parsed("/image").kinds, vec![Kind::Image]); + assert_eq!(parsed("t:code,json").kinds, vec![Kind::Code, Kind::Json]); + } + + #[test] + fn a_class_can_be_left_out() { + let filter = parsed("-k:image -/video"); + assert!(filter.kinds.is_empty()); + assert_eq!(filter.exclude_kinds, vec![Kind::Image, Kind::Video]); + assert_eq!(filter.query, None); + } + + #[test] + fn a_prefix_is_only_an_operator_when_its_value_is_known() { + assert_eq!(parsed("/nada").query.as_deref(), Some("/nada")); + assert_eq!(parsed("k:foto").query.as_deref(), Some("k:foto")); + assert_eq!( + parsed("k:image,foto").query.as_deref(), + Some("k:image,foto"), + "una lista con un desconocido no es media lista" + ); + assert_eq!(parsed("#FF8800").query.as_deref(), Some("#FF8800")); + assert_eq!(parsed("c:9").query.as_deref(), Some("c:9")); + assert_eq!(parsed("d:soon").query.as_deref(), Some("d:soon")); + assert_eq!(parsed("is:new").query.as_deref(), Some("is:new")); + assert_eq!(parsed("sort:size").query.as_deref(), Some("sort:size")); + } + + #[test] + fn an_application_by_key_by_symbol_and_with_spaces() { + assert_eq!(parsed("@slack").apps, vec!["slack"]); + assert_eq!(parsed("a:Code").apps, vec!["Code"]); + assert_eq!(parsed("app:\"Google Chrome\"").apps, vec!["Google Chrome"]); + assert_eq!(parsed("-@code").exclude_apps, vec!["code"]); + assert_eq!(parsed("a:slack,code").apps, vec!["slack", "code"]); + } + + #[test] + fn an_email_is_not_an_application() { + let filter = parsed("escribe a juan@ejemplo.test"); + assert!(filter.apps.is_empty()); + assert_eq!(filter.query.as_deref(), Some("escribe a juan@ejemplo.test")); + } + + #[test] + fn a_colour_by_name_or_by_number() { + assert_eq!(parsed("c:red").colors, vec![1]); + assert_eq!(parsed("#Blue").colors, vec![5]); + assert_eq!(parsed("color:3").colors, vec![3]); + assert_eq!(parsed("c:red,orange").colors, vec![1, 6]); + assert_eq!(parsed("c:none").colors, vec![0]); + } + + #[test] + fn a_relative_date_counts_back_from_now() { + assert_eq!(parsed("d:today").since, Some(CLOCK.day_start)); + assert_eq!(parsed("~1h").since, Some(CLOCK.now - HOUR)); + assert_eq!(parsed("since:7d").since, Some(CLOCK.now - WEEK)); + assert_eq!(parsed("date:30m").since, Some(CLOCK.now - 30 * MINUTE)); + assert_eq!(parsed("~2w").since, Some(CLOCK.now - 2 * WEEK)); + } + + #[test] + fn two_dates_keep_the_narrower_one() { + assert_eq!(parsed("~7d ~1h").since, Some(CLOCK.now - HOUR)); + assert_eq!(parsed("~1h ~7d").since, Some(CLOCK.now - HOUR)); + } + + #[test] + fn a_date_that_is_not_a_date_stays_text() { + assert_eq!(parsed("~0h").query.as_deref(), Some("~0h")); + assert_eq!(parsed("~h").query.as_deref(), Some("~h")); + assert_eq!(parsed("~1y").query.as_deref(), Some("~1y")); + assert_eq!(parsed("~ayer").query.as_deref(), Some("~ayer")); + } + + #[test] + fn an_absurd_amount_does_not_overflow() { + assert!(parsed("~99999999999999999d").since.is_some()); + let dawn = Clock { + now: i64::MIN + 1, + day_start: i64::MIN, + }; + assert_eq!(parse("~1h", &dawn).since, Some(i64::MIN)); + } + + #[test] + fn pinned_and_broken_are_states_not_words() { + assert!(parsed("is:pinned").pinned_only); + assert_eq!(parsed("is:broken").broken, Broken::Only); + assert_eq!(parsed("is:pinned").query, None); + } + + #[test] + fn the_label_is_searched_on_its_own_column() { + let filter = parsed("label:factura l:mayo"); + assert_eq!(filter.label.as_deref(), Some("factura mayo")); + assert_eq!(filter.query, None); + } + + #[test] + fn the_order_is_a_word_too() { + assert_eq!(parsed("sort:pasted").order, Order::MostPasted); + assert_eq!(parsed("order:used").order, Order::LastUsed); + let explicit = parsed("sort:recent"); + assert_eq!(explicit.order, Order::Recent); + assert_eq!(explicit.query, None, "es un operador, no texto"); + } + + #[test] + fn a_negated_state_or_date_is_not_an_operator() { + assert_eq!(parsed("-is:pinned").query.as_deref(), Some("-is:pinned")); + assert_eq!(parsed("-~1h").query.as_deref(), Some("-~1h")); + assert_eq!(parsed("-c:red").query.as_deref(), Some("-c:red")); + assert_eq!(parsed("-l:factura").query.as_deref(), Some("-l:factura")); + assert_eq!(parsed("-sort:used").query.as_deref(), Some("-sort:used")); + assert_eq!(parsed("-l:factura").label, None); + assert_eq!(parsed("-sort:used").order, Order::Recent); + } + + #[test] + fn the_units_are_milliseconds() { + assert_eq!(SECOND, 1_000); + assert_eq!(MINUTE, 60_000); + assert_eq!(HOUR, 3_600_000); + assert_eq!(DAY, 86_400_000); + assert_eq!(WEEK, 604_800_000); + } + + #[test] + fn an_empty_value_is_text() { + assert_eq!(parsed("k:").query.as_deref(), Some("k:")); + assert_eq!(parsed("@").query.as_deref(), Some("@")); + assert_eq!(parsed("a:,").query.as_deref(), Some("a:,")); + } + + #[test] + fn a_colon_inside_ordinary_text_is_not_a_key() { + assert_eq!( + parsed("https://ejemplo.test").query.as_deref(), + Some("https://ejemplo.test"), + "https no es una clave" + ); + assert_eq!(parsed("12:30").query.as_deref(), Some("12:30")); + assert_eq!(parsed("a-b:c").query.as_deref(), Some("a-b:c")); + } + + #[test] + fn the_example_from_the_design_note() { + let filter = parsed("token @code ~1h"); + assert_eq!( + filter, + Filter { + query: Some("token".into()), + apps: vec!["code".into()], + since: Some(CLOCK.now - HOUR), + ..Default::default() + } + ); + } + + #[test] + fn everything_at_once() { + let filter = parsed(" pedido k:json,text -@safari #red ~7d is:pinned l:mayo sort:used "); + assert_eq!(filter.query.as_deref(), Some("pedido")); + assert_eq!(filter.kinds, vec![Kind::Json, Kind::Text]); + assert_eq!(filter.exclude_apps, vec!["safari"]); + assert_eq!(filter.colors, vec![1]); + assert_eq!(filter.since, Some(CLOCK.now - WEEK)); + assert!(filter.pinned_only); + assert_eq!(filter.label.as_deref(), Some("mayo")); + assert_eq!(filter.order, Order::LastUsed); + } + + #[test] + fn quotes_group_words_and_then_disappear() { + assert_eq!( + parsed("\"dos palabras\"").query.as_deref(), + Some("dos palabras") + ); + assert_eq!( + parsed("sin cerrar \"la cita").query.as_deref(), + Some("sin cerrar la cita") + ); + } +} diff --git a/crates/cp-store/src/schema.rs b/crates/cp-store/src/schema.rs index f4a9e0b..1335059 100644 --- a/crates/cp-store/src/schema.rs +++ b/crates/cp-store/src/schema.rs @@ -1,6 +1,6 @@ use rusqlite::{Connection, Result}; -pub const SCHEMA_VERSION: u32 = 1; +pub const SCHEMA_VERSION: u32 = 2; pub fn migrate(db: &Connection) -> crate::Result { let found: u32 = db.query_row("PRAGMA user_version", [], |row| row.get(0))?; @@ -10,6 +10,13 @@ pub fn migrate(db: &Connection) -> crate::Result { supported: SCHEMA_VERSION, }); } + if found == 1 { + db.execute_batch(&format!( + "DROP INDEX IF EXISTS items_by_recency; + DROP INDEX IF EXISTS items_by_kind; + {ORDERING_INDEXES}" + ))?; + } if found < SCHEMA_VERSION { db.execute_batch(&format!("PRAGMA user_version = {SCHEMA_VERSION};"))?; return Ok(true); @@ -17,6 +24,11 @@ pub fn migrate(db: &Connection) -> crate::Result { Ok(false) } +const ORDERING_INDEXES: &str = " + CREATE INDEX IF NOT EXISTS items_by_recency ON items(modified_at DESC, id DESC); + CREATE INDEX IF NOT EXISTS items_by_kind ON items(kind, modified_at DESC, id DESC); +"; + pub fn configure(db: &Connection) -> Result<()> { db.execute_batch( r#" @@ -32,8 +44,12 @@ pub fn configure(db: &Connection) -> Result<()> { pub fn create(db: &Connection) -> Result<()> { configure(db)?; - db.execute_batch( - r#" + db.execute_batch(TABLES)?; + db.execute_batch(ORDERING_INDEXES)?; + Ok(()) +} + +const TABLES: &str = r#" CREATE TABLE IF NOT EXISTS items ( id INTEGER PRIMARY KEY, uuid TEXT NOT NULL UNIQUE, @@ -59,10 +75,8 @@ pub fn create(db: &Connection) -> Result<()> { deleted_at INTEGER ); - CREATE INDEX IF NOT EXISTS items_by_recency ON items(modified_at DESC); CREATE INDEX IF NOT EXISTS items_by_creation ON items(created_at); CREATE INDEX IF NOT EXISTS items_by_hash ON items(content_hash); - CREATE INDEX IF NOT EXISTS items_by_kind ON items(kind, modified_at DESC); CREATE INDEX IF NOT EXISTS items_by_color ON items(card_color); CREATE INDEX IF NOT EXISTS items_pinned ON items(pinned) WHERE pinned = 1; CREATE INDEX IF NOT EXISTS items_broken ON items(broken_since) @@ -70,6 +84,10 @@ pub fn create(db: &Connection) -> Result<()> { CREATE INDEX IF NOT EXISTS items_by_version ON items(updated_at); CREATE INDEX IF NOT EXISTS items_deleted ON items(deleted_at) WHERE deleted_at IS NOT NULL; + CREATE INDEX IF NOT EXISTS items_by_app ON items(search_app); + CREATE INDEX IF NOT EXISTS items_by_pastes ON items(paste_count DESC, id DESC); + CREATE INDEX IF NOT EXISTS items_by_use + ON items(COALESCE(last_used_at, -1) DESC, id DESC); CREATE TABLE IF NOT EXISTS item_formats ( item_id INTEGER NOT NULL REFERENCES items(id) ON DELETE CASCADE, @@ -131,9 +149,7 @@ pub fn create(db: &Connection) -> Result<()> { INSERT INTO items_fts(rowid, search_text, search_label, search_app, search_ocr) VALUES (new.id, new.search_text, new.search_label, new.search_app, new.search_ocr); END; - "#, - ) -} + "#; #[cfg(test)] mod tests { @@ -207,4 +223,40 @@ mod tests { create(&db).expect("primera"); create(&db).expect("segunda"); } + + fn index_sql(db: &Connection, name: &str) -> String { + db.query_row( + "SELECT sql FROM sqlite_master WHERE type = 'index' AND name = ?1", + [name], + |row| row.get(0), + ) + .expect("el índice existe") + } + + #[test] + fn a_version_one_database_gets_its_ordering_indexes_rebuilt() { + let db = Connection::open_in_memory().expect("abre"); + create(&db).expect("esquema"); + db.execute_batch( + "DROP INDEX items_by_recency; + DROP INDEX items_by_kind; + CREATE INDEX items_by_recency ON items(modified_at DESC); + CREATE INDEX items_by_kind ON items(kind, modified_at DESC); + PRAGMA user_version = 1;", + ) + .expect("una base como la dejó la versión 1"); + create(&db).expect("abrir la vuelve a crear sin tocar lo que existe"); + assert!( + !index_sql(&db, "items_by_recency").contains("id DESC"), + "IF NOT EXISTS no rehace un índice viejo: por eso hace falta migrar" + ); + + assert!(migrate(&db).expect("migra")); + assert!(index_sql(&db, "items_by_recency").contains("modified_at DESC, id DESC")); + assert!(index_sql(&db, "items_by_kind").contains("kind, modified_at DESC, id DESC")); + let version: u32 = db + .query_row("PRAGMA user_version", [], |row| row.get(0)) + .expect("consulta"); + assert_eq!(version, SCHEMA_VERSION); + } } diff --git a/crates/cp-store/src/store.rs b/crates/cp-store/src/store.rs index 0741314..3cf0d87 100644 --- a/crates/cp-store/src/store.rs +++ b/crates/cp-store/src/store.rs @@ -1,32 +1,210 @@ use crate::{Error, Result}; use cp_core::item::{Item, Payload}; -use cp_core::search::fold; -use rusqlite::{Connection, OptionalExtension, params}; +use cp_core::kind::Kind; +use cp_core::search::{EXCERPT_CHARS, Excerpt, excerpt, fold, terms_of}; +use rusqlite::types::ToSql; +use rusqlite::{Connection, OptionalExtension, params, params_from_iter}; -fn fts_expression(folded: &str) -> Option { - let terms: Vec = folded - .split_whitespace() - .filter(|word| word.chars().any(char::is_alphanumeric)) +fn fts_expression(query: &str) -> Option { + let terms: Vec = terms_of(query) + .iter() .map(|word| format!("\"{}\"*", word.replace('"', "\"\""))) .collect(); (!terms.is_empty()).then(|| terms.join(" ")) } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Where { + Text, + Label, + App, + Ocr, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Snippet { + pub found_in: Where, + pub excerpt: Excerpt, +} + #[derive(Debug, Clone, PartialEq, Eq)] pub struct Listed { pub id: i64, pub modified_at: i64, - pub kind: Option, + pub created_at: i64, + pub kind: Option, pub preview: String, + pub app: Option, + pub label: Option, + pub color: i64, + pub thumb_path: Option, + pub paste_count: i64, + pub last_used_at: Option, + pub broken_since: Option, pub pinned: bool, + pub snippet: Option, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Cursor { + key: i64, + id: i64, +} + +#[derive(Debug, Clone, PartialEq, Eq, Default)] +pub struct Page { + pub rows: Vec, + pub next: Option, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum Order { + #[default] + Recent, + MostPasted, + LastUsed, +} + +impl Order { + fn key(self) -> &'static str { + match self { + Order::Recent => "items.modified_at", + Order::MostPasted => "items.paste_count", + Order::LastUsed => "COALESCE(items.last_used_at, -1)", + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum Broken { + #[default] + Hidden, + Shown, + Only, } -#[derive(Debug, Clone, Default)] +#[derive(Debug, Clone, Default, PartialEq, Eq)] pub struct Filter { pub query: Option, - pub kinds: Vec, + pub label: Option, + pub kinds: Vec, + pub exclude_kinds: Vec, + pub apps: Vec, + pub exclude_apps: Vec, pub colors: Vec, pub pinned_only: bool, + pub since: Option, + pub broken: Broken, + pub order: Order, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Facet { + pub kind: Kind, + pub count: i64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct AppCount { + pub app: String, + pub count: i64, +} + +struct Clauses { + joins_index: bool, + conditions: Vec, + bound: Vec>, +} + +impl Clauses { + fn of(filter: &Filter, with_kinds: bool) -> Option { + let mut clauses = Self { + joins_index: false, + conditions: vec!["items.deleted_at IS NULL".into()], + bound: Vec::new(), + }; + let mut index = Vec::new(); + if let Some(text) = &filter.query { + index.push(fts_expression(text)?); + } + if let Some(label) = &filter.label { + index.push(format!("search_label : ({})", fts_expression(label)?)); + } + if !index.is_empty() { + clauses.joins_index = true; + clauses.conditions.push("items_fts MATCH ?".into()); + clauses.bound.push(Box::new(index.join(" AND "))); + } + if with_kinds { + clauses.kinds(&filter.kinds, false); + } + clauses.kinds(&filter.exclude_kinds, true); + clauses.apps("IN", &filter.apps); + clauses.apps("NOT IN", &filter.exclude_apps); + if !filter.colors.is_empty() { + let list = filter + .colors + .iter() + .map(i64::to_string) + .collect::>() + .join(", "); + clauses + .conditions + .push(format!("items.card_color IN ({list})")); + } + if filter.pinned_only { + clauses.conditions.push("items.pinned = 1".into()); + } + if let Some(since) = filter.since { + clauses.conditions.push("items.modified_at >= ?".into()); + clauses.bound.push(Box::new(since)); + } + match filter.broken { + Broken::Hidden => clauses.conditions.push("items.broken_since IS NULL".into()), + Broken::Only => clauses + .conditions + .push("items.broken_since IS NOT NULL".into()), + Broken::Shown => {} + } + Some(clauses) + } + + fn kinds(&mut self, kinds: &[Kind], excluded: bool) { + if kinds.is_empty() { + return; + } + let list = kinds + .iter() + .map(|kind| format!("'{}'", kind.as_str())) + .collect::>() + .join(", "); + self.conditions.push(if excluded { + format!("(items.kind IS NULL OR items.kind NOT IN ({list}))") + } else { + format!("items.kind IN ({list})") + }); + } + + fn apps(&mut self, verb: &str, apps: &[String]) { + if apps.is_empty() { + return; + } + let marks = vec!["?"; apps.len()].join(", "); + self.conditions + .push(format!("items.search_app {verb} ({marks})")); + for app in apps { + self.bound.push(Box::new(fold(app))); + } + } + + fn source(&self) -> String { + let join = if self.joins_index { + "JOIN items_fts ON items.id = items_fts.rowid " + } else { + "" + }; + format!("FROM items {join}WHERE {}", self.conditions.join(" AND ")) + } } pub struct Store { @@ -267,6 +445,10 @@ impl Store { WHERE id = ?1", params![id, at], )?; + self.release(id) + } + + fn release(&self, id: i64) -> Result<()> { if let Some(blobs) = &self.blobs { for digest in self.blobs_of(id)? { if self.blob_is_shared(&digest, id)? { @@ -284,6 +466,21 @@ impl Store { Ok(()) } + fn ids_where(&self, condition: &str, bound: &[&dyn ToSql]) -> Result> { + let mut stmt = self + .db + .prepare(&format!("SELECT id FROM items WHERE {condition}"))?; + let rows = stmt.query_map(bound, |row| row.get(0))?; + Ok(rows.collect::>()?) + } + + fn erase_all(&self, ids: &[i64], at: i64) -> Result { + for id in ids { + self.erase(*id, at)?; + } + Ok(ids.len()) + } + fn blobs_of(&self, id: i64) -> Result> { let mut stmt = self.db.prepare( "SELECT blob_path FROM item_formats WHERE item_id = ?1 AND blob_path IS NOT NULL", @@ -353,6 +550,11 @@ impl Store { params![uuid, kind, preview, created_at, hash, fold(preview)], )?; let id = self.db.last_insert_rowid(); + self.write_formats(id, item)?; + Ok(id) + } + + fn write_formats(&self, id: i64, item: &Item) -> Result<()> { for format in &item.formats { let (inline, blob, size) = match &format.payload { Payload::Inline(bytes) => (Some(bytes.clone()), None, Some(bytes.len() as i64)), @@ -373,7 +575,45 @@ impl Store { params![id, format.id, size, inline, blob], )?; } - Ok(id) + Ok(()) + } + + pub fn update_text(&self, id: i64, text: &str, at: i64) -> Result<()> { + let payload = Payload::stored(text.as_bytes().to_vec()); + if let Payload::TooBig { size } = payload { + return Err(Error::TooBig { size }); + } + let edited = Item { + kind: Some(cp_core::kind::classify_text(text)), + formats: vec![cp_core::item::Format { + id: cp_core::item::SYNTHETIC_TEXT.into(), + payload, + }], + }; + if self.blobs.is_none() && edited.needs_blob_store() { + let (format, size) = edited.oversized_format().expect("lo acaba de decir"); + return Err(Error::NeedsBlobStore { format, size }); + } + let changed = self.db.execute( + "UPDATE items + SET kind = ?2, preview_text = ?3, search_text = ?4, search_ocr = '', + content_hash = ?5, thumb_path = NULL, broken_since = NULL, updated_at = ?6 + WHERE id = ?1 AND deleted_at IS NULL", + params![ + id, + edited.kind.map(|kind| kind.as_str()), + text, + fold(text), + edited.fingerprint() as i64, + at + ], + )?; + if changed == 0 { + return Err(Error::NoSuchItem { id }); + } + self.release(id)?; + self.write_formats(id, &edited)?; + self.checkpoint() } pub fn find_by_hash(&self, item: &Item) -> Result> { @@ -405,11 +645,22 @@ impl Store { } pub fn purge_broken_before(&self, cutoff: i64) -> Result { - Ok(self.db.execute( - "DELETE FROM items - WHERE broken_since IS NOT NULL AND broken_since < ?1 AND pinned = 0", - [cutoff], - )?) + let doomed = self.ids_where( + "broken_since IS NOT NULL AND broken_since < ?1 AND pinned = 0", + &[&cutoff], + )?; + for id in &doomed { + self.release(*id)?; + self.db.execute("DELETE FROM items WHERE id = ?1", [id])?; + } + self.checkpoint()?; + Ok(doomed.len()) + } + + pub fn mark_present(&self, id: i64) -> Result<()> { + self.db + .execute("UPDATE items SET broken_since = NULL WHERE id = ?1", [id])?; + Ok(()) } pub fn pin(&self, id: i64) -> Result<()> { @@ -426,105 +677,220 @@ impl Store { )?) } - pub fn list(&self, filter: &Filter, limit: usize, after: Option) -> Result> { - let expression = filter - .query - .as_deref() - .map(|text| fts_expression(&fold(text))); - if matches!(expression, Some(None)) { - return Ok(Vec::new()); - } - let expression = expression.flatten(); - - let mut sql = String::from( - "SELECT items.id, items.modified_at, items.kind, items.preview_text, items.pinned - FROM items ", - ); - if expression.is_some() { - sql.push_str("JOIN items_fts ON items.id = items_fts.rowid "); - } - sql.push_str("WHERE items.deleted_at IS NULL "); - if expression.is_some() { - sql.push_str("AND items_fts MATCH :match "); - } - if !filter.kinds.is_empty() { - let list = filter - .kinds - .iter() - .map(|kind| format!("'{}'", kind.as_str())) - .collect::>() - .join(", "); - sql.push_str(&format!("AND items.kind IN ({list}) ")); - } - if !filter.colors.is_empty() { - let list = filter - .colors - .iter() - .map(i64::to_string) - .collect::>() - .join(", "); - sql.push_str(&format!("AND items.card_color IN ({list}) ")); - } - if filter.pinned_only { - sql.push_str("AND items.pinned = 1 "); + pub fn list(&self, filter: &Filter, limit: usize, after: Option) -> Result { + let Some(mut clauses) = Clauses::of(filter, true) else { + return Ok(Page::default()); + }; + let key = filter.order.key(); + if let Some(cursor) = after { + clauses + .conditions + .push(format!("({key} < ? OR ({key} = ? AND items.id < ?))")); + clauses.bound.push(Box::new(cursor.key)); + clauses.bound.push(Box::new(cursor.key)); + clauses.bound.push(Box::new(cursor.id)); } - sql.push_str( - "AND (:after IS NULL OR items.modified_at < :after) - ORDER BY items.modified_at DESC LIMIT :limit", + let sql = format!( + "SELECT items.id, items.modified_at, items.created_at, items.kind, + items.preview_text, items.app_source, items.label, items.card_color, + items.thumb_path, items.paste_count, items.last_used_at, + items.broken_since, items.pinned, items.search_ocr, {key} + {} ORDER BY {key} DESC, items.id DESC LIMIT ?", + clauses.source() ); + let fetch = i64::try_from(limit).map_or(i64::MAX, |limit| limit.saturating_add(1)); + clauses.bound.push(Box::new(fetch)); + let terms = filter.query.as_deref().map(terms_of).unwrap_or_default(); let mut stmt = self.db.prepare(&sql)?; - let limit = limit as i64; - let mut bound: Vec<(&str, &dyn rusqlite::ToSql)> = - vec![(":after", &after), (":limit", &limit)]; - if let Some(text) = &expression { - bound.push((":match", text)); - } - let rows = stmt.query_map(bound.as_slice(), |row| { - Ok(Listed { + let rows = stmt.query_map(params_from_iter(clauses.bound.iter()), |row| { + let ocr: String = row.get(13)?; + let mut listed = Listed { id: row.get(0)?, modified_at: row.get(1)?, - kind: row.get(2)?, - preview: row.get(3)?, - pinned: row.get::<_, i64>(4)? == 1, + created_at: row.get(2)?, + kind: row + .get::<_, Option>(3)? + .and_then(|name| Kind::from_name(&name)), + preview: row.get(4)?, + app: row.get(5)?, + label: row.get(6)?, + color: row.get(7)?, + thumb_path: row.get(8)?, + paste_count: row.get(9)?, + last_used_at: row.get(10)?, + broken_since: row.get(11)?, + pinned: row.get::<_, i64>(12)? == 1, + snippet: None, + }; + listed.snippet = snippet_of(&listed, &ocr, &terms); + Ok((listed, row.get::<_, i64>(14)?)) + })?; + let mut keyed: Vec<(Listed, i64)> = rows.collect::>()?; + let more = keyed.len() > limit; + keyed.truncate(limit); + let next = match keyed.last() { + Some((last, key)) if more => Some(Cursor { + key: *key, + id: last.id, + }), + _ => None, + }; + Ok(Page { + rows: keyed.into_iter().map(|(listed, _)| listed).collect(), + next, + }) + } + + pub fn facets(&self, filter: &Filter) -> Result> { + let Some(mut clauses) = Clauses::of(filter, false) else { + return Ok(Vec::new()); + }; + clauses.conditions.push("items.kind IS NOT NULL".into()); + let sql = format!( + "SELECT items.kind, COUNT(*) {} GROUP BY items.kind + ORDER BY COUNT(*) DESC, items.kind", + clauses.source() + ); + let mut stmt = self.db.prepare(&sql)?; + let rows = stmt.query_map(params_from_iter(clauses.bound.iter()), |row| { + Ok((row.get::<_, String>(0)?, row.get::<_, i64>(1)?)) + })?; + let mut facets = Vec::new(); + for row in rows { + let (name, count) = row?; + if let Some(kind) = Kind::from_name(&name) { + facets.push(Facet { kind, count }); + } + } + Ok(facets) + } + + pub fn distinct_apps(&self) -> Result> { + let mut stmt = self.db.prepare( + "SELECT MIN(app_source), COUNT(*) FROM items + WHERE deleted_at IS NULL AND app_source IS NOT NULL + GROUP BY search_app + ORDER BY COUNT(*) DESC, MIN(app_source)", + )?; + let rows = stmt.query_map([], |row| { + Ok(AppCount { + app: row.get(0)?, + count: row.get(1)?, }) })?; Ok(rows.collect::>()?) } pub fn clear_older_than(&self, cutoff: i64) -> Result { - let doomed: Vec = { - let mut stmt = self.db.prepare( - "SELECT id FROM items - WHERE modified_at < ?1 AND pinned = 0 AND deleted_at IS NULL", - )?; - let rows = stmt.query_map([cutoff], |row| row.get(0))?; - rows.collect::>()? - }; - for id in &doomed { - self.erase(*id, cutoff)?; - } + let doomed = self.ids_where( + "modified_at < ?1 AND pinned = 0 AND deleted_at IS NULL", + &[&cutoff], + )?; + let removed = self.erase_all(&doomed, cutoff)?; self.checkpoint()?; - Ok(doomed.len()) + Ok(removed) } pub fn clear_all_unpinned(&self, at: i64) -> Result { - self.clear_older_than_matching(at, "pinned = 0") + let doomed = self.ids_where("pinned = 0 AND deleted_at IS NULL", &[])?; + let removed = self.erase_all(&doomed, at)?; + self.checkpoint()?; + Ok(removed) } - fn clear_older_than_matching(&self, at: i64, condition: &str) -> Result { - let doomed: Vec = { - let mut stmt = self.db.prepare(&format!( - "SELECT id FROM items WHERE {condition} AND deleted_at IS NULL" - ))?; - let rows = stmt.query_map([], |row| row.get(0))?; - rows.collect::>()? - }; - for id in &doomed { - self.erase(*id, at)?; + pub fn usage(&self) -> Result { + let items = self.count()?; + let inline: i64 = self.db.query_row( + "SELECT COALESCE(SUM(LENGTH(inline_data)), 0) FROM item_formats", + [], + |row| row.get(0), + )?; + let blobs: i64 = self.db.query_row( + "SELECT COALESCE(SUM(size_bytes), 0) FROM ( + SELECT DISTINCT blob_path, size_bytes FROM item_formats + WHERE blob_path IS NOT NULL)", + [], + |row| row.get(0), + )?; + Ok(Usage { + items, + bytes: inline + blobs, + }) + } + + pub fn sweep(&self, policy: &Policy, now: i64) -> Result { + let mut swept = Swept::default(); + if let Some(grace) = policy.broken_for { + swept.broken = self.purge_broken_before(now - grace)?; + } + if let Some(age) = policy.keep_for { + let doomed = self.ids_where( + "modified_at < ?1 AND pinned = 0 AND deleted_at IS NULL", + &[&(now - age)], + )?; + swept.expired = self.erase_all(&doomed, now)?; + } + if let Some(keep) = policy.keep_at_most { + let excess = (self.count()? - keep).max(0); + let doomed = self.ids_where( + "pinned = 0 AND deleted_at IS NULL ORDER BY modified_at, id LIMIT ?1", + &[&excess], + )?; + swept.over_count = self.erase_all(&doomed, now)?; + } + if let Some(limit) = policy.bytes_at_most { + swept.over_bytes = self.evict_until_under(limit, now)?; + } + if let Some(blobs) = &self.blobs { + let referenced = self.referenced_blobs()?; + swept.orphans = blobs.sweep(&|digest| referenced.contains(digest))?; } self.checkpoint()?; - Ok(doomed.len()) + Ok(swept) + } + + fn evict_until_under(&self, limit: i64, at: i64) -> Result { + let mut evicted = 0; + let mut left = usize::MAX; + loop { + let usage = self.usage()?.bytes; + let candidates = self.eviction_candidates()?; + if usage <= limit || candidates.len() >= left { + return Ok(evicted); + } + left = candidates.len(); + let mut freed = 0; + for (id, bytes) in candidates { + self.erase(id, at)?; + evicted += 1; + freed += bytes; + if freed >= usage - limit { + break; + } + } + } + } + + fn eviction_candidates(&self) -> Result> { + let mut stmt = self.db.prepare( + "SELECT items.id, COALESCE(SUM(COALESCE(LENGTH(f.inline_data), f.size_bytes, 0)), 0) + FROM items LEFT JOIN item_formats f + ON f.item_id = items.id AND (f.inline_data IS NOT NULL OR f.blob_path IS NOT NULL) + WHERE items.pinned = 0 AND items.deleted_at IS NULL + GROUP BY items.id + ORDER BY items.modified_at, items.id", + )?; + let rows = stmt.query_map([], |row| Ok((row.get(0)?, row.get(1)?)))?; + Ok(rows.collect::>()?) + } + + fn referenced_blobs(&self) -> Result> { + let mut stmt = self + .db + .prepare("SELECT DISTINCT blob_path FROM item_formats WHERE blob_path IS NOT NULL")?; + let rows = stmt.query_map([], |row| row.get::<_, String>(0))?; + Ok(rows.collect::>()?) } pub const PAGE: usize = 100; @@ -539,7 +905,7 @@ impl Store { limit: usize, after: Option, ) -> Result> { - let Some(expression) = fts_expression(&fold(query)) else { + let Some(expression) = fts_expression(query) else { return Ok(Vec::new()); }; let mut stmt = self.db.prepare( @@ -559,7 +925,7 @@ impl Store { } pub fn search_page(&self, query: &str, limit: usize, offset: usize) -> Result> { - let Some(expression) = fts_expression(&fold(query)) else { + let Some(expression) = fts_expression(query) else { return Ok(Vec::new()); }; let mut stmt = self.db.prepare( @@ -578,6 +944,45 @@ impl Store { } } +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct Policy { + pub keep_for: Option, + pub keep_at_most: Option, + pub bytes_at_most: Option, + pub broken_for: Option, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct Swept { + pub broken: usize, + pub expired: usize, + pub over_count: usize, + pub over_bytes: usize, + pub orphans: usize, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Usage { + pub items: i64, + pub bytes: i64, +} + +fn snippet_of(listed: &Listed, ocr: &str, terms: &[String]) -> Option { + if terms.is_empty() { + return None; + } + let sources = [ + (Where::Text, Some(listed.preview.as_str())), + (Where::Label, listed.label.as_deref()), + (Where::App, listed.app.as_deref()), + (Where::Ocr, Some(ocr)), + ]; + sources.into_iter().find_map(|(found_in, text)| { + let excerpt = excerpt(text?, terms, EXCERPT_CHARS)?; + Some(Snippet { found_in, excerpt }) + }) +} + #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Restricted { Mode(u32), @@ -1257,10 +1662,16 @@ mod tests { #[test] fn pasting_does_not_move_the_item_up_the_list() { let store = a_little_history(); - let listed = store.list(&Filter::default(), 10, None).expect("listado"); + let listed = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; let oldest = listed.last().expect("hay").id; store.record_paste(oldest, 999).expect("pega"); - let after = store.list(&Filter::default(), 10, None).expect("listado"); + let after = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; assert_eq!( after.last().map(|one| one.id), Some(oldest), @@ -1271,13 +1682,16 @@ mod tests { #[test] fn the_colour_can_be_set_and_filtered_by() { let store = a_little_history(); - let listed = store.list(&Filter::default(), 10, None).expect("listado"); + let listed = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; store.set_color(listed[0].id, 3, 100).expect("color"); let filter = Filter { colors: vec![3], ..Default::default() }; - let coloured = store.list(&filter, 10, None).expect("listado"); + let coloured = store.list(&filter, 10, None).expect("listado").rows; assert_eq!(coloured.len(), 1); assert_eq!(coloured[0].id, listed[0].id); } @@ -1908,7 +2322,10 @@ mod tests { #[test] fn the_panel_can_ask_for_the_latest_without_searching_anything() { let store = a_little_history(); - let listed = store.list(&Filter::default(), 10, None).expect("listado"); + let listed = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; assert_eq!(listed.len(), 4, "sin término se devuelve el historial"); assert_eq!( listed.first().map(|one| one.preview.as_str()), @@ -1924,9 +2341,9 @@ mod tests { kinds: vec![Kind::Text], ..Default::default() }; - let listed = store.list(&filter, 10, None).expect("listado"); + let listed = store.list(&filter, 10, None).expect("listado").rows; assert_eq!(listed.len(), 2); - assert!(listed.iter().all(|one| one.kind.as_deref() == Some("text"))); + assert!(listed.iter().all(|one| one.kind == Some(Kind::Text))); } #[test] @@ -1936,7 +2353,10 @@ mod tests { kinds: vec![Kind::Email, Kind::Color], ..Default::default() }; - assert_eq!(store.list(&filter, 10, None).expect("listado").len(), 2); + assert_eq!( + store.list(&filter, 10, None).expect("listado").rows.len(), + 2 + ); } #[test] @@ -1947,7 +2367,10 @@ mod tests { kinds: vec![Kind::Text], ..Default::default() }; - assert_eq!(store.list(&filter, 10, None).expect("listado").len(), 2); + assert_eq!( + store.list(&filter, 10, None).expect("listado").rows.len(), + 2 + ); let narrower = Filter { query: Some("nota".into()), @@ -1955,7 +2378,11 @@ mod tests { ..Default::default() }; assert!( - store.list(&narrower, 10, None).expect("listado").is_empty(), + store + .list(&narrower, 10, None) + .expect("listado") + .rows + .is_empty(), "el filtro y el término se aplican los dos" ); } @@ -1963,32 +2390,41 @@ mod tests { #[test] fn only_pinned_can_be_asked_for() { let store = a_little_history(); - let listed = store.list(&Filter::default(), 10, None).expect("listado"); + let listed = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; let id = listed.first().expect("hay").id; store.pin(id).expect("fija"); let filter = Filter { pinned_only: true, ..Default::default() }; - let pinned = store.list(&filter, 10, None).expect("listado"); + let pinned = store.list(&filter, 10, None).expect("listado").rows; assert_eq!(pinned.len(), 1); assert!(pinned[0].pinned); } #[test] - fn the_list_pages_with_the_same_cursor_as_the_search() { + fn the_list_hands_out_a_cursor_only_while_there_is_more() { let store = a_little_history(); let first = store.list(&Filter::default(), 2, None).expect("página"); - assert_eq!(first.len(), 2); - let next = store - .list( - &Filter::default(), - 2, - first.last().map(|one| one.modified_at), - ) + assert_eq!(first.rows.len(), 2); + let cursor = first.next.expect("quedan dos más"); + let second = store + .list(&Filter::default(), 2, Some(cursor)) .expect("siguiente"); - assert_eq!(next.len(), 2); - assert!(next.iter().all(|one| !first.contains(one))); + assert_eq!(second.rows.len(), 2); + assert!(second.rows.iter().all(|one| !first.rows.contains(one))); + assert_eq!(second.next, None, "la última página no promete otra"); + } + + #[test] + fn a_page_that_ends_exactly_at_the_last_row_promises_nothing_more() { + let store = a_little_history(); + let whole = store.list(&Filter::default(), 4, None).expect("página"); + assert_eq!(whole.rows.len(), 4); + assert_eq!(whole.next, None); } #[test] @@ -1999,7 +2435,11 @@ mod tests { ..Default::default() }; assert!( - store.list(&filter, 10, None).expect("listado").is_empty(), + store + .list(&filter, 10, None) + .expect("listado") + .rows + .is_empty(), "pedir buscar algo imposible no puede devolver el historial entero" ); } @@ -2007,12 +2447,16 @@ mod tests { #[test] fn deleted_items_never_show_up_in_the_list() { let store = a_little_history(); - let listed = store.list(&Filter::default(), 10, None).expect("listado"); + let listed = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; store.mark_deleted(listed[0].id, 99).expect("borra"); assert_eq!( store .list(&Filter::default(), 10, None) .expect("listado") + .rows .len(), 3 ); @@ -2021,7 +2465,10 @@ mod tests { #[test] fn retention_takes_the_old_and_leaves_what_was_pinned() { let store = a_little_history(); - let listed = store.list(&Filter::default(), 10, None).expect("listado"); + let listed = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; let oldest = listed.last().expect("hay").id; store.pin(oldest).expect("fija el más viejo"); @@ -2030,7 +2477,10 @@ mod tests { removed, 2, "se van los de antes del corte que no estén fijados" ); - let left = store.list(&Filter::default(), 10, None).expect("listado"); + let left = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; assert_eq!(left.len(), 2); assert!( left.iter().any(|one| one.id == oldest), @@ -2041,7 +2491,10 @@ mod tests { #[test] fn clearing_everything_still_respects_what_was_pinned() { let store = a_little_history(); - let listed = store.list(&Filter::default(), 10, None).expect("listado"); + let listed = store + .list(&Filter::default(), 10, None) + .expect("listado") + .rows; store.pin(listed[0].id).expect("fija"); let removed = store.clear_all_unpinned(100).expect("vacía"); assert_eq!(removed, 3); @@ -2120,3 +2573,1110 @@ mod identity { ); } } + +#[cfg(test)] +mod listing { + use super::*; + use cp_core::item::Format; + + fn text_item(text: &str, kind: Kind) -> Item { + Item { + kind: Some(kind), + formats: vec![Format { + id: "public.utf8-plain-text".into(), + payload: Payload::Inline(text.as_bytes().to_vec()), + }], + } + } + + fn history() -> Store { + let store = Store::in_memory().expect("esquema"); + let rows = [ + ("uuid-1", "primera nota", Kind::Text, 10, "Safari"), + ("uuid-2", "alguien@ejemplo.test", Kind::Email, 20, "Slack"), + ("uuid-3", "#FF8800", Kind::Color, 30, "Slack"), + ("uuid-4", "segunda nota", Kind::Text, 40, "Code"), + ("uuid-5", "fn main() {}", Kind::Code, 50, "Code"), + ]; + for (uuid, text, kind, at, app) in rows { + let id = store + .insert_item(uuid, &text_item(text, kind), text, at) + .expect("insert"); + store.set_source(id, app, at).expect("origen"); + } + store + } + + fn all(store: &Store, filter: &Filter) -> Vec { + store.list(filter, 100, None).expect("listado").rows + } + + fn previews(rows: &[Listed]) -> Vec<&str> { + rows.iter().map(|one| one.preview.as_str()).collect() + } + + #[test] + fn the_card_gets_everything_the_row_knows() { + let store = history(); + let id = all(&store, &Filter::default())[0].id; + store.set_label(id, Some("Arranque"), 60).expect("etiqueta"); + store.set_color(id, 5, 61).expect("color"); + store.record_paste(id, 62).expect("pega"); + store.pin(id).expect("fija"); + let card = all(&store, &Filter::default()) + .into_iter() + .find(|one| one.id == id) + .expect("está"); + assert_eq!(card.preview, "fn main() {}"); + assert_eq!(card.kind, Some(Kind::Code)); + assert_eq!(card.app.as_deref(), Some("Code")); + assert_eq!(card.label.as_deref(), Some("Arranque")); + assert_eq!(card.color, 5); + assert_eq!(card.paste_count, 1); + assert_eq!(card.last_used_at, Some(62)); + assert_eq!(card.created_at, 50); + assert_eq!(card.modified_at, 50, "pegar no lo mueve"); + assert!(card.pinned); + assert_eq!(card.broken_since, None); + assert_eq!(card.thumb_path, None); + assert_eq!(card.snippet, None, "sin término no hay fragmento"); + } + + #[test] + fn searching_marks_the_fragment_that_matched() { + let store = history(); + let filter = Filter { + query: Some("segun".into()), + ..Default::default() + }; + let rows = all(&store, &filter); + assert_eq!(rows.len(), 1); + let snippet = rows[0].snippet.as_ref().expect("fragmento"); + assert_eq!(snippet.found_in, Where::Text); + let marked: Vec<&str> = snippet + .excerpt + .segments + .iter() + .filter(|one| one.matched) + .map(|one| one.text.as_str()) + .collect(); + assert_eq!(marked, vec!["segun"]); + assert_eq!(snippet.excerpt.plain(), "segunda nota"); + } + + #[test] + fn a_hit_on_the_label_says_so() { + let store = history(); + let id = all(&store, &Filter::default())[0].id; + store + .set_label(id, Some("Factura mayo"), 60) + .expect("etiqueta"); + let filter = Filter { + query: Some("factura".into()), + ..Default::default() + }; + let rows = all(&store, &filter); + assert_eq!(rows.len(), 1); + let snippet = rows[0].snippet.as_ref().expect("fragmento"); + assert_eq!(snippet.found_in, Where::Label); + assert_eq!(snippet.excerpt.plain(), "Factura mayo"); + } + + #[test] + fn a_hit_on_what_was_read_inside_an_image_says_so() { + let store = Store::in_memory().expect("esquema"); + let image = Item { + kind: Some(Kind::Image), + formats: vec![Format { + id: "public.png".into(), + payload: Payload::Inline(vec![1]), + }], + }; + let id = store + .insert_item("uuid-img", &image, "", 1) + .expect("insert"); + store + .set_ocr_text(id, "Pedido AB-4417 entrega", 2) + .expect("ocr"); + let filter = Filter { + query: Some("ab-4417".into()), + ..Default::default() + }; + let rows = all(&store, &filter); + assert_eq!(rows.len(), 1); + let snippet = rows[0].snippet.as_ref().expect("fragmento"); + assert_eq!(snippet.found_in, Where::Ocr); + assert!(snippet.excerpt.plain().contains("ab-4417")); + } + + #[test] + fn a_hit_on_the_source_application_says_so() { + let store = history(); + let filter = Filter { + query: Some("slack".into()), + ..Default::default() + }; + let rows = all(&store, &filter); + assert_eq!(rows.len(), 2); + assert!( + rows.iter() + .all(|one| one.snippet.as_ref().map(|s| s.found_in) == Some(Where::App)) + ); + } + + #[test] + fn a_class_can_be_left_out() { + let store = history(); + let filter = Filter { + exclude_kinds: vec![Kind::Text, Kind::Code], + ..Default::default() + }; + assert_eq!( + previews(&all(&store, &filter)), + vec!["#FF8800", "alguien@ejemplo.test"] + ); + } + + #[test] + fn the_source_application_filters_by_equality_not_by_search() { + let store = history(); + let id = all(&store, &Filter::default())[0].id; + store + .set_label(id, Some("pegar en slack"), 60) + .expect("etiqueta"); + let filter = Filter { + apps: vec!["slack".into()], + ..Default::default() + }; + let rows = all(&store, &filter); + assert_eq!(rows.len(), 2, "solo lo copiado desde Slack, sin mayúsculas"); + assert!(rows.iter().all(|one| one.app.as_deref() == Some("Slack"))); + } + + #[test] + fn several_applications_and_a_negated_one() { + let store = history(); + let either = Filter { + apps: vec!["Safari".into(), "Code".into()], + ..Default::default() + }; + assert_eq!(all(&store, &either).len(), 3); + let not_code = Filter { + exclude_apps: vec!["code".into()], + ..Default::default() + }; + assert_eq!(all(&store, ¬_code).len(), 3); + } + + #[test] + fn since_keeps_what_was_copied_from_that_moment_on() { + let store = history(); + let filter = Filter { + since: Some(30), + ..Default::default() + }; + assert_eq!(all(&store, &filter).len(), 3, "el 30 incluido"); + } + + #[test] + fn since_counts_a_recopy_as_copied_again() { + let store = history(); + let oldest = all(&store, &Filter::default()).last().expect("hay").id; + store.reactivate(oldest, 100).expect("recopiado"); + let filter = Filter { + since: Some(100), + ..Default::default() + }; + assert_eq!( + previews(&all(&store, &filter)), + vec!["primera nota"], + "lo que se vuelve a copiar hoy es de hoy" + ); + } + + #[test] + fn broken_items_are_hidden_unless_asked_for() { + let store = history(); + let id = all(&store, &Filter::default())[0].id; + store.mark_broken(id, 99).expect("roto"); + assert_eq!(all(&store, &Filter::default()).len(), 4); + let shown = Filter { + broken: Broken::Shown, + ..Default::default() + }; + assert_eq!(all(&store, &shown).len(), 5); + let only = Filter { + broken: Broken::Only, + ..Default::default() + }; + let rows = all(&store, &only); + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].broken_since, Some(99)); + } + + #[test] + fn a_scoped_label_query_does_not_match_the_content() { + let store = history(); + let rows = all(&store, &Filter::default()); + store + .set_label(rows[1].id, Some("nota"), 60) + .expect("etiqueta"); + let by_label = Filter { + label: Some("nota".into()), + ..Default::default() + }; + let found = all(&store, &by_label); + assert_eq!( + found.len(), + 1, + "«nota» está en dos contenidos y una etiqueta" + ); + assert_eq!(found[0].id, rows[1].id); + } + + #[test] + fn a_label_query_and_a_text_query_both_apply() { + let store = history(); + let rows = all(&store, &Filter::default()); + store + .set_label(rows[0].id, Some("arranque"), 60) + .expect("etiqueta"); + store + .set_label(rows[1].id, Some("arranque"), 61) + .expect("etiqueta"); + let filter = Filter { + query: Some("main".into()), + label: Some("arranque".into()), + ..Default::default() + }; + assert_eq!(previews(&all(&store, &filter)), vec!["fn main() {}"]); + } + + #[test] + fn a_label_query_with_nothing_usable_finds_nothing() { + let store = history(); + let filter = Filter { + label: Some("!!!".into()), + ..Default::default() + }; + assert!(all(&store, &filter).is_empty()); + } + + #[test] + fn most_pasted_comes_first_and_ties_break_the_same_way_every_time() { + let store = history(); + let rows = all(&store, &Filter::default()); + for _ in 0..3 { + store.record_paste(rows[4].id, 70).expect("pega"); + } + store.record_paste(rows[2].id, 71).expect("pega"); + let filter = Filter { + order: Order::MostPasted, + ..Default::default() + }; + let ordered = all(&store, &filter); + assert_eq!(ordered[0].id, rows[4].id); + assert_eq!(ordered[1].id, rows[2].id); + assert_eq!( + ordered[2..].iter().map(|one| one.id).collect::>(), + vec![rows[0].id, rows[1].id, rows[3].id], + "a igual cuenta, el más nuevo primero" + ); + } + + #[test] + fn last_used_puts_what_was_never_pasted_at_the_end() { + let store = history(); + let rows = all(&store, &Filter::default()); + store.record_paste(rows[3].id, 80).expect("pega"); + store.record_paste(rows[1].id, 90).expect("pega"); + let filter = Filter { + order: Order::LastUsed, + ..Default::default() + }; + let ordered = all(&store, &filter); + assert_eq!(ordered[0].id, rows[1].id); + assert_eq!(ordered[1].id, rows[3].id); + assert!(ordered[2..].iter().all(|one| one.last_used_at.is_none())); + } + + fn walk(store: &Store, filter: &Filter, page: usize) -> Vec { + let mut seen = Vec::new(); + let mut after = None; + loop { + let got = store.list(filter, page, after).expect("página"); + seen.extend(got.rows.iter().map(|one| one.id)); + match got.next { + Some(cursor) => after = Some(cursor), + None => return seen, + } + } + } + + #[test] + fn every_order_pages_without_repeating_or_skipping_even_with_ties() { + let store = Store::in_memory().expect("esquema"); + for at in 0..23 { + let id = store + .insert_item( + &format!("uuid-{at}"), + &text_item(&format!("nota {at}"), Kind::Text), + &format!("nota {at}"), + at % 4, + ) + .expect("insert"); + for _ in 0..(at % 3) { + store.record_paste(id, at % 5).expect("pega"); + } + } + for order in [Order::Recent, Order::MostPasted, Order::LastUsed] { + let filter = Filter { + order, + ..Default::default() + }; + let mut ids = walk(&store, &filter, 4); + assert_eq!(ids.len(), 23, "{order:?} se saltó filas"); + ids.sort_unstable(); + ids.dedup(); + assert_eq!(ids.len(), 23, "{order:?} repitió filas"); + } + } + + #[test] + fn the_tabs_count_only_the_classes_that_exist_within_the_search() { + let store = history(); + let facets = store.facets(&Filter::default()).expect("facetas"); + assert_eq!( + facets, + vec![ + Facet { + kind: Kind::Text, + count: 2 + }, + Facet { + kind: Kind::Code, + count: 1 + }, + Facet { + kind: Kind::Color, + count: 1 + }, + Facet { + kind: Kind::Email, + count: 1 + }, + ], + "por cantidad, y a igual cantidad por nombre" + ); + let within = Filter { + apps: vec!["Slack".into()], + kinds: vec![Kind::Text], + ..Default::default() + }; + let facets = store.facets(&within).expect("facetas"); + assert_eq!( + facets.iter().map(|one| one.kind).collect::>(), + vec![Kind::Color, Kind::Email], + "la pestaña elegida no recorta las demás; la app y el término sí" + ); + } + + #[test] + fn an_item_without_a_class_has_no_tab() { + let store = Store::in_memory().expect("esquema"); + store + .insert_item( + "uuid-none", + &Item { + kind: None, + formats: vec![], + }, + "", + 1, + ) + .expect("insert"); + assert!( + store + .facets(&Filter::default()) + .expect("facetas") + .is_empty() + ); + } + + #[test] + fn an_impossible_search_has_no_tabs_either() { + let store = history(); + let filter = Filter { + query: Some("!!!".into()), + ..Default::default() + }; + assert!(store.facets(&filter).expect("facetas").is_empty()); + } + + #[test] + fn the_applications_come_with_their_counts_most_used_first() { + let store = history(); + let apps = store.distinct_apps().expect("apps"); + assert_eq!( + apps, + vec![ + AppCount { + app: "Code".into(), + count: 2 + }, + AppCount { + app: "Slack".into(), + count: 2 + }, + AppCount { + app: "Safari".into(), + count: 1 + }, + ] + ); + } + + #[test] + fn two_spellings_of_one_application_are_one_entry() { + let store = history(); + let id = all(&store, &Filter::default())[0].id; + store.set_source(id, "slack", 60).expect("origen"); + let apps = store.distinct_apps().expect("apps"); + let slack = apps.iter().find(|one| one.app == "Slack").expect("está"); + assert_eq!(slack.count, 3); + assert!(apps.iter().all(|one| one.app != "slack")); + } + + #[test] + fn deleted_items_count_for_nothing() { + let store = history(); + let id = all(&store, &Filter::default())[0].id; + store.mark_deleted(id, 99).expect("borra"); + let facets = store.facets(&Filter::default()).expect("facetas"); + assert!(facets.iter().all(|one| one.kind != Kind::Code)); + let apps = store.distinct_apps().expect("apps"); + assert_eq!( + apps.iter() + .find(|one| one.app == "Code") + .map(|one| one.count), + Some(1) + ); + } + + #[test] + fn no_order_sorts_the_history_in_memory() { + let store = history(); + for order in [Order::Recent, Order::MostPasted, Order::LastUsed] { + let key = order.key(); + let plan: Vec = store + .db + .prepare(&format!( + "EXPLAIN QUERY PLAN SELECT items.id FROM items + WHERE items.deleted_at IS NULL AND items.broken_since IS NULL + ORDER BY {key} DESC, items.id DESC LIMIT 50" + )) + .expect("prepara") + .query_map([], |row| row.get::<_, String>(3)) + .expect("plan") + .map(|row| row.expect("fila")) + .collect(); + assert!( + !plan.iter().any(|step| step.contains("TEMP B-TREE")), + "{order:?} ordena en memoria: {plan:?}" + ); + } + } + + #[test] + fn leaving_a_class_out_keeps_what_has_no_class() { + let store = history(); + store + .insert_item( + "uuid-none", + &Item { + kind: None, + formats: vec![], + }, + "sin clase", + 60, + ) + .expect("insert"); + let filter = Filter { + exclude_kinds: vec![Kind::Image], + ..Default::default() + }; + assert_eq!( + all(&store, &filter).len(), + 6, + "excluir imágenes no puede esconder lo que no es nada" + ); + } + + #[test] + fn an_excluded_class_has_no_tab() { + let store = history(); + let filter = Filter { + exclude_kinds: vec![Kind::Text], + ..Default::default() + }; + let facets = store.facets(&filter).expect("facetas"); + assert!(facets.iter().all(|one| one.kind != Kind::Text)); + assert_eq!(facets.len(), 3); + } + + #[test] + fn asking_for_no_rows_is_an_empty_page_not_a_panic() { + let store = history(); + let page = store.list(&Filter::default(), 0, None).expect("página"); + assert!(page.rows.is_empty()); + assert_eq!(page.next, None); + let huge = store + .list(&Filter::default(), usize::MAX, None) + .expect("página"); + assert_eq!(huge.rows.len(), 5); + assert_eq!(huge.next, None); + } + + #[test] + fn a_broken_item_can_be_found_again() { + let store = history(); + let id = all(&store, &Filter::default())[0].id; + store.mark_broken(id, 99).expect("roto"); + assert_eq!(all(&store, &Filter::default()).len(), 4); + store.mark_present(id).expect("vuelve"); + let rows = all(&store, &Filter::default()); + assert_eq!(rows.len(), 5, "el volumen se volvió a montar"); + assert_eq!(rows[0].broken_since, None); + assert_eq!( + store.purge_broken_before(i64::MAX).expect("purga"), + 0, + "y ya no está en el plazo de nadie" + ); + } + + #[test] + fn nothing_a_person_can_type_as_an_application_breaks_the_query() { + let store = history(); + for app in ["'; DROP TABLE items; --", "\"", "%", "Straße"] { + let filter = Filter { + apps: vec![app.into()], + ..Default::default() + }; + store + .list(&filter, 10, None) + .unwrap_or_else(|why| panic!("«{app}» rompió el listado: {why}")); + } + assert_eq!(store.count().expect("cuenta"), 5, "la tabla sigue ahí"); + } +} + +#[cfg(test)] +mod housekeeping { + use super::*; + use cp_core::item::Format; + + fn on_disk() -> (tempfile::TempDir, Store) { + let dir = tempfile::tempdir().expect("carpeta"); + let store = Store::open(&dir.path().join("history.db")).expect("abre"); + (dir, store) + } + + fn image(byte: u8, size: usize) -> Item { + Item { + kind: Some(Kind::Image), + formats: vec![Format { + id: "public.png".into(), + payload: Payload::stored(vec![byte; size]), + }], + } + } + + fn fill(store: &Store, count: i64) -> Vec { + (1..=count) + .map(|at| { + store + .insert_text(&format!("uuid-{at}"), &format!("nota {at}"), at) + .expect("insert") + }) + .collect() + } + + fn alive(store: &Store) -> Vec { + store + .list(&Filter::default(), 100, None) + .expect("listado") + .rows + .iter() + .map(|one| one.id) + .collect() + } + + #[test] + fn a_policy_with_nothing_set_sweeps_nothing() { + let store = Store::in_memory().expect("esquema"); + fill(&store, 5); + let swept = store.sweep(&Policy::default(), 100).expect("barre"); + assert_eq!(swept, Swept::default()); + assert_eq!(store.count().expect("cuenta"), 5); + } + + #[test] + fn age_takes_the_old_and_leaves_what_was_pinned() { + let store = Store::in_memory().expect("esquema"); + let ids = fill(&store, 5); + store.pin(ids[0]).expect("fija"); + let policy = Policy { + keep_for: Some(3), + ..Default::default() + }; + let swept = store.sweep(&policy, 6).expect("barre"); + assert_eq!(swept.expired, 1, "el de antes del 3 que no está fijado"); + assert_eq!(alive(&store), vec![ids[4], ids[3], ids[2], ids[0]]); + } + + #[test] + fn a_count_limit_evicts_the_oldest_unpinned_beyond_it() { + let store = Store::in_memory().expect("esquema"); + let ids = fill(&store, 6); + store.pin(ids[0]).expect("fija el más viejo"); + let policy = Policy { + keep_at_most: Some(4), + ..Default::default() + }; + let swept = store.sweep(&policy, 10).expect("barre"); + assert_eq!(swept.over_count, 2); + assert_eq!( + alive(&store), + vec![ids[5], ids[4], ids[3], ids[0]], + "el fijado cuenta para el límite pero no se va" + ); + assert_eq!(store.sweep(&policy, 11).expect("otra vez").over_count, 0); + } + + #[test] + fn a_count_limit_already_met_touches_nothing() { + let store = Store::in_memory().expect("esquema"); + fill(&store, 3); + for keep in [3, 5] { + let policy = Policy { + keep_at_most: Some(keep), + ..Default::default() + }; + assert_eq!(store.sweep(&policy, 10).expect("barre").over_count, 0); + assert_eq!(store.count().expect("cuenta"), 3, "con {keep} de límite"); + } + } + + #[test] + fn usage_counts_what_is_actually_kept_and_shared_bytes_once() { + let (_dir, store) = on_disk(); + let empty = store.usage().expect("uso"); + assert_eq!((empty.items, empty.bytes), (0, 0)); + store.insert_text("uuid-t", "hola", 1).expect("insert"); + store + .insert_item("uuid-a", &image(1, 200_000), "", 2) + .expect("a"); + store + .insert_item("uuid-b", &image(1, 200_000), "", 3) + .expect("b"); + let announced = Item { + kind: Some(Kind::Image), + formats: vec![Format { + id: "public.tiff".into(), + payload: Payload::Announced { + size: Some(4_000_000), + }, + }], + }; + store.insert_item("uuid-c", &announced, "", 4).expect("c"); + let usage = store.usage().expect("uso"); + assert_eq!(usage.items, 4); + assert_eq!( + usage.bytes, 200_000, + "el texto no tiene formatos guardados, la imagen compartida cuenta una vez, lo anunciado nada" + ); + } + + #[test] + fn a_byte_quota_evicts_the_oldest_until_it_fits() { + let (dir, store) = on_disk(); + for at in 1..=4 { + store + .insert_item( + &format!("uuid-{at}"), + &image(at as u8, 100_000), + "", + at as i64, + ) + .expect("insert"); + } + let policy = Policy { + bytes_at_most: Some(200_000), + ..Default::default() + }; + let swept = store.sweep(&policy, 10).expect("barre"); + assert_eq!(swept.over_bytes, 2, "quedar justo en la cuota es caber"); + assert_eq!(store.usage().expect("uso").bytes, 200_000); + assert_eq!(alive(&store).len(), 2); + let files = crate::blobs::files_under(&dir.path().join("blobs")).len(); + assert_eq!(files, 2, "los bytes desalojados se fueron del disco"); + } + + #[test] + fn a_quota_never_evicts_what_is_pinned_even_if_it_stays_over() { + let (_dir, store) = on_disk(); + let id = store + .insert_item("uuid-1", &image(1, 200_000), "", 1) + .expect("insert"); + store.pin(id).expect("fija"); + let policy = Policy { + bytes_at_most: Some(1_000), + ..Default::default() + }; + let swept = store.sweep(&policy, 10).expect("barre"); + assert_eq!(swept.over_bytes, 0); + assert_eq!(store.count().expect("cuenta"), 1); + } + + #[test] + fn a_shared_blob_is_freed_only_when_its_last_owner_goes() { + let (_dir, store) = on_disk(); + store + .insert_item("uuid-1", &image(7, 100_000), "", 1) + .expect("a"); + store + .insert_item("uuid-2", &image(7, 100_000), "", 2) + .expect("b"); + let policy = Policy { + bytes_at_most: Some(50_000), + ..Default::default() + }; + let swept = store.sweep(&policy, 10).expect("barre"); + assert_eq!( + swept.over_bytes, 2, + "borrar el primero no libera nada, así que sigue con el segundo" + ); + assert_eq!(store.usage().expect("uso").bytes, 0); + } + + #[test] + fn purging_on_its_own_leaves_nothing_in_the_log_either() { + let dir = tempfile::tempdir().expect("carpeta"); + let store = Store::open(&dir.path().join("history.db")).expect("abre"); + let secret = "ruta-secreta-del-archivo"; + let id = store.insert_text("uuid-r", secret, 1).expect("insert"); + store.mark_broken(id, 2).expect("roto"); + store.purge_broken_before(10).expect("purga"); + let wal = std::fs::read(dir.path().join("history.db-wal")).unwrap_or_default(); + assert!( + !wal.windows(secret.len()) + .any(|window| window == secret.as_bytes()), + "la purga también es un borrado" + ); + } + + #[test] + fn broken_items_go_after_their_grace_and_take_their_bytes_along() { + let (dir, store) = on_disk(); + let id = store + .insert_item("uuid-roto", &image(3, 100_000), "", 1) + .expect("insert"); + store.mark_broken(id, 5).expect("roto"); + let policy = Policy { + broken_for: Some(10), + ..Default::default() + }; + assert_eq!(store.sweep(&policy, 14).expect("aún no").broken, 0); + assert_eq!(store.sweep(&policy, 16).expect("ya").broken, 1); + let files = crate::blobs::files_under(&dir.path().join("blobs")).len(); + assert_eq!(files, 0, "purgar un roto no puede dejar su imagen en disco"); + } + + #[test] + fn a_blob_nobody_points_at_is_swept_once_it_has_settled() { + let (dir, store) = on_disk(); + let blobs = crate::Blobs::at(&dir.path().join("blobs")).expect("blobs"); + let digest = blobs.put(b"de una escritura interrumpida").expect("guarda"); + let path = dir + .path() + .join("blobs") + .join(&digest[0..2]) + .join(&digest[2..4]) + .join(&digest); + std::fs::File::options() + .write(true) + .open(&path) + .expect("abre") + .set_modified(std::time::UNIX_EPOCH) + .expect("envejece"); + let swept = store.sweep(&Policy::default(), 10).expect("barre"); + assert_eq!(swept.orphans, 1); + assert!(!blobs.exists(&digest)); + } + + #[test] + fn a_blob_with_an_owner_is_never_an_orphan() { + let (dir, store) = on_disk(); + store + .insert_item("uuid-1", &image(9, 100_000), "", 1) + .expect("insert"); + for path in crate::blobs::files_under(&dir.path().join("blobs")) { + std::fs::File::options() + .write(true) + .open(&path) + .expect("abre") + .set_modified(std::time::UNIX_EPOCH) + .expect("envejece"); + } + assert_eq!( + store.sweep(&Policy::default(), 10).expect("barre").orphans, + 0 + ); + assert!(store.payload_of(1, "public.png").expect("lee").is_some()); + } + + #[test] + fn everything_at_once_reports_each_count() { + let (_dir, store) = on_disk(); + let ids = fill(&store, 6); + store.mark_broken(ids[0], 1).expect("roto"); + let policy = Policy { + keep_for: Some(6), + keep_at_most: Some(2), + bytes_at_most: Some(i64::MAX), + broken_for: Some(1), + }; + let swept = store.sweep(&policy, 10).expect("barre"); + assert_eq!( + swept, + Swept { + broken: 1, + expired: 2, + over_count: 1, + over_bytes: 0, + orphans: 0, + }, + "cada regla cuenta lo suyo, en orden, sin contar dos veces" + ); + assert_eq!(alive(&store), vec![ids[5], ids[4]]); + } + + #[test] + fn a_sweep_leaves_nothing_in_the_write_ahead_log() { + let (dir, store) = on_disk(); + let secret = "clave-que-se-va"; + store.insert_text("uuid-s", secret, 1).expect("insert"); + let policy = Policy { + keep_for: Some(1), + ..Default::default() + }; + store.sweep(&policy, 10).expect("barre"); + for file in ["history.db", "history.db-wal"] { + let bytes = std::fs::read(dir.path().join(file)).unwrap_or_default(); + assert!( + !bytes + .windows(secret.len()) + .any(|window| window == secret.as_bytes()), + "«{secret}» sigue legible en {file}" + ); + } + } + + fn captured(text: &str) -> Item { + Item { + kind: Some(Kind::Text), + formats: vec![ + Format { + id: "public.utf8-plain-text".into(), + payload: Payload::Inline(text.as_bytes().to_vec()), + }, + Format { + id: "public.rtf".into(), + payload: Payload::Inline(format!("{{\\rtf1 {text}}}").into_bytes()), + }, + ], + } + } + + #[test] + fn editing_replaces_the_content_and_drops_the_renderings_that_no_longer_match() { + let store = Store::in_memory().expect("esquema"); + let id = store + .insert_item("uuid-e", &captured("hola mundo"), "hola mundo", 1) + .expect("insert"); + store.update_text(id, "adiós mundo", 2).expect("edita"); + + let card = &store + .list(&Filter::default(), 10, None) + .expect("lista") + .rows[0]; + assert_eq!(card.preview, "adiós mundo"); + assert_eq!( + card.modified_at, 1, + "editar no lo sube: el usuario ya lo tiene delante" + ); + assert_eq!( + store.formats_of(id).expect("formatos"), + vec![cp_core::item::SYNTHETIC_TEXT.to_string()], + "el RTF decía «hola» y pegarlo sería pegar lo viejo" + ); + assert_eq!( + store + .payload_of(id, cp_core::item::SYNTHETIC_TEXT) + .expect("lee") + .as_deref(), + Some("adiós mundo".as_bytes()) + ); + } + + #[test] + fn the_edited_text_is_what_gets_found_and_classified() { + let store = Store::in_memory().expect("esquema"); + let id = store + .insert_item("uuid-e", &captured("hola mundo"), "hola mundo", 1) + .expect("insert"); + store.update_text(id, "#FF8800", 2).expect("edita"); + assert!(store.search("hola").expect("busca").is_empty()); + assert_eq!(store.search("ff8800").expect("busca").len(), 1); + let card = &store + .list(&Filter::default(), 10, None) + .expect("lista") + .rows[0]; + assert_eq!(card.kind, Some(Kind::Color)); + assert_eq!( + store.find_by_hash(&Item::plain("#FF8800")).expect("hash"), + Some(id), + "la identidad es la del texto nuevo" + ); + assert!( + store + .changed_since(1) + .expect("cambios") + .contains(&"uuid-e".to_string()), + "la versión avanza" + ); + } + + #[test] + fn editing_an_image_into_text_takes_its_bytes_off_the_disk() { + let (dir, store) = on_disk(); + let id = store + .insert_item("uuid-img", &image(2, 100_000), "", 1) + .expect("insert"); + store.set_ocr_text(id, "texto leído", 2).expect("ocr"); + store.set_meta(id, "width", "800").expect("meta"); + store.update_text(id, "texto leído", 3).expect("edita"); + assert_eq!( + crate::blobs::files_under(&dir.path().join("blobs")).len(), + 0 + ); + assert!(store.all_meta(id).expect("meta").is_empty()); + assert_eq!( + store.search("leido").expect("busca").len(), + 1, + "ahora es contenido" + ); + assert!(store.pending_ocr(10).expect("ocr").is_empty()); + } + + #[test] + fn what_was_edited_away_is_not_left_lying_in_the_database_or_its_log() { + let dir = tempfile::tempdir().expect("carpeta"); + let store = Store::open(&dir.path().join("history.db")).expect("abre"); + let secret = "hunter2-la-de-antes"; + let id = store.insert_text("uuid-s", secret, 1).expect("insert"); + store.checkpoint().expect("ya está en la base principal"); + store.update_text(id, "texto inocente", 2).expect("edita"); + for file in ["history.db", "history.db-wal"] { + let bytes = std::fs::read(dir.path().join(file)).unwrap_or_default(); + assert!( + !bytes + .windows(secret.len()) + .any(|window| window == secret.as_bytes()), + "«{secret}» sigue legible en {file}" + ); + } + } + + #[test] + fn editing_a_broken_file_makes_it_a_whole_text_again() { + let store = Store::in_memory().expect("esquema"); + let id = store + .insert_text("uuid-roto", "/tmp/se-fue.txt", 1) + .expect("insert"); + store.mark_broken(id, 5).expect("roto"); + store.update_text(id, "lo que decía", 6).expect("edita"); + let rows = store + .list(&Filter::default(), 10, None) + .expect("lista") + .rows; + assert_eq!(rows.len(), 1, "un texto no puede estar roto"); + assert_eq!(rows[0].broken_since, None); + } + + #[test] + fn editing_what_does_not_exist_or_was_deleted_is_refused() { + let store = Store::in_memory().expect("esquema"); + assert!(matches!( + store.update_text(404, "nada", 1), + Err(Error::NoSuchItem { id: 404 }) + )); + let id = store.insert_text("uuid-d", "algo", 1).expect("insert"); + store.mark_deleted(id, 2).expect("borra"); + assert!(store.update_text(id, "resucita", 3).is_err()); + assert_eq!(store.count().expect("cuenta"), 0); + } + + #[test] + fn a_long_edit_goes_to_disk_and_an_absurd_one_is_refused() { + let (_dir, store) = on_disk(); + let id = store.insert_text("uuid-l", "corto", 1).expect("insert"); + let long = "x".repeat(cp_core::item::INLINE_UP_TO + 1); + store.update_text(id, &long, 2).expect("edita"); + assert_eq!(store.usage().expect("uso").bytes as usize, long.len()); + + let memory = Store::in_memory().expect("esquema"); + let id = memory.insert_text("uuid-m", "corto", 1).expect("insert"); + assert!( + matches!( + memory.update_text(id, &long, 2), + Err(Error::NeedsBlobStore { .. }) + ), + "sin carpeta no hay dónde dejarlo" + ); + assert_eq!( + memory.search("corto").expect("busca").len(), + 1, + "y lo de antes sigue intacto" + ); + } + + #[test] + fn the_parser_and_the_list_speak_the_same_filter() { + let store = Store::in_memory().expect("esquema"); + let rows = [ + ("uuid-1", "reunión lunes", "Slack", 10), + ("uuid-2", "reunión martes", "Code", 20), + ("uuid-3", "otra cosa", "Slack", 30), + ]; + for (uuid, text, app, at) in rows { + let id = store.insert_text(uuid, text, at).expect("insert"); + store.set_source(id, app, at).expect("origen"); + } + let clock = crate::query::Clock { + now: 40, + day_start: 0, + }; + let filter = crate::query::parse("reunion @slack", &clock); + let found = store.list(&filter, 10, None).expect("lista").rows; + assert_eq!(found.len(), 1); + assert_eq!(found[0].preview, "reunión lunes"); + assert_eq!( + found[0] + .snippet + .as_ref() + .map(|snippet| snippet.excerpt.plain()), + Some("reunión lunes".into()) + ); + } +} From bfeb3b15abb6c3a2d0ae0730a81325db7dfc01c7 Mon Sep 17 00:00:00 2001 From: rgdevment Date: Sun, 20 Sep 2026 23:31:36 -0300 Subject: [PATCH 2/7] fix(core): what the audit of paste-as and the insisting capture found MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The HTML-to-Markdown scanner sliced at byte one, so any rich copy whose text after a tag began with an accented letter panicked; it now advances by characters, gives up on a `<` with no `>` left in one pass instead of rescanning per bracket, respects comments and quoted attributes, and never slices before the start of an anchor whose surroundings were trimmed inside `
`. `dedent` strips the whitespace prefix every line
shares, character by character, instead of a byte count that could land
inside a non-breaking space. The curl form escapes single quotes: a JWT
signature may carry any character, and the rendered line is meant for a
shell. Percentages in rgb channels and in alpha are scaled. Quotes keep
every paragraph; fences outgrow the backticks inside; odd URLs go in
angle brackets.

`Form::AsIs` is gone — pasting the original item is the interface's job
— and every form `forms_for` offers now renders, which a test asserts
over ten contents.

`capture_insisting` no longer spawns a reader per attempt: `reading::begin`
hands back a `Pending` that can be waited on more than once, so a late
answer from Excel is collected instead of queued behind a fresh read,
and a hung source blocks one thread rather than five. Probe C6 allows a
legitimate attempt to take up to `PATIENCE`.
---
 crates/cp-core/src/paste_as.rs          | 412 +++++++++++++++++++-----
 crates/cp-mac-sys/src/reading.rs        |  43 ++-
 crates/cp-mac/examples/probe/battery.rs |   2 +-
 crates/cp-mac/src/capture.rs            |   5 +-
 crates/cp-mac/src/content.rs            |  12 +-
 5 files changed, 374 insertions(+), 100 deletions(-)

diff --git a/crates/cp-core/src/paste_as.rs b/crates/cp-core/src/paste_as.rs
index 70aada5..f7501b8 100644
--- a/crates/cp-core/src/paste_as.rs
+++ b/crates/cp-core/src/paste_as.rs
@@ -15,7 +15,6 @@ pub struct Content<'a> {
 
 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
 pub enum Form {
-    AsIs,
     PlainText,
     Markdown,
     JsonPretty,
@@ -49,7 +48,7 @@ pub enum Rendered {
 pub const JPEG_QUALITY: u8 = 85;
 
 pub fn forms_for(content: &Content) -> Vec {
-    let mut forms = vec![Form::AsIs];
+    let mut forms = Vec::new();
     if content.rich && content.text.is_some() {
         forms.push(Form::PlainText);
     }
@@ -75,8 +74,11 @@ pub fn forms_for(content: &Content) -> Vec {
             }
         }
         Some(Kind::Link) => {
-            forms.extend([Form::LinkMarkdown, Form::LinkDomain]);
-            if content.title.is_some_and(|title| !title.trim().is_empty()) {
+            forms.push(Form::LinkMarkdown);
+            if domain_of(text).is_some() {
+                forms.push(Form::LinkDomain);
+            }
+            if title_of(content).is_some() {
                 forms.push(Form::LinkTitled);
             }
         }
@@ -106,13 +108,6 @@ pub fn forms_for(content: &Content) -> Vec {
 pub fn render(form: Form, content: &Content) -> Option {
     let text = content.text.map(str::trim).unwrap_or_default();
     let rendered = match form {
-        Form::AsIs => {
-            return match (content.text, content.png) {
-                (Some(text), _) => Some(Rendered::Text(text.to_owned())),
-                (None, Some(png)) => Some(Rendered::Png(png.to_vec())),
-                (None, None) => None,
-            };
-        }
         Form::PlainText => content.text?.to_owned(),
         Form::Markdown => markdown_of_html(content.html?),
         Form::JsonPretty => serde_json::to_string_pretty(&json_of(text)?).ok()?,
@@ -122,18 +117,21 @@ pub fn render(form: Form, content: &Content) -> Option {
         Form::ColorHex => hex_of(parse_color(text)?),
         Form::ColorRgb => rgb_of(parse_color(text)?),
         Form::ColorHsl => hsl_of(parse_color(text)?),
-        Form::LinkMarkdown => format!("[{}]({text})", content.title.map(str::trim).unwrap_or(text)),
+        Form::LinkMarkdown => markdown_link(title_of(content).unwrap_or(text), text),
         Form::LinkDomain => domain_of(text)?.to_owned(),
-        Form::LinkTitled => format!("{} — {text}", content.title?.trim()),
+        Form::LinkTitled => format!("{} — {text}", title_of(content)?),
         Form::CodeOneLine => one_line(text),
-        Form::CodeBlock => format!("```\n{text}\n```"),
+        Form::CodeBlock => fenced(text),
         Form::CodeDedented => dedent(content.text?),
         Form::TokenHeader => format!("Authorization: Bearer {text}"),
         Form::TokenClaims => {
             let claims = crate::token::claims_of(text)?;
             serde_json::to_string_pretty(&Value::Object(claims.payload)).ok()?
         }
-        Form::TokenCurl => format!("curl -H 'Authorization: Bearer {text}' \"$URL\""),
+        Form::TokenCurl => format!(
+            "curl -H 'Authorization: Bearer {}' \"$URL\"",
+            text.replace('\'', "'\\''")
+        ),
         Form::ImageJpeg => return jpeg_of(content.png?).map(Rendered::Jpeg),
         Form::ImageOcr => content.ocr?.trim().to_owned(),
         Form::Path => content.paths.join("\n"),
@@ -141,6 +139,28 @@ pub fn render(form: Form, content: &Content) -> Option {
     Some(Rendered::Text(rendered))
 }
 
+fn title_of<'a>(content: &Content<'a>) -> Option<&'a str> {
+    content
+        .title
+        .map(str::trim)
+        .filter(|title| !title.is_empty())
+}
+
+fn markdown_link(label: &str, url: &str) -> String {
+    let label = label.replace(['[', ']'], " ");
+    if url.contains([' ', '(', ')', '<', '>']) {
+        format!("[{}](<{}>)", label.trim(), url.replace(['<', '>'], ""))
+    } else {
+        format!("[{}]({url})", label.trim())
+    }
+}
+
+fn fenced(text: &str) -> String {
+    let longest = text.split(|c| c != '`').map(str::len).max().unwrap_or(0);
+    let fence = "`".repeat(longest.max(2) + 1);
+    format!("{fence}\n{text}\n{fence}")
+}
+
 fn json_of(text: &str) -> Option {
     serde_json::from_str(text).ok()
 }
@@ -235,19 +255,37 @@ pub fn parse_color(text: &str) -> Option {
     let (kind, inner) = lower
         .split_once('(')
         .and_then(|(kind, rest)| Some((kind, rest.strip_suffix(')')?)))?;
-    let parts: Vec = inner
+    let parts: Vec<(f64, bool)> = inner
         .split(',')
-        .map(|part| part.trim().trim_end_matches('%').parse::().ok())
+        .map(|part| {
+            let part = part.trim();
+            let percent = part.ends_with('%');
+            part.trim_end_matches('%')
+                .parse::()
+                .ok()
+                .map(|value| (value, percent))
+        })
         .collect::>()?;
-    let alpha = |value: Option<&f64>| value.copied().unwrap_or(1.0).clamp(0.0, 1.0);
+    let alpha = |value: Option<&(f64, bool)>| match value {
+        Some((value, true)) => (value / 100.0).clamp(0.0, 1.0),
+        Some((value, false)) => value.clamp(0.0, 1.0),
+        None => 1.0,
+    };
+    let byte = |(value, percent): (f64, bool)| {
+        channel(if percent {
+            value * 255.0 / 100.0
+        } else {
+            value
+        })
+    };
     match (kind, parts.as_slice()) {
         ("rgb", [r, g, b]) | ("rgba", [r, g, b, _]) => Some(Rgba {
-            r: channel(*r),
-            g: channel(*g),
-            b: channel(*b),
+            r: byte(*r),
+            g: byte(*g),
+            b: byte(*b),
             a: alpha(parts.get(3)),
         }),
-        ("hsl", [h, s, l]) | ("hsla", [h, s, l, _]) => {
+        ("hsl", [(h, _), (s, _), (l, _)]) | ("hsla", [(h, _), (s, _), (l, _), _]) => {
             let (r, g, b) = rgb_of_hsl(*h, s / 100.0, l / 100.0);
             Some(Rgba {
                 r,
@@ -391,20 +429,25 @@ fn one_line(text: &str) -> String {
 }
 
 fn dedent(text: &str) -> String {
-    let indent = text
+    let common = text
         .lines()
         .filter(|line| !line.trim().is_empty())
-        .map(|line| line.len() - line.trim_start().len())
-        .min()
-        .unwrap_or(0);
+        .map(|line| &line[..line.len() - line.trim_start().len()])
+        .reduce(|shared, next| {
+            let kept = shared
+                .chars()
+                .zip(next.chars())
+                .take_while(|(a, b)| a == b)
+                .map(|(a, _)| a.len_utf8())
+                .sum();
+            &shared[..kept]
+        })
+        .unwrap_or("");
     let mut out = text
         .lines()
         .map(|line| {
-            if line.len() >= indent {
-                &line[indent..]
-            } else {
-                line.trim_start()
-            }
+            line.strip_prefix(common)
+                .unwrap_or_else(|| line.trim_start())
         })
         .collect::>()
         .join("\n");
@@ -447,19 +490,43 @@ pub fn markdown_of_html(html: &str) -> String {
                 .chars()
                 .next()
                 .is_some_and(|c| c.is_ascii_alphabetic() || "/!?".contains(c))
-            && let Some(end) = after.find('>')
         {
-            writer.tag(&after[..end]);
-            rest = &after[end + 1..];
-            continue;
+            match tag_end(after) {
+                Some(end) => {
+                    writer.tag(&after[..end]);
+                    rest = &after[end + 1..];
+                    continue;
+                }
+                None => {
+                    writer.text(&decode_entities(rest));
+                    break;
+                }
+            }
         }
-        let next = rest[1..].find('<').map_or(rest.len(), |at| at + 1);
+        let first = rest.chars().next().map_or(1, char::len_utf8);
+        let next = rest[first..].find('<').map_or(rest.len(), |at| at + first);
         writer.text(&decode_entities(&rest[..next]));
         rest = &rest[next..];
     }
     writer.finish()
 }
 
+fn tag_end(after: &str) -> Option {
+    if after.starts_with("!--") {
+        return after.find("-->").map(|at| at + 2);
+    }
+    let mut quote: Option = None;
+    for (at, c) in after.char_indices() {
+        match (quote, c) {
+            (Some(open), c) if c == open => quote = None,
+            (None, '"' | '\'') => quote = Some(c),
+            (None, '>') => return Some(at),
+            _ => {}
+        }
+    }
+    None
+}
+
 fn fragment_of(html: &str) -> &str {
     let start = html
         .find("")
@@ -480,7 +547,7 @@ struct MarkdownWriter {
     link: Option<(String, usize)>,
     skipping: Option<&'static str>,
     preformatted: bool,
-    quoting: bool,
+    quoting: Option,
     opening: String,
     fresh: bool,
 }
@@ -493,7 +560,7 @@ impl Default for MarkdownWriter {
             link: None,
             skipping: None,
             preformatted: false,
-            quoting: false,
+            quoting: None,
             opening: String::new(),
             fresh: true,
         }
@@ -582,11 +649,27 @@ impl MarkdownWriter {
             }
             ("blockquote", false) => {
                 self.blank_line();
-                self.quoting = true;
-                self.out.push_str("> ");
+                self.quoting = Some(self.out.len());
             }
             ("blockquote", true) => {
-                self.quoting = false;
+                if let Some(from) = self.quoting.take() {
+                    self.newline();
+                    let quoted = self.out[from..]
+                        .trim_end()
+                        .lines()
+                        .map(|line| {
+                            if line.is_empty() {
+                                ">".to_owned()
+                            } else {
+                                format!("> {line}")
+                            }
+                        })
+                        .collect::>()
+                        .join("\n");
+                    self.out.truncate(from);
+                    self.out.push_str("ed);
+                    self.fresh = false;
+                }
                 self.blank_line();
             }
             ("a", false) => {
@@ -595,6 +678,7 @@ impl MarkdownWriter {
             }
             ("a", true) => {
                 if let Some((href, from)) = self.link.take() {
+                    let from = from.min(self.out.len());
                     let label = self.out[from..].trim().to_owned();
                     self.out.truncate(from);
                     if href.is_empty() {
@@ -630,6 +714,10 @@ impl MarkdownWriter {
             self.opening = unopened.to_owned();
             return;
         }
+        if self.preformatted {
+            self.out.push_str(marker);
+            return;
+        }
         let kept = self.out.trim_end_matches(' ').len();
         let had_space = kept < self.out.len();
         self.out.truncate(kept);
@@ -777,11 +865,52 @@ mod tests {
     }
 
     #[test]
-    fn every_item_can_at_least_be_pasted_as_it_is() {
-        let content = text_of(Kind::Text, "hola");
-        assert_eq!(forms_for(&content), vec![Form::AsIs]);
-        assert_eq!(rendered(Form::AsIs, &content), "hola");
-        assert_eq!(render(Form::AsIs, &Content::default()), None);
+    fn plain_text_has_no_forms_of_its_own() {
+        assert!(forms_for(&text_of(Kind::Text, "hola")).is_empty());
+        assert!(forms_for(&Content::default()).is_empty());
+    }
+
+    #[test]
+    fn every_form_that_is_offered_can_be_rendered() {
+        let png = tiny_png();
+        let jwt = "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiJjcC0zIn0.firma";
+        let contents = [
+            Content {
+                kind: Some(Kind::Text),
+                text: Some("hola"),
+                html: Some("hola"),
+                rich: true,
+                ..Default::default()
+            },
+            text_of(Kind::Json, r#"[{"a": 1}, {"b": 2}]"#),
+            text_of(Kind::Json, "[1, 2]"),
+            text_of(Kind::Color, "hsla(10, 20%, 30%, 40%)"),
+            Content {
+                title: Some("Ejemplo"),
+                ..text_of(Kind::Link, "https://ejemplo.test/a b")
+            },
+            text_of(Kind::Link, "https://localhost/"),
+            text_of(Kind::Code, "  x\n  y"),
+            text_of(Kind::Token, jwt),
+            text_of(Kind::Token, "ghp_not_a_real_token_for_tests_0000000000"),
+            Content {
+                kind: Some(Kind::Image),
+                png: Some(&png),
+                ocr: Some("leído"),
+                paths: vec!["/tmp/a.png".into()],
+                ..Default::default()
+            },
+        ];
+        for content in &contents {
+            let forms = forms_for(content);
+            assert!(!forms.is_empty(), "{content:?}");
+            for form in forms {
+                assert!(
+                    render(form, content).is_some(),
+                    "{form:?} sobre {content:?}"
+                );
+            }
+        }
     }
 
     #[test]
@@ -793,16 +922,13 @@ mod tests {
             rich: true,
             ..Default::default()
         };
-        assert_eq!(
-            forms_for(&with_html),
-            vec![Form::AsIs, Form::PlainText, Form::Markdown]
-        );
+        assert_eq!(forms_for(&with_html), vec![Form::PlainText, Form::Markdown]);
         assert_eq!(rendered(Form::Markdown, &with_html), "**hola**");
         let rtf_only = Content {
             rich: true,
             ..text_of(Kind::Text, "hola")
         };
-        assert_eq!(forms_for(&rtf_only), vec![Form::AsIs, Form::PlainText]);
+        assert_eq!(forms_for(&rtf_only), vec![Form::PlainText]);
     }
 
     #[test]
@@ -814,7 +940,6 @@ mod tests {
         assert_eq!(
             forms_for(&content),
             vec![
-                Form::AsIs,
                 Form::JsonPretty,
                 Form::JsonMinified,
                 Form::JsonKeys,
@@ -857,11 +982,11 @@ mod tests {
         let content = text_of(Kind::Json, "[1, 2, 3]");
         assert_eq!(
             forms_for(&content),
-            vec![Form::AsIs, Form::JsonPretty, Form::JsonMinified]
+            vec![Form::JsonPretty, Form::JsonMinified]
         );
         assert_eq!(render(Form::JsonKeys, &content), None);
         let broken = text_of(Kind::Json, "{no es json}");
-        assert_eq!(forms_for(&broken), vec![Form::AsIs]);
+        assert!(forms_for(&broken).is_empty());
     }
 
     #[test]
@@ -869,7 +994,7 @@ mod tests {
         let content = text_of(Kind::Color, "#FF8800");
         assert_eq!(
             forms_for(&content),
-            vec![Form::AsIs, Form::ColorHex, Form::ColorRgb, Form::ColorHsl]
+            vec![Form::ColorHex, Form::ColorRgb, Form::ColorHsl]
         );
         assert_eq!(rendered(Form::ColorHex, &content), "#FF8800");
         assert_eq!(rendered(Form::ColorRgb, &content), "rgb(255, 136, 0)");
@@ -1003,10 +1128,7 @@ mod tests {
 
     #[test]
     fn a_colour_that_does_not_parse_offers_nothing_extra() {
-        assert_eq!(
-            forms_for(&text_of(Kind::Color, "#GGGGGG")),
-            vec![Form::AsIs]
-        );
+        assert!(forms_for(&text_of(Kind::Color, "#GGGGGG")).is_empty());
         assert_eq!(parse_color("rgb(1, 2)"), None);
         assert_eq!(parse_color("hsl(1, 2, 3, 4, 5)"), None);
         assert_eq!(parse_color("#12345"), None);
@@ -1015,10 +1137,7 @@ mod tests {
     #[test]
     fn a_link_is_offered_as_markdown_domain_and_with_its_title() {
         let bare = text_of(Kind::Link, "https://www.ejemplo.test/ruta?x=1#f");
-        assert_eq!(
-            forms_for(&bare),
-            vec![Form::AsIs, Form::LinkMarkdown, Form::LinkDomain]
-        );
+        assert_eq!(forms_for(&bare), vec![Form::LinkMarkdown, Form::LinkDomain]);
         assert_eq!(
             rendered(Form::LinkMarkdown, &bare),
             "[https://www.ejemplo.test/ruta?x=1#f](https://www.ejemplo.test/ruta?x=1#f)"
@@ -1084,12 +1203,7 @@ mod tests {
         let content = text_of(Kind::Code, source);
         assert_eq!(
             forms_for(&content),
-            vec![
-                Form::AsIs,
-                Form::CodeOneLine,
-                Form::CodeBlock,
-                Form::CodeDedented
-            ]
+            vec![Form::CodeOneLine, Form::CodeBlock, Form::CodeDedented]
         );
         assert_eq!(
             rendered(Form::CodeOneLine, &content),
@@ -1114,10 +1228,7 @@ mod tests {
     #[test]
     fn a_token_becomes_a_header_or_a_curl_and_a_jwt_shows_its_claims() {
         let opaque = text_of(Kind::Token, "ghp_not_a_real_token_for_tests_0000000000");
-        assert_eq!(
-            forms_for(&opaque),
-            vec![Form::AsIs, Form::TokenHeader, Form::TokenCurl]
-        );
+        assert_eq!(forms_for(&opaque), vec![Form::TokenHeader, Form::TokenCurl]);
         assert_eq!(
             rendered(Form::TokenHeader, &opaque),
             "Authorization: Bearer ghp_not_a_real_token_for_tests_0000000000"
@@ -1173,25 +1284,18 @@ mod tests {
             png: Some(&png),
             ..Default::default()
         };
-        assert_eq!(forms_for(&silent), vec![Form::AsIs, Form::ImageJpeg]);
-        assert_eq!(
-            render(Form::AsIs, &silent),
-            Some(Rendered::Png(png.clone()))
-        );
+        assert_eq!(forms_for(&silent), vec![Form::ImageJpeg]);
         let read = Content {
             ocr: Some(" Pedido 4417 "),
             ..silent.clone()
         };
-        assert_eq!(
-            forms_for(&read),
-            vec![Form::AsIs, Form::ImageJpeg, Form::ImageOcr]
-        );
+        assert_eq!(forms_for(&read), vec![Form::ImageJpeg, Form::ImageOcr]);
         assert_eq!(rendered(Form::ImageOcr, &read), "Pedido 4417");
         let blank = Content {
             ocr: Some("   "),
             ..silent.clone()
         };
-        assert_eq!(forms_for(&blank), vec![Form::AsIs, Form::ImageJpeg]);
+        assert_eq!(forms_for(&blank), vec![Form::ImageJpeg]);
     }
 
     #[test]
@@ -1250,7 +1354,7 @@ mod tests {
             paths: vec!["/tmp/uno.txt".into(), "/tmp/dos.txt".into()],
             ..Default::default()
         };
-        assert_eq!(forms_for(&content), vec![Form::AsIs, Form::Path]);
+        assert_eq!(forms_for(&content), vec![Form::Path]);
         assert_eq!(rendered(Form::Path, &content), "/tmp/uno.txt\n/tmp/dos.txt");
     }
 
@@ -1387,6 +1491,146 @@ mod tests {
         assert_eq!(markdown_of_html("
a\n
b"), "```\na\n```\n\nb"); } + #[test] + fn a_multibyte_character_right_after_a_tag_does_not_panic() { + assert_eq!(markdown_of_html("

ñandú

"), "ñandú"); + assert_eq!(markdown_of_html("—€"), "**—**€"); + assert_eq!(markdown_of_html("ñ<ñ"), "ñ<ñ"); + } + + #[test] + fn an_angle_bracket_that_never_closes_is_text_and_costs_one_pass() { + let hostile = " b -->x"), "x"); + assert_eq!( + markdown_of_html("b\">t"), + "[t](x)" + ); + assert_eq!(markdown_of_html("