diff --git a/rust/src/browser/leveldb/local_storage.rs b/rust/src/browser/leveldb/local_storage.rs new file mode 100644 index 0000000000..a2e3d5236e --- /dev/null +++ b/rust/src/browser/leveldb/local_storage.rs @@ -0,0 +1,70 @@ +//! Chromium `Local Storage` layer on top of the LevelDB reader. +//! +//! Chromium stores each origin's `localStorage` items under the key +//! `_\0` where `` is the serialized origin (`https://www.kimi.ai`) and +//! `` starts with a format byte: `0x00` for UTF-16LE text, `0x01` for Latin-1. Values use +//! the same format byte. Other keys in the database (`VERSION`, `META:`, +//! `METAACCESS:`) are bookkeeping and are ignored. + +use super::{Entry, LevelDbError, read_entries}; +use std::path::{Path, PathBuf}; + +const KEY_PREFIX: u8 = b'_'; +const ORIGIN_TERMINATOR: u8 = 0; +const FORMAT_UTF16LE: u8 = 0; +const FORMAT_LATIN1: u8 = 1; + +/// One decoded `localStorage` item. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct LocalStorageEntry { + pub key: String, + pub value: String, +} + +/// Directory holding Local Storage for a Chromium profile directory (`Default`, `Profile 1`, ...). +pub fn local_storage_dir(profile_dir: &Path) -> PathBuf { + profile_dir.join("Local Storage").join("leveldb") +} + +/// Read the `localStorage` items stored for `origin` (for example `https://www.kimi.ai`) in the +/// LevelDB directory `dir`, sorted by key. A trailing slash on `origin` is ignored. +pub fn read_local_storage_entries( + dir: &Path, + origin: &str, +) -> Result, LevelDbError> { + Ok(decode_origin_entries(&read_entries(dir)?, origin)) +} + +pub(super) fn decode_origin_entries(entries: &[Entry], origin: &str) -> Vec { + let mut prefix = Vec::with_capacity(origin.len() + 2); + prefix.push(KEY_PREFIX); + prefix.extend_from_slice(origin.trim_end_matches('/').as_bytes()); + prefix.push(ORIGIN_TERMINATOR); + + entries + .iter() + .filter_map(|entry| { + let key = decode_text(entry.key.strip_prefix(prefix.as_slice())?)?; + let value = decode_text(&entry.value)?; + Some(LocalStorageEntry { key, value }) + }) + .collect() +} + +/// Decode a format-byte-prefixed Chromium string; `None` for an unknown format or bad UTF-16. +fn decode_text(bytes: &[u8]) -> Option { + let (&format, data) = bytes.split_first()?; + match format { + FORMAT_LATIN1 => Some(data.iter().map(|&byte| char::from(byte)).collect()), + FORMAT_UTF16LE if data.len() % 2 == 0 => { + let units: Vec = data + .as_chunks::<2>() + .0 + .iter() + .map(|pair| u16::from_le_bytes(*pair)) + .collect(); + String::from_utf16(&units).ok() + } + _ => None, + } +} diff --git a/rust/src/browser/leveldb/log.rs b/rust/src/browser/leveldb/log.rs new file mode 100644 index 0000000000..1256a818cb --- /dev/null +++ b/rust/src/browser/leveldb/log.rs @@ -0,0 +1,122 @@ +//! LevelDB write-ahead log (`*.log`) reader. +//! +//! A log is a sequence of 32 KiB blocks holding physical records (7 byte header: masked CRC32C, +//! length, type) that are reassembled into logical records. Each logical record is a write batch: +//! `sequence: u64le`, `count: u32le`, then `count` operations (`1 key value` puts and `0 key` +//! deletes, with varint-length-prefixed byte strings). Operation `i` has sequence `sequence + i`. +//! +//! The live log of a running browser routinely ends in a half-written record, so a truncated or +//! malformed tail ends the scan quietly and everything before it is kept. Record checksums are +//! not verified: this is a best-effort read of another program's cache, not a database recovery. + +use super::Record; +use super::varint::{read_length_prefixed, read_u32_le, read_u64_le}; + +const BLOCK_SIZE: usize = 32 * 1024; +const HEADER_SIZE: usize = 7; +const BATCH_HEADER_SIZE: usize = 12; + +const TYPE_ZERO: u8 = 0; +const TYPE_FULL: u8 = 1; +const TYPE_FIRST: u8 = 2; +const TYPE_MIDDLE: u8 = 3; +const TYPE_LAST: u8 = 4; + +const OP_DELETE: u8 = 0; +const OP_PUT: u8 = 1; + +/// Feed every put/delete found in `data` to `emit`. +pub(super) fn read_log(data: &[u8], emit: &mut impl FnMut(Record)) { + let mut assembled: Vec = Vec::new(); + let mut in_fragmented = false; + + let mut offset = 0usize; + while offset < data.len() { + let block_remaining = BLOCK_SIZE - (offset % BLOCK_SIZE); + if block_remaining < HEADER_SIZE { + offset += block_remaining; + continue; + } + let Some(header) = data.get(offset..offset + HEADER_SIZE) else { + return; + }; + let length = usize::from(u16::from_le_bytes([header[4], header[5]])); + let kind = header[6]; + if kind == TYPE_ZERO && length == 0 { + // Zero padding: the rest of this block is unused. + offset += block_remaining; + continue; + } + let payload_start = offset + HEADER_SIZE; + if HEADER_SIZE + length > block_remaining { + return; + } + let Some(payload) = data.get(payload_start..payload_start + length) else { + return; + }; + offset = payload_start + length; + + match kind { + TYPE_FULL => { + assembled.clear(); + in_fragmented = false; + read_batch(payload, emit); + } + TYPE_FIRST => { + assembled.clear(); + assembled.extend_from_slice(payload); + in_fragmented = true; + } + TYPE_MIDDLE if in_fragmented => assembled.extend_from_slice(payload), + TYPE_LAST if in_fragmented => { + assembled.extend_from_slice(payload); + in_fragmented = false; + read_batch(&assembled, emit); + assembled.clear(); + } + _ => return, + } + } +} + +fn read_batch(batch: &[u8], emit: &mut impl FnMut(Record)) { + let (Some(sequence), Some(count)) = (read_u64_le(batch, 0), read_u32_le(batch, 8)) else { + return; + }; + let mut rest = batch.get(BATCH_HEADER_SIZE..).unwrap_or_default(); + let mut records = Vec::new(); + for index in 0..u64::from(count) { + let Some((&op, after_op)) = rest.split_first() else { + return; + }; + let Some((key, after_key)) = read_length_prefixed(after_op) else { + return; + }; + let value = match op { + OP_PUT => { + let Some((value, after_value)) = read_length_prefixed(after_key) else { + return; + }; + rest = after_value; + Some(value.to_vec()) + } + OP_DELETE => { + rest = after_key; + None + } + _ => return, + }; + let Some(sequence) = sequence.checked_add(index) else { + return; + }; + records.push(Record { + key: key.to_vec(), + sequence, + value, + }); + } + if !rest.is_empty() { + return; + } + records.into_iter().for_each(emit); +} diff --git a/rust/src/browser/leveldb/mod.rs b/rust/src/browser/leveldb/mod.rs new file mode 100644 index 0000000000..aa95a9cb42 --- /dev/null +++ b/rust/src/browser/leveldb/mod.rs @@ -0,0 +1,144 @@ +//! Read-only LevelDB reader for Chromium browser storage (Local Storage, and other LevelDB +//! directories a browser keeps under a profile). +//! +//! This is a small hand-written reader rather than a dependency: it understands the write-ahead +//! log and sorted-table formats, including Snappy-compressed table blocks, and resolves each user +//! key to its newest value by sequence number. It does not open the database, take its `LOCK`, or +//! write anything, so it can be pointed at a profile of a running browser. +//! +//! Known limits, all acceptable for a best-effort credential import: +//! - The `MANIFEST` is not consulted, so a table that compaction already obsoleted but the +//! browser has not yet removed can still be read. Newer sequence numbers win, so this only +//! matters for a deleted key whose tombstone was compacted away. +//! - Checksums are not verified and only the bytewise comparator is assumed (which is what +//! `Local Storage` uses; a full scan does not depend on ordering anyway). +//! - Blocks using a compression type other than none/Snappy are skipped. +//! - Files larger than [`MAX_FILE_BYTES`] and decoded table blocks larger than +//! [`MAX_BLOCK_BYTES`] are skipped to bound each input allocation. The returned snapshot still +//! grows with the database's live data. + +pub mod local_storage; +mod log; +pub mod snappy; +mod table; +mod varint; + +#[cfg(test)] +mod tests; + +use std::collections::BTreeMap; +use std::ffi::OsStr; +use std::io::Read; +use std::path::Path; + +/// Largest single log or table file that will be read into memory. +pub const MAX_FILE_BYTES: u64 = 64 * 1024 * 1024; +/// Largest decoded table block accepted. +pub const MAX_BLOCK_BYTES: usize = 16 * 1024 * 1024; + +/// One live key/value pair of the database. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Entry { + pub key: Vec, + pub value: Vec, +} + +/// Failure to read a LevelDB directory at all (per-file problems are skipped, not reported). +#[derive(Debug, thiserror::Error)] +pub enum LevelDbError { + #[error("cannot read LevelDB directory: {0}")] + Io(#[from] std::io::Error), +} + +/// A put (`value: Some`) or delete (`value: None`) with the sequence number it was written at. +pub(crate) struct Record { + pub(crate) key: Vec, + pub(crate) sequence: u64, + pub(crate) value: Option>, +} + +/// Read every live entry of the LevelDB directory `dir`, sorted by key. +/// +/// Keys that were deleted (or whose newest record is a delete) are omitted. Unreadable or corrupt +/// files are skipped with a debug log so the remaining data is still returned. +pub fn read_entries(dir: &Path) -> Result, LevelDbError> { + let mut newest: BTreeMap, (u64, Option>)> = BTreeMap::new(); + let mut absorb = |record: Record| match newest.get_mut(&record.key) { + Some(existing) if existing.0 >= record.sequence => {} + Some(existing) => *existing = (record.sequence, record.value), + None => { + newest.insert(record.key, (record.sequence, record.value)); + } + }; + + for dir_entry in std::fs::read_dir(dir)? { + let Ok(dir_entry) = dir_entry else { continue }; + let path = dir_entry.path(); + let Some(kind) = FileKind::of(&path) else { + continue; + }; + let data = match read_bounded_file(&path) { + Ok(data) => data, + Err(error) => { + tracing::debug!(file = ?path.file_name(), %error, "skipping unreadable LevelDB file"); + continue; + } + }; + match kind { + FileKind::Log => log::read_log(&data, &mut absorb), + FileKind::Table => match table::read_table(&data, &mut absorb) { + Ok(0) => {} + Ok(skipped) => tracing::debug!( + file = ?path.file_name(), + skipped, + "skipped undecodable LevelDB table blocks" + ), + Err(error) => { + tracing::debug!(file = ?path.file_name(), %error, "skipping malformed LevelDB table"); + } + }, + } + } + + Ok(newest + .into_iter() + .filter_map(|(key, (_, value))| value.map(|value| Entry { key, value })) + .collect()) +} + +#[derive(Clone, Copy)] +enum FileKind { + Log, + Table, +} + +impl FileKind { + fn of(path: &Path) -> Option { + let extension = path.extension().and_then(OsStr::to_str)?; + if extension.eq_ignore_ascii_case("log") { + Some(Self::Log) + } else if extension.eq_ignore_ascii_case("ldb") || extension.eq_ignore_ascii_case("sst") { + Some(Self::Table) + } else { + None + } + } +} + +fn read_bounded_file(path: &Path) -> std::io::Result> { + let file = std::fs::File::open(path)?; + let size = file.metadata()?.len(); + if size > MAX_FILE_BYTES { + return Err(std::io::Error::other(format!( + "file is {size} bytes, over the {MAX_FILE_BYTES} byte limit" + ))); + } + let mut data = Vec::new(); + file.take(MAX_FILE_BYTES + 1).read_to_end(&mut data)?; + if data.len() as u64 > MAX_FILE_BYTES { + return Err(std::io::Error::other(format!( + "file grew beyond the {MAX_FILE_BYTES} byte limit while being read" + ))); + } + Ok(data) +} diff --git a/rust/src/browser/leveldb/snappy.rs b/rust/src/browser/leveldb/snappy.rs new file mode 100644 index 0000000000..3436a621df --- /dev/null +++ b/rust/src/browser/leveldb/snappy.rs @@ -0,0 +1,116 @@ +//! Raw Snappy block decompression (the format LevelDB uses for compressed table blocks). +//! +//! Format reference: . +//! The stream is a varint uncompressed length followed by literal and copy elements. +//! Output size is bounded by the caller so a hostile length preamble cannot exhaust memory. + +use super::varint::read_varint32; + +/// Why a Snappy block could not be decoded. +#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] +pub enum SnappyError { + #[error("snappy length preamble is missing or malformed")] + BadPreamble, + #[error("snappy output of {0} bytes exceeds the {1} byte limit")] + TooLarge(usize, usize), + #[error("snappy stream is truncated")] + Truncated, + #[error("snappy copy references data before the start of the output")] + BadOffset, + #[error("snappy stream length does not match its preamble")] + LengthMismatch, +} + +/// Decompress one raw Snappy block, refusing to produce more than `max_len` bytes. +pub fn decompress(input: &[u8], max_len: usize) -> Result, SnappyError> { + let (expected, mut pos) = read_varint32(input).ok_or(SnappyError::BadPreamble)?; + let expected = expected as usize; + if expected > max_len { + return Err(SnappyError::TooLarge(expected, max_len)); + } + + let mut out = Vec::with_capacity(expected); + while pos < input.len() { + let tag = input[pos]; + pos += 1; + match tag & 0b11 { + 0b00 => { + let mut len = usize::from(tag >> 2); + if len >= 60 { + let extra = len - 59; + let bytes = input.get(pos..pos + extra).ok_or(SnappyError::Truncated)?; + pos += extra; + len = bytes + .iter() + .rev() + .fold(0usize, |acc, b| (acc << 8) | usize::from(*b)); + } + len = len.checked_add(1).ok_or(SnappyError::Truncated)?; + let literal = pos + .checked_add(len) + .and_then(|end| input.get(pos..end)) + .ok_or(SnappyError::Truncated)?; + if out.len() + literal.len() > expected { + return Err(SnappyError::LengthMismatch); + } + out.extend_from_slice(literal); + pos += len; + } + kind => { + let (len, offset) = match kind { + 0b01 => { + let low = *input.get(pos).ok_or(SnappyError::Truncated)?; + pos += 1; + ( + 4 + usize::from((tag >> 2) & 0b111), + (usize::from(tag >> 5) << 8) | usize::from(low), + ) + } + 0b10 => { + let bytes = input.get(pos..pos + 2).ok_or(SnappyError::Truncated)?; + pos += 2; + ( + 1 + usize::from(tag >> 2), + usize::from(u16::from_le_bytes([bytes[0], bytes[1]])), + ) + } + _ => { + let bytes = input.get(pos..pos + 4).ok_or(SnappyError::Truncated)?; + pos += 4; + ( + 1 + usize::from(tag >> 2), + u32::from_le_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]) as usize, + ) + } + }; + copy_within_output(&mut out, offset, len, expected)?; + } + } + } + + if out.len() == expected { + Ok(out) + } else { + Err(SnappyError::LengthMismatch) + } +} + +/// Append `len` bytes copied from `offset` bytes back; the ranges may overlap (run-length style). +fn copy_within_output( + out: &mut Vec, + offset: usize, + len: usize, + expected: usize, +) -> Result<(), SnappyError> { + if offset == 0 || offset > out.len() { + return Err(SnappyError::BadOffset); + } + if out.len() + len > expected { + return Err(SnappyError::LengthMismatch); + } + let start = out.len() - offset; + for i in 0..len { + out.push(out[start + i]); + } + Ok(()) +} diff --git a/rust/src/browser/leveldb/table.rs b/rust/src/browser/leveldb/table.rs new file mode 100644 index 0000000000..30e8bacfe4 --- /dev/null +++ b/rust/src/browser/leveldb/table.rs @@ -0,0 +1,235 @@ +//! LevelDB sorted-table (`*.ldb` / `*.sst`) reader. +//! +//! Layout: data blocks, a metaindex block, an index block, and a 48 byte footer that ends with a +//! magic number. The footer holds block handles (varint offset, varint size) for the metaindex and +//! index blocks; index entries map a separator key to the handle of a data block. Every block is +//! followed by a 5 byte trailer (compression type, CRC32C). Entries inside a block share key +//! prefixes with the previous entry, and a restart array closes the block. +//! +//! Only the index and data blocks are read; the filter/metaindex blocks are irrelevant to a full +//! scan. Data-block keys are internal keys: the user key followed by `sequence << 8 | kind`. +//! Checksums are not verified (see `log.rs`); structural checks keep malformed input from +//! panicking or over-allocating. + +use super::snappy; +use super::varint::{read_u32_le, read_u64_le, read_varint32, read_varint64}; +use super::{MAX_BLOCK_BYTES, Record}; + +const FOOTER_SIZE: usize = 48; +const TABLE_MAGIC: u64 = 0xdb47_7524_8b80_fb57; +const BLOCK_TRAILER_SIZE: usize = 5; + +const COMPRESSION_NONE: u8 = 0; +const COMPRESSION_SNAPPY: u8 = 1; + +const KIND_DELETE: u64 = 0; +const KIND_VALUE: u64 = 1; + +#[derive(Debug, thiserror::Error)] +pub(super) enum TableError { + #[error("file is too short or has a bad table footer")] + BadFooter, + #[error("block handle points outside the file")] + BadHandle, + #[error("table block is {0} bytes, over the {1} byte limit")] + BlockTooLarge(usize, usize), + #[error("unsupported block compression type {0}")] + UnsupportedCompression(u8), + #[error("block is malformed")] + BadBlock, + #[error("snappy block failed to decode: {0}")] + Snappy(#[from] snappy::SnappyError), +} + +#[derive(Clone, Copy)] +struct BlockHandle { + offset: usize, + size: usize, +} + +impl BlockHandle { + /// Parse a handle from the front of `input`; returns it with the bytes consumed. + fn parse(input: &[u8]) -> Option<(Self, usize)> { + let (offset, first) = read_varint64(input)?; + let (size, second) = read_varint64(input.get(first..)?)?; + Some(( + Self { + offset: usize::try_from(offset).ok()?, + size: usize::try_from(size).ok()?, + }, + first + second, + )) + } +} + +/// Feed every put/delete found in the table `data` to `emit`. +/// +/// A data block that cannot be decoded is skipped (and counted in the returned failure count) so +/// one bad block does not hide the rest of the table; a bad footer or index block is an error. +pub(super) fn read_table(data: &[u8], emit: &mut impl FnMut(Record)) -> Result { + let footer_start = data + .len() + .checked_sub(FOOTER_SIZE) + .ok_or(TableError::BadFooter)?; + let footer = &data[footer_start..]; + if read_u64_le(footer, FOOTER_SIZE - 8) != Some(TABLE_MAGIC) { + return Err(TableError::BadFooter); + } + let footer_handles = &footer[..FOOTER_SIZE - 8]; + let (_metaindex, used) = BlockHandle::parse(footer_handles).ok_or(TableError::BadFooter)?; + let (index, _) = BlockHandle::parse(&footer_handles[used..]).ok_or(TableError::BadFooter)?; + + let index_block = read_block(data, index, footer_start)?; + let index_entries: Vec<_> = BlockEntries::new(&index_block)?.collect::>()?; + let mut skipped_blocks = 0usize; + for (_separator, handle_bytes) in index_entries { + let entry = BlockHandle::parse(handle_bytes) + .and_then(|(handle, used)| (used == handle_bytes.len()).then_some(handle)) + .ok_or(TableError::BadBlock); + let block = entry.and_then(|handle| read_block(data, handle, index.offset)); + let Ok(block) = block else { + skipped_blocks += 1; + continue; + }; + match decode_block_records(&block) { + Ok(records) => records.into_iter().for_each(&mut *emit), + Err(_) => skipped_blocks += 1, + } + } + Ok(skipped_blocks) +} + +fn decode_block_records(block: &[u8]) -> Result, TableError> { + let mut records = Vec::new(); + for entry in BlockEntries::new(block)? { + let (internal_key, value) = entry?; + let Some(split) = internal_key.len().checked_sub(8) else { + return Err(TableError::BadBlock); + }; + let (user_key, trailer) = internal_key.split_at(split); + let packed = read_u64_le(trailer, 0).ok_or(TableError::BadBlock)?; + let value = match packed & 0xff { + KIND_VALUE => Some(value.to_vec()), + KIND_DELETE => None, + _ => return Err(TableError::BadBlock), + }; + records.push(Record { + key: user_key.to_vec(), + sequence: packed >> 8, + value, + }); + } + Ok(records) +} + +/// Return the decompressed contents of the block at `handle` (without its trailer). +fn read_block(data: &[u8], handle: BlockHandle, upper_bound: usize) -> Result, TableError> { + if handle.offset >= upper_bound { + return Err(TableError::BadHandle); + } + let end = handle + .offset + .checked_add(handle.size) + .and_then(|end| end.checked_add(BLOCK_TRAILER_SIZE)) + .ok_or(TableError::BadHandle)?; + if end > upper_bound || data.len() < end { + return Err(TableError::BadHandle); + } + let contents_end = end - BLOCK_TRAILER_SIZE; + let raw = data + .get(handle.offset..contents_end) + .ok_or(TableError::BadHandle)?; + let compression = data[contents_end]; + match compression { + COMPRESSION_NONE if raw.len() <= MAX_BLOCK_BYTES => Ok(raw.to_vec()), + COMPRESSION_NONE => Err(TableError::BlockTooLarge(raw.len(), MAX_BLOCK_BYTES)), + COMPRESSION_SNAPPY => Ok(snappy::decompress(raw, MAX_BLOCK_BYTES)?), + other => Err(TableError::UnsupportedCompression(other)), + } +} + +/// Iterator over the `(key, value)` entries of a decoded block, undoing prefix compression. +struct BlockEntries<'a> { + entries: &'a [u8], + key: Vec, + failed: bool, +} + +impl<'a> BlockEntries<'a> { + fn new(block: &'a [u8]) -> Result { + let restart_count = + read_u32_le(block, block.len().saturating_sub(4)).ok_or(TableError::BadBlock)? as usize; + if restart_count == 0 { + return Err(TableError::BadBlock); + } + let restarts_len = restart_count + .checked_mul(4) + .and_then(|len| len.checked_add(4)) + .ok_or(TableError::BadBlock)?; + let entries_end = block + .len() + .checked_sub(restarts_len) + .ok_or(TableError::BadBlock)?; + let mut previous_restart = None; + for index in 0..restart_count { + let restart = + read_u32_le(block, entries_end + index * 4).ok_or(TableError::BadBlock)? as usize; + if (index == 0 && restart != 0) + || restart > entries_end + || (entries_end > 0 && restart == entries_end) + || previous_restart.is_some_and(|previous| restart <= previous) + { + return Err(TableError::BadBlock); + } + previous_restart = Some(restart); + } + Ok(Self { + entries: &block[..entries_end], + key: Vec::new(), + failed: false, + }) + } +} + +impl<'a> Iterator for BlockEntries<'a> { + type Item = Result<(Vec, &'a [u8]), TableError>; + + fn next(&mut self) -> Option { + if self.failed || self.entries.is_empty() { + return None; + } + let parsed = parse_entry(self.entries, &self.key); + match parsed { + Some((key, value, rest)) => { + self.key.clone_from(&key); + self.entries = rest; + Some(Ok((key, value))) + } + None => { + self.failed = true; + Some(Err(TableError::BadBlock)) + } + } + } +} + +/// Decode one entry: `shared`, `non_shared`, `value_len` varints, key delta, value. +fn parse_entry<'a>( + entries: &'a [u8], + previous_key: &[u8], +) -> Option<(Vec, &'a [u8], &'a [u8])> { + let (shared, a) = read_varint32(entries)?; + let (non_shared, b) = read_varint32(entries.get(a..)?)?; + let (value_len, c) = read_varint32(entries.get(a + b..)?)?; + let body = entries.get(a + b + c..)?; + let (shared, non_shared, value_len) = + (shared as usize, non_shared as usize, value_len as usize); + let payload_len = non_shared.checked_add(value_len)?; + if shared > previous_key.len() || body.len() < payload_len { + return None; + } + let mut key = Vec::with_capacity(shared + non_shared); + key.extend_from_slice(&previous_key[..shared]); + key.extend_from_slice(&body[..non_shared]); + Some((key, &body[non_shared..payload_len], &body[payload_len..])) +} diff --git a/rust/src/browser/leveldb/tests.rs b/rust/src/browser/leveldb/tests.rs new file mode 100644 index 0000000000..7110e9d215 --- /dev/null +++ b/rust/src/browser/leveldb/tests.rs @@ -0,0 +1,625 @@ +//! Tests for the LevelDB reader. Fixtures are built in-memory from the documented on-disk +//! formats (log block/record framing, table block/footer layout, Snappy elements). + +#![allow( + clippy::cast_possible_truncation, + reason = "fixture builders encode small in-memory sizes into fixed-width fields" +)] + +use super::local_storage::{ + LocalStorageEntry, decode_origin_entries, local_storage_dir, read_local_storage_entries, +}; +use super::snappy::{self, SnappyError}; +use super::table::{TableError, read_table}; +use super::{Entry, Record, log, read_entries}; + +// --------------------------------------------------------------------------------------------- +// Fixture builders +// --------------------------------------------------------------------------------------------- + +fn varint(mut value: u64) -> Vec { + let mut out = Vec::new(); + while value >= 0x80 { + out.push((value as u8 & 0x7f) | 0x80); + value >>= 7; + } + out.push(value as u8); + out +} + +/// A valid Snappy stream made only of literal elements (no back-references). +fn snappy_literal(data: &[u8]) -> Vec { + let mut out = varint(data.len() as u64); + for chunk in data.chunks(60) { + out.push(((chunk.len() - 1) as u8) << 2); + out.extend_from_slice(chunk); + } + out +} + +enum Op<'a> { + Put(&'a [u8], &'a [u8]), + Delete(&'a [u8]), +} + +fn write_batch(sequence: u64, ops: &[Op]) -> Vec { + let mut out = sequence.to_le_bytes().to_vec(); + out.extend_from_slice(&(ops.len() as u32).to_le_bytes()); + for op in ops { + match op { + Op::Put(key, value) => { + out.push(1); + out.extend(varint(key.len() as u64)); + out.extend_from_slice(key); + out.extend(varint(value.len() as u64)); + out.extend_from_slice(value); + } + Op::Delete(key) => { + out.push(0); + out.extend(varint(key.len() as u64)); + out.extend_from_slice(key); + } + } + } + out +} + +/// Frame batches as a LevelDB log: 32 KiB blocks, fragmenting records that cross a block edge. +fn write_log(batches: &[Vec]) -> Vec { + const BLOCK: usize = 32 * 1024; + let mut out = Vec::new(); + for batch in batches { + let mut remaining = batch.as_slice(); + let mut first = true; + loop { + let left_in_block = BLOCK - out.len() % BLOCK; + if left_in_block < 7 { + out.extend(std::iter::repeat_n(0u8, left_in_block)); + continue; + } + let chunk_len = remaining.len().min(left_in_block - 7); + let last = chunk_len == remaining.len(); + let kind: u8 = match (first, last) { + (true, true) => 1, + (true, false) => 2, + (false, false) => 3, + (false, true) => 4, + }; + out.extend_from_slice(&[0, 0, 0, 0]); // checksum (not verified by the reader) + out.extend_from_slice(&(chunk_len as u16).to_le_bytes()); + out.push(kind); + out.extend_from_slice(&remaining[..chunk_len]); + remaining = &remaining[chunk_len..]; + first = false; + if last { + break; + } + } + } + out +} + +fn internal_key(user_key: &[u8], sequence: u64, kind: u8) -> Vec { + let mut key = user_key.to_vec(); + key.extend_from_slice(&((sequence << 8) | u64::from(kind)).to_le_bytes()); + key +} + +/// Encode a block with prefix compression and a restart every `restart_interval` entries. +fn write_block(entries: &[(Vec, Vec)], restart_interval: usize) -> Vec { + let mut out = Vec::new(); + let mut restarts = vec![0u32]; + let mut previous: &[u8] = &[]; + for (index, (key, value)) in entries.iter().enumerate() { + let shared = if index % restart_interval == 0 { + if index > 0 { + restarts.push(out.len() as u32); + } + 0 + } else { + previous.iter().zip(key).take_while(|(a, b)| a == b).count() + }; + out.extend(varint(shared as u64)); + out.extend(varint((key.len() - shared) as u64)); + out.extend(varint(value.len() as u64)); + out.extend_from_slice(&key[shared..]); + out.extend_from_slice(value); + previous = key; + } + for restart in &restarts { + out.extend_from_slice(&restart.to_le_bytes()); + } + out.extend_from_slice(&(restarts.len() as u32).to_le_bytes()); + out +} + +#[derive(Clone, Copy, PartialEq)] +enum Compression { + None, + Snappy, +} + +/// A table record: user key, sequence, kind (1 = value, 0 = delete), value. +type TableRecord<'a> = (&'a [u8], u64, u8, &'a [u8]); + +/// Build a table with one data block per slice of `blocks`. +fn write_table(blocks: &[(&[TableRecord], Compression)]) -> Vec { + let mut file = Vec::new(); + let mut index_entries = Vec::new(); + for (records, compression) in blocks { + let entries: Vec<_> = records + .iter() + .map(|(key, sequence, kind, value)| { + (internal_key(key, *sequence, *kind), value.to_vec()) + }) + .collect(); + let block = write_block(&entries, 2); + let (contents, tag) = match compression { + Compression::None => (block, 0u8), + Compression::Snappy => (snappy_literal(&block), 1u8), + }; + let mut handle = varint(file.len() as u64); + handle.extend(varint(contents.len() as u64)); + file.extend_from_slice(&contents); + file.push(tag); + file.extend_from_slice(&[0, 0, 0, 0]); + index_entries.push((entries.last().unwrap().0.clone(), handle)); + } + + let index = write_block(&index_entries, 1); + let mut index_handle = varint(file.len() as u64); + index_handle.extend(varint(index.len() as u64)); + file.extend_from_slice(&index); + file.extend_from_slice(&[0, 0, 0, 0, 0]); + + let mut footer = varint(0); // metaindex handle: offset 0, size 0 (never read) + footer.extend(varint(0)); + footer.extend_from_slice(&index_handle); + footer.resize(40, 0); + footer.extend_from_slice(&0xdb47_7524_8b80_fb57u64.to_le_bytes()); + file.extend_from_slice(&footer); + file +} + +type Collected = Vec<(Vec, u64, Option>)>; + +fn collect_log(data: &[u8]) -> Collected { + let mut records = Vec::new(); + log::read_log(data, &mut |r: Record| { + records.push((r.key, r.sequence, r.value)); + }); + records +} + +fn collect_table(data: &[u8]) -> Result { + let mut records = Vec::new(); + read_table(data, &mut |r: Record| { + records.push((r.key, r.sequence, r.value)); + })?; + Ok(records) +} + +// --------------------------------------------------------------------------------------------- +// Snappy +// --------------------------------------------------------------------------------------------- + +#[test] +fn snappy_decodes_literal() { + let mut stream = vec![5, 4 << 2]; + stream.extend_from_slice(b"hello"); + assert_eq!(snappy::decompress(&stream, 1024).unwrap(), b"hello"); +} + +#[test] +fn snappy_decodes_long_literal_with_extra_length_byte() { + let data: Vec = (0..100u8).collect(); + let mut stream = varint(100); + stream.extend_from_slice(&[60 << 2, 99]); + stream.extend_from_slice(&data); + assert_eq!(snappy::decompress(&stream, 1024).unwrap(), data); +} + +#[test] +fn snappy_decodes_overlapping_copy_with_one_byte_offset() { + // "abc" literal, then copy of length 9 at offset 3 (copy-1 tag: len-4 in bits 2..5). + let stream = [12, 2 << 2, b'a', b'b', b'c', 0b01 | (5 << 2), 3]; + assert_eq!(snappy::decompress(&stream, 1024).unwrap(), b"abcabcabcabc"); +} + +#[test] +fn snappy_decodes_copy_with_two_byte_offset() { + let mut stream = vec![15, 9 << 2]; + stream.extend_from_slice(b"0123456789"); + stream.extend_from_slice(&[0b10 | (4 << 2), 10, 0]); + assert_eq!( + snappy::decompress(&stream, 1024).unwrap(), + b"012345678901234" + ); +} + +#[test] +fn snappy_decodes_copy_with_four_byte_offset() { + let mut stream = vec![14, 9 << 2]; + stream.extend_from_slice(b"0123456789"); + stream.extend_from_slice(&[0b11 | (3 << 2), 10, 0, 0, 0]); + assert_eq!( + snappy::decompress(&stream, 1024).unwrap(), + b"01234567890123" + ); +} + +#[test] +fn snappy_rejects_malformed_streams() { + assert_eq!(snappy::decompress(&[], 16), Err(SnappyError::BadPreamble)); + // Declared length larger than the limit is refused before allocating. + assert_eq!( + snappy::decompress(&varint(u64::from(u32::MAX)), 1024), + Err(SnappyError::TooLarge(u32::MAX as usize, 1024)) + ); + // Literal runs past the end of input. + assert_eq!( + snappy::decompress(&[5, 4 << 2, b'h'], 16), + Err(SnappyError::Truncated) + ); + // Copy before any output exists. + assert_eq!( + snappy::decompress(&[4, 0b01, 1], 16), + Err(SnappyError::BadOffset) + ); + // Zero offset is invalid. + assert_eq!( + snappy::decompress(&[5, 0, b'a', 0b01, 0], 16), + Err(SnappyError::BadOffset) + ); + // Output shorter than the preamble promised. + assert_eq!( + snappy::decompress(&[5, 0, b'a'], 16), + Err(SnappyError::LengthMismatch) + ); + // Output longer than the preamble promised. + assert_eq!( + snappy::decompress(&[1, 1 << 2, b'a', b'b'], 16), + Err(SnappyError::LengthMismatch) + ); +} + +// --------------------------------------------------------------------------------------------- +// Write-ahead log +// --------------------------------------------------------------------------------------------- + +#[test] +fn log_yields_puts_and_deletes_with_incrementing_sequences() { + let data = write_log(&[write_batch( + 10, + &[Op::Put(b"a", b"1"), Op::Delete(b"b"), Op::Put(b"c", b"3")], + )]); + assert_eq!( + collect_log(&data), + vec![ + (b"a".to_vec(), 10, Some(b"1".to_vec())), + (b"b".to_vec(), 11, None), + (b"c".to_vec(), 12, Some(b"3".to_vec())), + ] + ); +} + +#[test] +fn log_reassembles_records_that_span_blocks() { + let big = vec![0xabu8; 80 * 1024]; + let data = write_log(&[ + write_batch(1, &[Op::Put(b"small", b"x")]), + write_batch(2, &[Op::Put(b"big", &big)]), + write_batch(3, &[Op::Put(b"after", b"y")]), + ]); + assert!(data.len() > 64 * 1024); + let records = collect_log(&data); + assert_eq!(records.len(), 3); + assert_eq!(records[1].2.as_deref(), Some(big.as_slice())); + assert_eq!(records[2].0, b"after"); +} + +#[test] +fn log_skips_block_trailer_padding() { + // Leave fewer than 7 bytes at the end of the first block so the writer pads it. + let filler = vec![1u8; 32 * 1024 - 7 - 12 - 1 - 2 - 1 - 1 - 4]; + let data = write_log(&[ + write_batch(1, &[Op::Put(b"k", &filler)]), + write_batch(2, &[Op::Put(b"z", b"after-padding")]), + ]); + let records = collect_log(&data); + assert_eq!(records.len(), 2); + assert_eq!(records[1].2.as_deref(), Some(b"after-padding".as_slice())); +} + +#[test] +fn log_keeps_records_before_a_truncated_tail() { + let mut data = write_log(&[ + write_batch(1, &[Op::Put(b"kept", b"1")]), + write_batch(2, &[Op::Put(b"lost", b"2")]), + ]); + data.truncate(data.len() - 3); + let records = collect_log(&data); + assert_eq!(records.len(), 1); + assert_eq!(records[0].0, b"kept"); +} + +#[test] +fn log_ignores_garbage_without_panicking() { + assert!(collect_log(&[0xff; 100]).is_empty()); + assert!(collect_log(&[]).is_empty()); + // Middle fragment with no preceding first fragment. + let orphan = [0, 0, 0, 0, 1, 0, 3, 9]; + assert!(collect_log(&orphan).is_empty()); +} + +// --------------------------------------------------------------------------------------------- +// Tables +// --------------------------------------------------------------------------------------------- + +#[test] +fn table_reads_prefix_compressed_multi_block_data() { + let first: &[TableRecord] = &[ + (b"app:alpha", 5, 1, b"one"), + (b"app:alpha2", 6, 1, b"two"), + (b"app:beta", 7, 1, b"three"), + ]; + let second: &[TableRecord] = &[(b"zeta", 8, 1, b"four"), (b"zeta-gone", 9, 0, b"")]; + let data = write_table(&[(first, Compression::None), (second, Compression::None)]); + + let records = collect_table(&data).unwrap(); + assert_eq!(records.len(), 5); + assert_eq!( + records[1], + (b"app:alpha2".to_vec(), 6, Some(b"two".to_vec())) + ); + assert_eq!(records[4], (b"zeta-gone".to_vec(), 9, None)); +} + +#[test] +fn table_reads_snappy_compressed_blocks() { + let long_value = vec![b'v'; 500]; + let records: &[TableRecord] = &[(b"k1", 1, 1, &long_value), (b"k2", 2, 1, b"short")]; + let data = write_table(&[(records, Compression::Snappy)]); + + let decoded = collect_table(&data).unwrap(); + assert_eq!(decoded.len(), 2); + assert_eq!(decoded[0].2.as_deref(), Some(long_value.as_slice())); +} + +#[test] +fn table_with_bad_magic_or_length_is_rejected() { + let records: &[TableRecord] = &[(b"k", 1, 1, b"v")]; + let mut data = write_table(&[(records, Compression::None)]); + assert!(matches!( + collect_table(&data[..20]), + Err(TableError::BadFooter) + )); + let last = data.len() - 1; + data[last] ^= 0xff; + assert!(matches!(collect_table(&data), Err(TableError::BadFooter))); +} + +#[test] +fn table_skips_an_undecodable_block_but_keeps_the_rest() { + let good: &[TableRecord] = &[(b"good", 1, 1, b"ok")]; + let bad: &[TableRecord] = &[(b"bad", 2, 1, b"nope")]; + let mut data = write_table(&[(bad, Compression::Snappy), (good, Compression::None)]); + // Corrupt the first block's compression tag to an unsupported type (2). + let bad_block_len = snappy_literal(&write_block( + &[(internal_key(b"bad", 2, 1), b"nope".to_vec())], + 2, + )) + .len(); + data[bad_block_len] = 2; + + let mut records = Vec::new(); + let skipped = read_table(&data, &mut |r: Record| records.push(r.key)).unwrap(); + assert_eq!(skipped, 1); + assert_eq!(records, vec![b"good".to_vec()]); +} + +#[test] +fn table_with_out_of_range_handle_does_not_panic() { + let records: &[TableRecord] = &[(b"k", 1, 1, b"v")]; + let mut data = write_table(&[(records, Compression::None)]); + // Point the index handle far past the end of the file. + let footer = data.len() - 48; + data[footer..footer + 40].fill(0); + data[footer + 2..footer + 6].copy_from_slice(&[0xff, 0xff, 0xff, 0x0f]); + assert!(collect_table(&data).is_err()); +} + +// --------------------------------------------------------------------------------------------- +// Directory reader +// --------------------------------------------------------------------------------------------- + +#[test] +fn read_entries_resolves_newest_sequence_across_log_and_table() { + let dir = tempfile::tempdir().unwrap(); + let table_records: &[TableRecord] = &[ + (b"keep", 1, 1, b"table-keep"), + (b"stale", 2, 1, b"table-old"), + (b"removed", 3, 1, b"table-removed"), + ]; + std::fs::write( + dir.path().join("000005.ldb"), + write_table(&[(table_records, Compression::None)]), + ) + .unwrap(); + std::fs::write( + dir.path().join("000006.log"), + write_log(&[write_batch( + 10, + &[ + Op::Put(b"stale", b"log-new"), + Op::Delete(b"removed"), + Op::Put(b"fresh", b"log-fresh"), + ], + )]), + ) + .unwrap(); + // Bookkeeping files and unrelated data are ignored. + std::fs::write(dir.path().join("LOCK"), b"").unwrap(); + std::fs::write(dir.path().join("CURRENT"), b"MANIFEST-000001\n").unwrap(); + std::fs::write(dir.path().join("notes.txt"), b"not a database").unwrap(); + + let entries = read_entries(dir.path()).unwrap(); + let pairs: Vec<(&[u8], &[u8])> = entries + .iter() + .map(|e| (e.key.as_slice(), e.value.as_slice())) + .collect(); + assert_eq!( + pairs, + vec![ + (b"fresh".as_slice(), b"log-fresh".as_slice()), + (b"keep".as_slice(), b"table-keep".as_slice()), + (b"stale".as_slice(), b"log-new".as_slice()), + ] + ); +} + +#[test] +fn read_entries_lets_a_newer_table_delete_beat_an_older_log_put() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write( + dir.path().join("000001.log"), + write_log(&[write_batch(1, &[Op::Put(b"gone", b"old")])]), + ) + .unwrap(); + let deletes: &[TableRecord] = &[(b"gone", 2, 0, b"")]; + std::fs::write( + dir.path().join("000002.ldb"), + write_table(&[(deletes, Compression::None)]), + ) + .unwrap(); + assert!(read_entries(dir.path()).unwrap().is_empty()); +} + +#[test] +fn read_entries_skips_corrupt_files_and_keeps_good_ones() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write(dir.path().join("000001.ldb"), b"definitely not a table").unwrap(); + std::fs::write( + dir.path().join("000002.log"), + write_log(&[write_batch(1, &[Op::Put(b"ok", b"1")])]), + ) + .unwrap(); + assert_eq!( + read_entries(dir.path()).unwrap(), + vec![Entry { + key: b"ok".to_vec(), + value: b"1".to_vec() + }] + ); +} + +#[test] +fn read_entries_reports_a_missing_directory() { + let dir = tempfile::tempdir().unwrap(); + assert!(read_entries(&dir.path().join("absent")).is_err()); +} + +// --------------------------------------------------------------------------------------------- +// Chromium Local Storage layer +// --------------------------------------------------------------------------------------------- + +fn latin1(text: &str) -> Vec { + let mut out = vec![1u8]; + out.extend(text.chars().map(|c| c as u8)); + out +} + +fn utf16(text: &str) -> Vec { + let mut out = vec![0u8]; + out.extend(text.encode_utf16().flat_map(u16::to_le_bytes)); + out +} + +fn storage_key(origin: &str, encoded_key: &[u8]) -> Vec { + let mut key = format!("_{origin}").into_bytes(); + key.push(0); + key.extend_from_slice(encoded_key); + key +} + +#[test] +fn local_storage_decodes_latin1_and_utf16_items_for_one_origin() { + let entries = vec![ + Entry { + key: b"META:https://www.kimi.ai".to_vec(), + value: b"\x08\x01".to_vec(), + }, + Entry { + key: storage_key("https://www.kimi.ai", &latin1("access_token")), + value: latin1("eyJ.body.sig"), + }, + Entry { + key: storage_key("https://www.kimi.ai", &utf16("name")), + value: utf16("Zoë \u{1F600}"), + }, + Entry { + key: storage_key("https://www.kimi.com", &latin1("access_token")), + value: latin1("other-origin"), + }, + Entry { + key: storage_key("https://www.kimi.ai", &latin1("bad_format")), + value: vec![7, b'x'], + }, + ]; + + let decoded = decode_origin_entries(&entries, "https://www.kimi.ai/"); + assert_eq!( + decoded, + vec![ + LocalStorageEntry { + key: "access_token".into(), + value: "eyJ.body.sig".into() + }, + LocalStorageEntry { + key: "name".into(), + value: "Zoë \u{1F600}".into() + }, + ] + ); +} + +#[test] +fn local_storage_origin_match_is_exact_not_a_prefix() { + let entries = vec![Entry { + key: storage_key("https://www.kimi.ai.evil.example", &latin1("access_token")), + value: latin1("nope"), + }]; + assert!(decode_origin_entries(&entries, "https://www.kimi.ai").is_empty()); +} + +#[test] +fn local_storage_rejects_odd_length_utf16_values() { + let entries = vec![Entry { + key: storage_key("https://a.example", &latin1("k")), + value: vec![0, b'x'], + }]; + assert!(decode_origin_entries(&entries, "https://a.example").is_empty()); +} + +#[test] +fn read_local_storage_entries_reads_a_profile_directory() { + let profile = tempfile::tempdir().unwrap(); + let dir = local_storage_dir(profile.path()); + std::fs::create_dir_all(&dir).unwrap(); + + let key = storage_key("https://www.kimi.ai", &latin1("access_token")); + let value = latin1("a.b.c"); + std::fs::write( + dir.join("000003.log"), + write_log(&[write_batch(4, &[Op::Put(&key, &value)])]), + ) + .unwrap(); + + assert_eq!( + read_local_storage_entries(&dir, "https://www.kimi.ai").unwrap(), + vec![LocalStorageEntry { + key: "access_token".into(), + value: "a.b.c".into() + }] + ); +} diff --git a/rust/src/browser/leveldb/varint.rs b/rust/src/browser/leveldb/varint.rs new file mode 100644 index 0000000000..c51511a711 --- /dev/null +++ b/rust/src/browser/leveldb/varint.rs @@ -0,0 +1,41 @@ +//! LevelDB varint and fixed-width helpers. + +/// Decode a little-endian base-128 varint of at most 32 bits; returns the value and bytes used. +pub(super) fn read_varint32(input: &[u8]) -> Option<(u32, usize)> { + let (value, used) = read_varint64(input)?; + Some((u32::try_from(value).ok()?, used)) +} + +/// Decode a little-endian base-128 varint of at most 64 bits; returns the value and bytes used. +pub(super) fn read_varint64(input: &[u8]) -> Option<(u64, usize)> { + let mut value = 0u64; + for (index, byte) in input.iter().take(10).enumerate() { + let bits = u64::from(byte & 0x7f); + if index == 9 && bits > 1 { + return None; + } + value |= bits << (7 * index); + if byte & 0x80 == 0 { + return Some((value, index + 1)); + } + } + None +} + +/// Split a varint-length-prefixed slice off the front of `input`, returning `(slice, rest)`. +pub(super) fn read_length_prefixed(input: &[u8]) -> Option<(&[u8], &[u8])> { + let (len, used) = read_varint32(input)?; + let rest = &input[used..]; + let len = len as usize; + (rest.len() >= len).then(|| rest.split_at(len)) +} + +pub(super) fn read_u32_le(input: &[u8], at: usize) -> Option { + let bytes = input.get(at..at.checked_add(4)?)?; + Some(u32::from_le_bytes(bytes.try_into().ok()?)) +} + +pub(super) fn read_u64_le(input: &[u8], at: usize) -> Option { + let bytes = input.get(at..at.checked_add(8)?)?; + Some(u64::from_le_bytes(bytes.try_into().ok()?)) +} diff --git a/rust/src/browser/mod.rs b/rust/src/browser/mod.rs index f428e1b41a..035fb6e6d2 100755 --- a/rust/src/browser/mod.rs +++ b/rust/src/browser/mod.rs @@ -3,6 +3,7 @@ pub mod cookie_cache; pub mod cookies; pub mod detection; +pub mod leveldb; pub mod watchdog; pub mod wsl_paths;