Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
70 changes: 70 additions & 0 deletions rust/src/browser/leveldb/local_storage.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,70 @@
//! Chromium `Local Storage` layer on top of the LevelDB reader.
//!
//! Chromium stores each origin's `localStorage` items under the key
//! `_<origin>\0<key>` where `<origin>` is the serialized origin (`https://www.kimi.ai`) and
//! `<key>` starts with a format byte: `0x00` for UTF-16LE text, `0x01` for Latin-1. Values use
//! the same format byte. Other keys in the database (`VERSION`, `META:<origin>`,
//! `METAACCESS:<origin>`) are bookkeeping and are ignored.

use super::{Entry, LevelDbError, read_entries};
use std::path::{Path, PathBuf};

const KEY_PREFIX: u8 = b'_';
const ORIGIN_TERMINATOR: u8 = 0;
const FORMAT_UTF16LE: u8 = 0;
const FORMAT_LATIN1: u8 = 1;

/// One decoded `localStorage` item.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct LocalStorageEntry {
pub key: String,
pub value: String,
}

/// Directory holding Local Storage for a Chromium profile directory (`Default`, `Profile 1`, ...).
pub fn local_storage_dir(profile_dir: &Path) -> PathBuf {
profile_dir.join("Local Storage").join("leveldb")
}

/// Read the `localStorage` items stored for `origin` (for example `https://www.kimi.ai`) in the
/// LevelDB directory `dir`, sorted by key. A trailing slash on `origin` is ignored.
pub fn read_local_storage_entries(
dir: &Path,
origin: &str,
) -> Result<Vec<LocalStorageEntry>, LevelDbError> {
Ok(decode_origin_entries(&read_entries(dir)?, origin))
}

pub(super) fn decode_origin_entries(entries: &[Entry], origin: &str) -> Vec<LocalStorageEntry> {
let mut prefix = Vec::with_capacity(origin.len() + 2);
prefix.push(KEY_PREFIX);
prefix.extend_from_slice(origin.trim_end_matches('/').as_bytes());
prefix.push(ORIGIN_TERMINATOR);

entries
.iter()
.filter_map(|entry| {
let key = decode_text(entry.key.strip_prefix(prefix.as_slice())?)?;
let value = decode_text(&entry.value)?;
Some(LocalStorageEntry { key, value })
})
.collect()
}

/// Decode a format-byte-prefixed Chromium string; `None` for an unknown format or bad UTF-16.
fn decode_text(bytes: &[u8]) -> Option<String> {
let (&format, data) = bytes.split_first()?;
match format {
FORMAT_LATIN1 => Some(data.iter().map(|&byte| char::from(byte)).collect()),
FORMAT_UTF16LE if data.len() % 2 == 0 => {
let units: Vec<u16> = data
.as_chunks::<2>()
.0
.iter()
.map(|pair| u16::from_le_bytes(*pair))
.collect();
String::from_utf16(&units).ok()
}
_ => None,
}
}
122 changes: 122 additions & 0 deletions rust/src/browser/leveldb/log.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,122 @@
//! LevelDB write-ahead log (`*.log`) reader.
//!
//! A log is a sequence of 32 KiB blocks holding physical records (7 byte header: masked CRC32C,
//! length, type) that are reassembled into logical records. Each logical record is a write batch:
//! `sequence: u64le`, `count: u32le`, then `count` operations (`1 key value` puts and `0 key`
//! deletes, with varint-length-prefixed byte strings). Operation `i` has sequence `sequence + i`.
//!
//! The live log of a running browser routinely ends in a half-written record, so a truncated or
//! malformed tail ends the scan quietly and everything before it is kept. Record checksums are
//! not verified: this is a best-effort read of another program's cache, not a database recovery.

use super::Record;
use super::varint::{read_length_prefixed, read_u32_le, read_u64_le};

const BLOCK_SIZE: usize = 32 * 1024;
const HEADER_SIZE: usize = 7;
const BATCH_HEADER_SIZE: usize = 12;

const TYPE_ZERO: u8 = 0;
const TYPE_FULL: u8 = 1;
const TYPE_FIRST: u8 = 2;
const TYPE_MIDDLE: u8 = 3;
const TYPE_LAST: u8 = 4;

const OP_DELETE: u8 = 0;
const OP_PUT: u8 = 1;

/// Feed every put/delete found in `data` to `emit`.
pub(super) fn read_log(data: &[u8], emit: &mut impl FnMut(Record)) {
let mut assembled: Vec<u8> = Vec::new();
let mut in_fragmented = false;

let mut offset = 0usize;
while offset < data.len() {
let block_remaining = BLOCK_SIZE - (offset % BLOCK_SIZE);
if block_remaining < HEADER_SIZE {
offset += block_remaining;
continue;
}
let Some(header) = data.get(offset..offset + HEADER_SIZE) else {
return;
};
let length = usize::from(u16::from_le_bytes([header[4], header[5]]));
let kind = header[6];
if kind == TYPE_ZERO && length == 0 {
// Zero padding: the rest of this block is unused.
offset += block_remaining;
continue;
}
let payload_start = offset + HEADER_SIZE;
if HEADER_SIZE + length > block_remaining {
return;
}
let Some(payload) = data.get(payload_start..payload_start + length) else {
return;
};
offset = payload_start + length;

match kind {
TYPE_FULL => {
assembled.clear();
in_fragmented = false;
read_batch(payload, emit);
}
TYPE_FIRST => {
assembled.clear();
assembled.extend_from_slice(payload);
in_fragmented = true;
}
TYPE_MIDDLE if in_fragmented => assembled.extend_from_slice(payload),
TYPE_LAST if in_fragmented => {
assembled.extend_from_slice(payload);
in_fragmented = false;
read_batch(&assembled, emit);
assembled.clear();
}
_ => return,
}
}
}

fn read_batch(batch: &[u8], emit: &mut impl FnMut(Record)) {
let (Some(sequence), Some(count)) = (read_u64_le(batch, 0), read_u32_le(batch, 8)) else {
return;
};
let mut rest = batch.get(BATCH_HEADER_SIZE..).unwrap_or_default();
let mut records = Vec::new();
for index in 0..u64::from(count) {
let Some((&op, after_op)) = rest.split_first() else {
return;
};
let Some((key, after_key)) = read_length_prefixed(after_op) else {
return;
};
let value = match op {
OP_PUT => {
let Some((value, after_value)) = read_length_prefixed(after_key) else {
return;
};
rest = after_value;
Some(value.to_vec())
}
OP_DELETE => {
rest = after_key;
None
}
_ => return,
};
let Some(sequence) = sequence.checked_add(index) else {
return;
};
records.push(Record {
key: key.to_vec(),
sequence,
value,
});
}
if !rest.is_empty() {
return;
}
records.into_iter().for_each(emit);
}
144 changes: 144 additions & 0 deletions rust/src/browser/leveldb/mod.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,144 @@
//! Read-only LevelDB reader for Chromium browser storage (Local Storage, and other LevelDB
//! directories a browser keeps under a profile).
//!
//! This is a small hand-written reader rather than a dependency: it understands the write-ahead
//! log and sorted-table formats, including Snappy-compressed table blocks, and resolves each user
//! key to its newest value by sequence number. It does not open the database, take its `LOCK`, or
//! write anything, so it can be pointed at a profile of a running browser.
//!
//! Known limits, all acceptable for a best-effort credential import:
//! - The `MANIFEST` is not consulted, so a table that compaction already obsoleted but the
//! browser has not yet removed can still be read. Newer sequence numbers win, so this only
//! matters for a deleted key whose tombstone was compacted away.
//! - Checksums are not verified and only the bytewise comparator is assumed (which is what
//! `Local Storage` uses; a full scan does not depend on ordering anyway).
//! - Blocks using a compression type other than none/Snappy are skipped.
//! - Files larger than [`MAX_FILE_BYTES`] and decoded table blocks larger than
//! [`MAX_BLOCK_BYTES`] are skipped to bound each input allocation. The returned snapshot still
//! grows with the database's live data.

pub mod local_storage;
mod log;
pub mod snappy;
mod table;
mod varint;

#[cfg(test)]
mod tests;

use std::collections::BTreeMap;
use std::ffi::OsStr;
use std::io::Read;
use std::path::Path;

/// Largest single log or table file that will be read into memory.
pub const MAX_FILE_BYTES: u64 = 64 * 1024 * 1024;
/// Largest decoded table block accepted.
pub const MAX_BLOCK_BYTES: usize = 16 * 1024 * 1024;

/// One live key/value pair of the database.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Entry {
pub key: Vec<u8>,
pub value: Vec<u8>,
}

/// Failure to read a LevelDB directory at all (per-file problems are skipped, not reported).
#[derive(Debug, thiserror::Error)]
pub enum LevelDbError {
#[error("cannot read LevelDB directory: {0}")]
Io(#[from] std::io::Error),
}

/// A put (`value: Some`) or delete (`value: None`) with the sequence number it was written at.
pub(crate) struct Record {
pub(crate) key: Vec<u8>,
pub(crate) sequence: u64,
pub(crate) value: Option<Vec<u8>>,
}

/// Read every live entry of the LevelDB directory `dir`, sorted by key.
///
/// Keys that were deleted (or whose newest record is a delete) are omitted. Unreadable or corrupt
/// files are skipped with a debug log so the remaining data is still returned.
pub fn read_entries(dir: &Path) -> Result<Vec<Entry>, LevelDbError> {
let mut newest: BTreeMap<Vec<u8>, (u64, Option<Vec<u8>>)> = BTreeMap::new();
let mut absorb = |record: Record| match newest.get_mut(&record.key) {
Some(existing) if existing.0 >= record.sequence => {}
Some(existing) => *existing = (record.sequence, record.value),
None => {
newest.insert(record.key, (record.sequence, record.value));
}
};

for dir_entry in std::fs::read_dir(dir)? {
let Ok(dir_entry) = dir_entry else { continue };
let path = dir_entry.path();
let Some(kind) = FileKind::of(&path) else {
continue;
};
let data = match read_bounded_file(&path) {
Ok(data) => data,
Err(error) => {
tracing::debug!(file = ?path.file_name(), %error, "skipping unreadable LevelDB file");
continue;
}
};
match kind {
FileKind::Log => log::read_log(&data, &mut absorb),
FileKind::Table => match table::read_table(&data, &mut absorb) {
Ok(0) => {}
Ok(skipped) => tracing::debug!(
file = ?path.file_name(),
skipped,
"skipped undecodable LevelDB table blocks"
),
Err(error) => {
tracing::debug!(file = ?path.file_name(), %error, "skipping malformed LevelDB table");
}
},
}
}

Ok(newest
.into_iter()
.filter_map(|(key, (_, value))| value.map(|value| Entry { key, value }))
.collect())
}

#[derive(Clone, Copy)]
enum FileKind {
Log,
Table,
}

impl FileKind {
fn of(path: &Path) -> Option<Self> {
let extension = path.extension().and_then(OsStr::to_str)?;
if extension.eq_ignore_ascii_case("log") {
Some(Self::Log)
} else if extension.eq_ignore_ascii_case("ldb") || extension.eq_ignore_ascii_case("sst") {
Some(Self::Table)
} else {
None
}
}
}

fn read_bounded_file(path: &Path) -> std::io::Result<Vec<u8>> {
let file = std::fs::File::open(path)?;
let size = file.metadata()?.len();
if size > MAX_FILE_BYTES {
return Err(std::io::Error::other(format!(
"file is {size} bytes, over the {MAX_FILE_BYTES} byte limit"
)));
}
let mut data = Vec::new();
file.take(MAX_FILE_BYTES + 1).read_to_end(&mut data)?;
if data.len() as u64 > MAX_FILE_BYTES {
return Err(std::io::Error::other(format!(
"file grew beyond the {MAX_FILE_BYTES} byte limit while being read"
)));
}
Ok(data)
}
Loading