AT Protocol PDS + AppView + Tauri Desktop Client, 160-char post limit. - PDS (Rust + axum + sqlx) - Auth: createAccount, createSession, refreshSession - Records: createRecord, deleteRecord (race-safe via SELECT FOR UPDATE) - Feed: feed.like.create, feed.repost.create - Sync: getRepo, getBlocks, getLatestCommit, getRecord (with MST proof), listRepos - Identity: resolveHandle - MST: spec-conformant (at-mst crate, 27 tests) - Repo: signed commits, TID counter (monotonic, 4096 wrap safe) - AppView (Rust + axum + sqlx) - Jetstream consumer (WebSocket, exponential backoff, 38k+ events indexed) - REST API: timeline/home (graph-aware), profile, search, post (with thread hydration) - Handle-sync worker (did:plc + did:web) - JSONB embed storage + thread columns (migration 0003) - Like/repost counter cache (migration 0004) - Tauri 2 + Svelte 5 Desktop Client - System tray (Show/Compose/Quit menu) - OS notifications (tauri-plugin-notification) - Auto-update (tauri-plugin-updater, placeholder endpoint) - Window-state (tauri-plugin-window-state) - 160-char compose with live counter - Image/Link embed rendering - LocalStorage-persisted like state - Timeline with poll (prepend new posts) - Custom TitleBar (transparent, no decorations) - Orange/IBM Plex Mono maarcade design Tests: 231 Rust + 9 vitest = 240 passed.
480 lines
15 KiB
Rust
480 lines
15 KiB
Rust
//! CAR v1 writer for atproto sync endpoints.
|
|
//!
|
|
//! The on-the-wire format follows
|
|
//! <https://ipld.io/specs/transport/car/carv1/> and is the same format used by
|
|
//! `com.atproto.sync.getRepo`, `getBlocks`, `getLatestCommit` and
|
|
//! `getRecord`.
|
|
//!
|
|
//! Layout:
|
|
//!
|
|
//! ```text
|
|
//! [ varint: header_len | DAG-CBOR header block ] (header)
|
|
//! [ varint: section_len | CID | block bytes ] (block 1)
|
|
//! [ varint: section_len | CID | block bytes ] (block 2)
|
|
//! ...
|
|
//! ```
|
|
//!
|
|
//! The header is `{ version: 1, roots: [CID, ...] }` encoded as DAG-CBOR. In
|
|
//! DAG-CBOR CID links carry the IANA-registered CBOR tag `42`, which the
|
|
//! `ciborium` crate does not emit for `cid::Cid` (it uses serde newtype-struct
|
|
//! tagging instead). We hand-encode the header bytes to keep the file
|
|
//! spec-compliant: a `Map(2)` with text keys `"version"` and `"roots"`, an
|
|
//! unsigned int `1` for the version, and a tagged byte string for each root
|
|
//! CID.
|
|
//!
|
|
//! Per the spec, CAR v1 stores the raw CID bytes (varint version + codec +
|
|
//! multihash) prefixed to every block, with a leading varint giving the total
|
|
//! length of the section (CID + block).
|
|
|
|
use anyhow::Result;
|
|
use cid::Cid;
|
|
|
|
/// Encode an unsigned CBOR head (major type in upper 3 bits) with a value.
|
|
///
|
|
/// Supports values up to `u32::MAX` which is more than enough for any realistic
|
|
/// header or array length.
|
|
fn cbor_head(out: &mut Vec<u8>, major: u8, n: u64) {
|
|
let m = (major & 0x07) << 5;
|
|
if n < 24 {
|
|
out.push(m | n as u8);
|
|
} else if n < 0x100 {
|
|
out.push(m | 24);
|
|
out.push(n as u8);
|
|
} else if n < 0x10000 {
|
|
out.push(m | 25);
|
|
out.push((n >> 8) as u8);
|
|
out.push(n as u8);
|
|
} else if n < 0x100_0000 {
|
|
out.push(m | 26);
|
|
out.push((n >> 16) as u8);
|
|
out.push((n >> 8) as u8);
|
|
out.push(n as u8);
|
|
} else {
|
|
out.push(m | 27);
|
|
out.push((n >> 24) as u8);
|
|
out.push((n >> 16) as u8);
|
|
out.push((n >> 8) as u8);
|
|
out.push(n as u8);
|
|
}
|
|
}
|
|
|
|
/// Append a CBOR text string.
|
|
fn cbor_text(out: &mut Vec<u8>, s: &str) {
|
|
cbor_head(out, 3, s.len() as u64);
|
|
out.extend_from_slice(s.as_bytes());
|
|
}
|
|
|
|
/// Append a CBOR byte string.
|
|
fn cbor_bytes(out: &mut Vec<u8>, b: &[u8]) {
|
|
cbor_head(out, 2, b.len() as u64);
|
|
out.extend_from_slice(b);
|
|
}
|
|
|
|
/// Append a CBOR tag wrapping the following value.
|
|
fn cbor_tag(out: &mut Vec<u8>, tag: u64) {
|
|
cbor_head(out, 6, tag);
|
|
}
|
|
|
|
/// Encode the CAR v1 DAG-CBOR header `{ version: 1, roots: [CID, ...] }`.
|
|
///
|
|
/// CIDs are encoded as `tag(42) + bytes(<raw-cid-bytes>)` per the DAG-CBOR
|
|
/// spec. This is the canonical IPLD CID-link form.
|
|
pub fn encode_header(roots: &[Cid]) -> Vec<u8> {
|
|
let mut out = Vec::new();
|
|
// Map(2): { "version": 1, "roots": [...] }
|
|
cbor_head(&mut out, 5, 2);
|
|
cbor_text(&mut out, "version");
|
|
cbor_head(&mut out, 0, 1);
|
|
cbor_text(&mut out, "roots");
|
|
cbor_head(&mut out, 4, roots.len() as u64);
|
|
for cid in roots {
|
|
cbor_tag(&mut out, 42);
|
|
cbor_bytes(&mut out, &cid.to_bytes());
|
|
}
|
|
out
|
|
}
|
|
|
|
/// Append a varint to `out` using LEB128 unsigned encoding.
|
|
fn write_varint(out: &mut Vec<u8>, n: u64) {
|
|
let mut buf = unsigned_varint::encode::u64_buffer();
|
|
let bytes = unsigned_varint::encode::u64(n, &mut buf);
|
|
out.extend_from_slice(bytes);
|
|
}
|
|
|
|
/// A single (CID, block_bytes) pair held in a [`CarWriter`].
|
|
#[derive(Debug, Clone)]
|
|
pub struct Block {
|
|
pub cid: Cid,
|
|
pub data: Vec<u8>,
|
|
}
|
|
|
|
/// Buffer for assembling a CAR v1 file.
|
|
///
|
|
/// Usage:
|
|
///
|
|
/// ```ignore
|
|
/// let mut w = CarWriter::new();
|
|
/// w.append(cid_a, &block_a);
|
|
/// w.append(cid_b, &block_b);
|
|
/// let bytes = w.finish(&[head_commit_cid]);
|
|
/// ```
|
|
///
|
|
/// The header's `roots` is provided at `finish` time so callers can defer
|
|
/// deciding what the root is until all blocks are queued.
|
|
#[derive(Debug, Default, Clone)]
|
|
pub struct CarWriter {
|
|
blocks: Vec<Block>,
|
|
}
|
|
|
|
impl CarWriter {
|
|
pub fn new() -> Self {
|
|
Self::default()
|
|
}
|
|
|
|
/// Append a (CID, block) pair. Duplicate CIDs are de-duplicated: the first
|
|
/// occurrence wins. CAR v1 allows duplicate blocks in principle but for
|
|
/// repo exports the spec says the root CID is unique and our callers don't
|
|
/// need to write the same block twice.
|
|
pub fn append(&mut self, cid: Cid, data: &[u8]) {
|
|
if self.blocks.iter().any(|b| b.cid == cid) {
|
|
return;
|
|
}
|
|
self.blocks.push(Block {
|
|
cid,
|
|
data: data.to_vec(),
|
|
});
|
|
}
|
|
|
|
/// Finalize the CAR stream. Writes the header followed by every queued
|
|
/// block as a length-prefixed CID+data section.
|
|
pub fn finish(&self, roots: &[Cid]) -> Vec<u8> {
|
|
let header = encode_header(roots);
|
|
let mut out = Vec::with_capacity(header.len() + self.blocks.len() * 64);
|
|
write_varint(&mut out, header.len() as u64);
|
|
out.extend_from_slice(&header);
|
|
for b in &self.blocks {
|
|
let cid_bytes = b.cid.to_bytes();
|
|
// Section length is the combined length of CID bytes + block data.
|
|
let section_len = (cid_bytes.len() + b.data.len()) as u64;
|
|
write_varint(&mut out, section_len);
|
|
out.extend_from_slice(&cid_bytes);
|
|
out.extend_from_slice(&b.data);
|
|
}
|
|
out
|
|
}
|
|
|
|
#[allow(dead_code)]
|
|
pub fn len(&self) -> usize {
|
|
self.blocks.len()
|
|
}
|
|
|
|
#[allow(dead_code)]
|
|
pub fn is_empty(&self) -> bool {
|
|
self.blocks.is_empty()
|
|
}
|
|
}
|
|
|
|
// -- minimal CAR reader (for tests / debug) --------------------------------
|
|
|
|
/// Header parsed out of a CAR file. `roots` are kept as raw CID byte vectors
|
|
/// so callers can re-parse them however they like.
|
|
#[allow(dead_code)]
|
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
|
pub struct CarHeader {
|
|
pub version: u64,
|
|
pub roots: Vec<Cid>,
|
|
}
|
|
|
|
/// A block parsed from a CAR file.
|
|
#[allow(dead_code)]
|
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
|
pub struct CarBlock {
|
|
pub cid: Cid,
|
|
pub data: Vec<u8>,
|
|
}
|
|
|
|
/// Parse a CAR v1 file. Returns the header and the list of blocks in order.
|
|
///
|
|
/// This is intentionally minimal — it does not validate CIDs, codec, or
|
|
/// DAG-CBOR, only structure. Used in unit/integration tests to round-trip
|
|
/// CAR files we just produced.
|
|
#[allow(dead_code)]
|
|
pub fn parse(bytes: &[u8]) -> Result<(CarHeader, Vec<CarBlock>)> {
|
|
let mut p = 0usize;
|
|
|
|
let (header_len, n) = read_varint(bytes, p)?;
|
|
p += n;
|
|
let header_end = p + header_len as usize;
|
|
if header_end > bytes.len() {
|
|
anyhow::bail!("CAR header length exceeds file");
|
|
}
|
|
let header_bytes = &bytes[p..header_end];
|
|
let header = decode_header(header_bytes)?;
|
|
p = header_end;
|
|
|
|
let mut blocks = Vec::new();
|
|
while p < bytes.len() {
|
|
let (section_len, n) = read_varint(bytes, p)?;
|
|
p += n;
|
|
let section_end = p + section_len as usize;
|
|
if section_end > bytes.len() {
|
|
anyhow::bail!("CAR section length exceeds file at offset {}", p - n);
|
|
}
|
|
let section = &bytes[p..section_end];
|
|
let (cid, data) = read_section(section)?;
|
|
blocks.push(CarBlock { cid, data });
|
|
p = section_end;
|
|
}
|
|
|
|
Ok((header, blocks))
|
|
}
|
|
|
|
fn read_varint(bytes: &[u8], offset: usize) -> Result<(u64, usize)> {
|
|
let mut value: u64 = 0;
|
|
let mut shift = 0u32;
|
|
let mut i = offset;
|
|
loop {
|
|
if i >= bytes.len() {
|
|
anyhow::bail!("varint extends past end of input");
|
|
}
|
|
let b = bytes[i];
|
|
i += 1;
|
|
value |= ((b & 0x7f) as u64) << shift;
|
|
if b & 0x80 == 0 {
|
|
return Ok((value, i - offset));
|
|
}
|
|
shift += 7;
|
|
if shift >= 64 {
|
|
anyhow::bail!("varint too long");
|
|
}
|
|
}
|
|
}
|
|
|
|
fn read_section(section: &[u8]) -> Result<(Cid, Vec<u8>)> {
|
|
let cid = Cid::read_bytes(section)
|
|
.map_err(|e| anyhow::anyhow!("invalid CID in CAR section: {e}"))?;
|
|
let cid_len = cid.encoded_len();
|
|
if cid_len > section.len() {
|
|
anyhow::bail!("section too short for CID");
|
|
}
|
|
let data = section[cid_len..].to_vec();
|
|
Ok((cid, data))
|
|
}
|
|
|
|
#[allow(dead_code)]
|
|
fn decode_header(bytes: &[u8]) -> Result<CarHeader> {
|
|
// The header is a tiny DAG-CBOR map. We decode only the structure we emit.
|
|
let mut p = 0usize;
|
|
let (n_items, consumed) = read_head_and_uint(bytes, p, 5)?;
|
|
p += consumed;
|
|
if n_items != 2 {
|
|
anyhow::bail!("CAR header must have 2 keys, got {n_items}");
|
|
}
|
|
|
|
let mut version: Option<u64> = None;
|
|
let mut roots: Vec<Cid> = Vec::new();
|
|
|
|
for _ in 0..2 {
|
|
let (key, consumed) = read_head_and_text(bytes, p)?;
|
|
p += consumed;
|
|
match key.as_str() {
|
|
"version" => {
|
|
let (v, c) = read_head_and_uint(bytes, p, 0)?;
|
|
p += c;
|
|
version = Some(v);
|
|
}
|
|
"roots" => {
|
|
let (n_roots, c) = read_head_and_uint(bytes, p, 4)?;
|
|
p += c;
|
|
for _ in 0..n_roots {
|
|
// tag(42)
|
|
let (_, c) = read_head_and_uint(bytes, p, 6)?;
|
|
p += c;
|
|
// bytes
|
|
let (n, c) = read_head_and_uint(bytes, p, 2)?;
|
|
p += c;
|
|
if p + n as usize > bytes.len() {
|
|
anyhow::bail!("CAR root CID bytes exceed header");
|
|
}
|
|
let cid_bytes = &bytes[p..p + n as usize];
|
|
let cid = Cid::read_bytes(cid_bytes)
|
|
.map_err(|e| anyhow::anyhow!("invalid root CID bytes: {e}"))?;
|
|
p += n as usize;
|
|
roots.push(cid);
|
|
}
|
|
}
|
|
other => anyhow::bail!("unknown CAR header key `{other}`"),
|
|
}
|
|
}
|
|
|
|
Ok(CarHeader {
|
|
version: version.unwrap_or(0),
|
|
roots,
|
|
})
|
|
}
|
|
|
|
/// Read a CBOR head (single byte for value < 24, otherwise head + varint
|
|
/// extension) and decode its value. Validates that the major type is
|
|
/// `expected_major`. Returns the decoded value and the number of bytes
|
|
/// consumed (head + any extension).
|
|
#[allow(dead_code)]
|
|
fn read_head_and_uint(
|
|
bytes: &[u8],
|
|
offset: usize,
|
|
expected_major: u8,
|
|
) -> Result<(u64, usize)> {
|
|
if offset >= bytes.len() {
|
|
anyhow::bail!("CBOR read past end of input");
|
|
}
|
|
let first = bytes[offset];
|
|
let major = first >> 5;
|
|
if major != expected_major {
|
|
anyhow::bail!(
|
|
"expected CBOR major {}, got {}",
|
|
expected_major,
|
|
major
|
|
);
|
|
}
|
|
let low = first & 0x1f;
|
|
let (value, extra) = match low {
|
|
0..=23 => (low as u64, 0usize),
|
|
24 => {
|
|
if offset + 2 > bytes.len() {
|
|
anyhow::bail!("truncated CBOR uint8");
|
|
}
|
|
(bytes[offset + 1] as u64, 1)
|
|
}
|
|
25 => {
|
|
if offset + 3 > bytes.len() {
|
|
anyhow::bail!("truncated CBOR uint16");
|
|
}
|
|
(
|
|
((bytes[offset + 1] as u64) << 8) | (bytes[offset + 2] as u64),
|
|
2,
|
|
)
|
|
}
|
|
26 => {
|
|
if offset + 5 > bytes.len() {
|
|
anyhow::bail!("truncated CBOR uint32");
|
|
}
|
|
let n = ((bytes[offset + 1] as u64) << 24)
|
|
| ((bytes[offset + 2] as u64) << 16)
|
|
| ((bytes[offset + 3] as u64) << 8)
|
|
| (bytes[offset + 4] as u64);
|
|
(n, 4)
|
|
}
|
|
27 => {
|
|
if offset + 9 > bytes.len() {
|
|
anyhow::bail!("truncated CBOR uint64");
|
|
}
|
|
let mut n = 0u64;
|
|
for i in 0..8 {
|
|
n = (n << 8) | (bytes[offset + 1 + i] as u64);
|
|
}
|
|
(n, 8)
|
|
}
|
|
other => anyhow::bail!("unsupported CBOR uint tag {other}"),
|
|
};
|
|
Ok((value, 1 + extra))
|
|
}
|
|
|
|
/// Read a CBOR text string with major type 3, returning the string and the
|
|
/// total number of bytes consumed.
|
|
#[allow(dead_code)]
|
|
fn read_head_and_text(
|
|
bytes: &[u8],
|
|
offset: usize,
|
|
) -> Result<(String, usize)> {
|
|
let (n, c) = read_head_and_uint(bytes, offset, 3)?;
|
|
if offset + c + n as usize > bytes.len() {
|
|
anyhow::bail!("CBOR text string exceeds buffer");
|
|
}
|
|
let s = std::str::from_utf8(&bytes[offset + c..offset + c + n as usize])
|
|
.map_err(|e| anyhow::anyhow!("invalid UTF-8 in CBOR text: {e}"))?;
|
|
Ok((s.to_string(), c + n as usize))
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
use at_crypto::cid::cid_for_cbor;
|
|
|
|
#[test]
|
|
fn header_encodes_cids_with_tag_42() {
|
|
let c1 = cid_for_cbor(b"a").unwrap();
|
|
let c2 = cid_for_cbor(b"b").unwrap();
|
|
let bytes = encode_header(&[c1, c2]);
|
|
// First byte: map(2) = 0xA2
|
|
assert_eq!(bytes[0], 0xA2, "first byte must be map(2)");
|
|
// Round-trip via our parser.
|
|
let h = decode_header(&bytes).unwrap();
|
|
assert_eq!(h.version, 1);
|
|
assert_eq!(h.roots, vec![c1, c2]);
|
|
}
|
|
|
|
#[test]
|
|
fn car_round_trip_with_one_block() {
|
|
let cid = cid_for_cbor(b"hello world").unwrap();
|
|
let mut w = CarWriter::new();
|
|
w.append(cid, b"hello world");
|
|
let car = w.finish(&[cid]);
|
|
let (h, blocks) = parse(&car).unwrap();
|
|
assert_eq!(h.version, 1);
|
|
assert_eq!(h.roots, vec![cid]);
|
|
assert_eq!(blocks.len(), 1);
|
|
assert_eq!(blocks[0].cid, cid);
|
|
assert_eq!(blocks[0].data, b"hello world");
|
|
}
|
|
|
|
#[test]
|
|
fn car_round_trip_with_many_blocks_and_no_dupes() {
|
|
let cids: Vec<Cid> = (0..5)
|
|
.map(|i| cid_for_cbor(format!("block-{i}").as_bytes()).unwrap())
|
|
.collect();
|
|
let mut w = CarWriter::new();
|
|
for (i, c) in cids.iter().enumerate() {
|
|
w.append(*c, format!("block-{i}").as_bytes());
|
|
}
|
|
// Re-appending the same CID should be a no-op.
|
|
w.append(cids[0], b"ignored");
|
|
assert_eq!(w.len(), 5);
|
|
let car = w.finish(&[cids[2]]);
|
|
let (h, blocks) = parse(&car).unwrap();
|
|
assert_eq!(h.roots, vec![cids[2]]);
|
|
assert_eq!(blocks.len(), 5);
|
|
for (i, b) in blocks.iter().enumerate() {
|
|
assert_eq!(b.cid, cids[i]);
|
|
assert_eq!(b.data, format!("block-{i}").as_bytes());
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn car_with_empty_roots() {
|
|
let cid = cid_for_cbor(b"only block").unwrap();
|
|
let mut w = CarWriter::new();
|
|
w.append(cid, b"only block");
|
|
let car = w.finish(&[]);
|
|
let (h, blocks) = parse(&car).unwrap();
|
|
assert_eq!(h.version, 1);
|
|
assert!(h.roots.is_empty());
|
|
assert_eq!(blocks.len(), 1);
|
|
}
|
|
|
|
#[test]
|
|
fn block_cid_verifies_under_sha256() {
|
|
// For DAG-CBOR blocks the CID is the SHA-256 of the bytes. Verify the
|
|
// CID we put in the CAR header matches a re-computed CID over the
|
|
// block data.
|
|
let data = b"some record bytes".to_vec();
|
|
let cid = cid_for_cbor(&data).unwrap();
|
|
let mut w = CarWriter::new();
|
|
w.append(cid, &data);
|
|
let car = w.finish(&[cid]);
|
|
let (_h, blocks) = parse(&car).unwrap();
|
|
for b in &blocks {
|
|
let recomputed = cid_for_cbor(&b.data).unwrap();
|
|
assert_eq!(b.cid, recomputed);
|
|
}
|
|
}
|
|
}
|