maarcadetweet: initial commit
AT Protocol PDS + AppView + Tauri Desktop Client, 160-char post limit. - PDS (Rust + axum + sqlx) - Auth: createAccount, createSession, refreshSession - Records: createRecord, deleteRecord (race-safe via SELECT FOR UPDATE) - Feed: feed.like.create, feed.repost.create - Sync: getRepo, getBlocks, getLatestCommit, getRecord (with MST proof), listRepos - Identity: resolveHandle - MST: spec-conformant (at-mst crate, 27 tests) - Repo: signed commits, TID counter (monotonic, 4096 wrap safe) - AppView (Rust + axum + sqlx) - Jetstream consumer (WebSocket, exponential backoff, 38k+ events indexed) - REST API: timeline/home (graph-aware), profile, search, post (with thread hydration) - Handle-sync worker (did:plc + did:web) - JSONB embed storage + thread columns (migration 0003) - Like/repost counter cache (migration 0004) - Tauri 2 + Svelte 5 Desktop Client - System tray (Show/Compose/Quit menu) - OS notifications (tauri-plugin-notification) - Auto-update (tauri-plugin-updater, placeholder endpoint) - Window-state (tauri-plugin-window-state) - 160-char compose with live counter - Image/Link embed rendering - LocalStorage-persisted like state - Timeline with poll (prepend new posts) - Custom TitleBar (transparent, no decorations) - Orange/IBM Plex Mono maarcade design Tests: 231 Rust + 9 vitest = 240 passed.
This commit is contained in:
@@ -0,0 +1,479 @@
|
||||
//! CAR v1 writer for atproto sync endpoints.
|
||||
//!
|
||||
//! The on-the-wire format follows
|
||||
//! <https://ipld.io/specs/transport/car/carv1/> and is the same format used by
|
||||
//! `com.atproto.sync.getRepo`, `getBlocks`, `getLatestCommit` and
|
||||
//! `getRecord`.
|
||||
//!
|
||||
//! Layout:
|
||||
//!
|
||||
//! ```text
|
||||
//! [ varint: header_len | DAG-CBOR header block ] (header)
|
||||
//! [ varint: section_len | CID | block bytes ] (block 1)
|
||||
//! [ varint: section_len | CID | block bytes ] (block 2)
|
||||
//! ...
|
||||
//! ```
|
||||
//!
|
||||
//! The header is `{ version: 1, roots: [CID, ...] }` encoded as DAG-CBOR. In
|
||||
//! DAG-CBOR CID links carry the IANA-registered CBOR tag `42`, which the
|
||||
//! `ciborium` crate does not emit for `cid::Cid` (it uses serde newtype-struct
|
||||
//! tagging instead). We hand-encode the header bytes to keep the file
|
||||
//! spec-compliant: a `Map(2)` with text keys `"version"` and `"roots"`, an
|
||||
//! unsigned int `1` for the version, and a tagged byte string for each root
|
||||
//! CID.
|
||||
//!
|
||||
//! Per the spec, CAR v1 stores the raw CID bytes (varint version + codec +
|
||||
//! multihash) prefixed to every block, with a leading varint giving the total
|
||||
//! length of the section (CID + block).
|
||||
|
||||
use anyhow::Result;
|
||||
use cid::Cid;
|
||||
|
||||
/// Encode an unsigned CBOR head (major type in upper 3 bits) with a value.
|
||||
///
|
||||
/// Supports values up to `u32::MAX` which is more than enough for any realistic
|
||||
/// header or array length.
|
||||
fn cbor_head(out: &mut Vec<u8>, major: u8, n: u64) {
|
||||
let m = (major & 0x07) << 5;
|
||||
if n < 24 {
|
||||
out.push(m | n as u8);
|
||||
} else if n < 0x100 {
|
||||
out.push(m | 24);
|
||||
out.push(n as u8);
|
||||
} else if n < 0x10000 {
|
||||
out.push(m | 25);
|
||||
out.push((n >> 8) as u8);
|
||||
out.push(n as u8);
|
||||
} else if n < 0x100_0000 {
|
||||
out.push(m | 26);
|
||||
out.push((n >> 16) as u8);
|
||||
out.push((n >> 8) as u8);
|
||||
out.push(n as u8);
|
||||
} else {
|
||||
out.push(m | 27);
|
||||
out.push((n >> 24) as u8);
|
||||
out.push((n >> 16) as u8);
|
||||
out.push((n >> 8) as u8);
|
||||
out.push(n as u8);
|
||||
}
|
||||
}
|
||||
|
||||
/// Append a CBOR text string.
|
||||
fn cbor_text(out: &mut Vec<u8>, s: &str) {
|
||||
cbor_head(out, 3, s.len() as u64);
|
||||
out.extend_from_slice(s.as_bytes());
|
||||
}
|
||||
|
||||
/// Append a CBOR byte string.
|
||||
fn cbor_bytes(out: &mut Vec<u8>, b: &[u8]) {
|
||||
cbor_head(out, 2, b.len() as u64);
|
||||
out.extend_from_slice(b);
|
||||
}
|
||||
|
||||
/// Append a CBOR tag wrapping the following value.
|
||||
fn cbor_tag(out: &mut Vec<u8>, tag: u64) {
|
||||
cbor_head(out, 6, tag);
|
||||
}
|
||||
|
||||
/// Encode the CAR v1 DAG-CBOR header `{ version: 1, roots: [CID, ...] }`.
|
||||
///
|
||||
/// CIDs are encoded as `tag(42) + bytes(<raw-cid-bytes>)` per the DAG-CBOR
|
||||
/// spec. This is the canonical IPLD CID-link form.
|
||||
pub fn encode_header(roots: &[Cid]) -> Vec<u8> {
|
||||
let mut out = Vec::new();
|
||||
// Map(2): { "version": 1, "roots": [...] }
|
||||
cbor_head(&mut out, 5, 2);
|
||||
cbor_text(&mut out, "version");
|
||||
cbor_head(&mut out, 0, 1);
|
||||
cbor_text(&mut out, "roots");
|
||||
cbor_head(&mut out, 4, roots.len() as u64);
|
||||
for cid in roots {
|
||||
cbor_tag(&mut out, 42);
|
||||
cbor_bytes(&mut out, &cid.to_bytes());
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Append a varint to `out` using LEB128 unsigned encoding.
|
||||
fn write_varint(out: &mut Vec<u8>, n: u64) {
|
||||
let mut buf = unsigned_varint::encode::u64_buffer();
|
||||
let bytes = unsigned_varint::encode::u64(n, &mut buf);
|
||||
out.extend_from_slice(bytes);
|
||||
}
|
||||
|
||||
/// A single (CID, block_bytes) pair held in a [`CarWriter`].
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Block {
|
||||
pub cid: Cid,
|
||||
pub data: Vec<u8>,
|
||||
}
|
||||
|
||||
/// Buffer for assembling a CAR v1 file.
|
||||
///
|
||||
/// Usage:
|
||||
///
|
||||
/// ```ignore
|
||||
/// let mut w = CarWriter::new();
|
||||
/// w.append(cid_a, &block_a);
|
||||
/// w.append(cid_b, &block_b);
|
||||
/// let bytes = w.finish(&[head_commit_cid]);
|
||||
/// ```
|
||||
///
|
||||
/// The header's `roots` is provided at `finish` time so callers can defer
|
||||
/// deciding what the root is until all blocks are queued.
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct CarWriter {
|
||||
blocks: Vec<Block>,
|
||||
}
|
||||
|
||||
impl CarWriter {
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Append a (CID, block) pair. Duplicate CIDs are de-duplicated: the first
|
||||
/// occurrence wins. CAR v1 allows duplicate blocks in principle but for
|
||||
/// repo exports the spec says the root CID is unique and our callers don't
|
||||
/// need to write the same block twice.
|
||||
pub fn append(&mut self, cid: Cid, data: &[u8]) {
|
||||
if self.blocks.iter().any(|b| b.cid == cid) {
|
||||
return;
|
||||
}
|
||||
self.blocks.push(Block {
|
||||
cid,
|
||||
data: data.to_vec(),
|
||||
});
|
||||
}
|
||||
|
||||
/// Finalize the CAR stream. Writes the header followed by every queued
|
||||
/// block as a length-prefixed CID+data section.
|
||||
pub fn finish(&self, roots: &[Cid]) -> Vec<u8> {
|
||||
let header = encode_header(roots);
|
||||
let mut out = Vec::with_capacity(header.len() + self.blocks.len() * 64);
|
||||
write_varint(&mut out, header.len() as u64);
|
||||
out.extend_from_slice(&header);
|
||||
for b in &self.blocks {
|
||||
let cid_bytes = b.cid.to_bytes();
|
||||
// Section length is the combined length of CID bytes + block data.
|
||||
let section_len = (cid_bytes.len() + b.data.len()) as u64;
|
||||
write_varint(&mut out, section_len);
|
||||
out.extend_from_slice(&cid_bytes);
|
||||
out.extend_from_slice(&b.data);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub fn len(&self) -> usize {
|
||||
self.blocks.len()
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.blocks.is_empty()
|
||||
}
|
||||
}
|
||||
|
||||
// -- minimal CAR reader (for tests / debug) --------------------------------
|
||||
|
||||
/// Header parsed out of a CAR file. `roots` are kept as raw CID byte vectors
|
||||
/// so callers can re-parse them however they like.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct CarHeader {
|
||||
pub version: u64,
|
||||
pub roots: Vec<Cid>,
|
||||
}
|
||||
|
||||
/// A block parsed from a CAR file.
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct CarBlock {
|
||||
pub cid: Cid,
|
||||
pub data: Vec<u8>,
|
||||
}
|
||||
|
||||
/// Parse a CAR v1 file. Returns the header and the list of blocks in order.
|
||||
///
|
||||
/// This is intentionally minimal — it does not validate CIDs, codec, or
|
||||
/// DAG-CBOR, only structure. Used in unit/integration tests to round-trip
|
||||
/// CAR files we just produced.
|
||||
#[allow(dead_code)]
|
||||
pub fn parse(bytes: &[u8]) -> Result<(CarHeader, Vec<CarBlock>)> {
|
||||
let mut p = 0usize;
|
||||
|
||||
let (header_len, n) = read_varint(bytes, p)?;
|
||||
p += n;
|
||||
let header_end = p + header_len as usize;
|
||||
if header_end > bytes.len() {
|
||||
anyhow::bail!("CAR header length exceeds file");
|
||||
}
|
||||
let header_bytes = &bytes[p..header_end];
|
||||
let header = decode_header(header_bytes)?;
|
||||
p = header_end;
|
||||
|
||||
let mut blocks = Vec::new();
|
||||
while p < bytes.len() {
|
||||
let (section_len, n) = read_varint(bytes, p)?;
|
||||
p += n;
|
||||
let section_end = p + section_len as usize;
|
||||
if section_end > bytes.len() {
|
||||
anyhow::bail!("CAR section length exceeds file at offset {}", p - n);
|
||||
}
|
||||
let section = &bytes[p..section_end];
|
||||
let (cid, data) = read_section(section)?;
|
||||
blocks.push(CarBlock { cid, data });
|
||||
p = section_end;
|
||||
}
|
||||
|
||||
Ok((header, blocks))
|
||||
}
|
||||
|
||||
fn read_varint(bytes: &[u8], offset: usize) -> Result<(u64, usize)> {
|
||||
let mut value: u64 = 0;
|
||||
let mut shift = 0u32;
|
||||
let mut i = offset;
|
||||
loop {
|
||||
if i >= bytes.len() {
|
||||
anyhow::bail!("varint extends past end of input");
|
||||
}
|
||||
let b = bytes[i];
|
||||
i += 1;
|
||||
value |= ((b & 0x7f) as u64) << shift;
|
||||
if b & 0x80 == 0 {
|
||||
return Ok((value, i - offset));
|
||||
}
|
||||
shift += 7;
|
||||
if shift >= 64 {
|
||||
anyhow::bail!("varint too long");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn read_section(section: &[u8]) -> Result<(Cid, Vec<u8>)> {
|
||||
let cid = Cid::read_bytes(section)
|
||||
.map_err(|e| anyhow::anyhow!("invalid CID in CAR section: {e}"))?;
|
||||
let cid_len = cid.encoded_len();
|
||||
if cid_len > section.len() {
|
||||
anyhow::bail!("section too short for CID");
|
||||
}
|
||||
let data = section[cid_len..].to_vec();
|
||||
Ok((cid, data))
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
fn decode_header(bytes: &[u8]) -> Result<CarHeader> {
|
||||
// The header is a tiny DAG-CBOR map. We decode only the structure we emit.
|
||||
let mut p = 0usize;
|
||||
let (n_items, consumed) = read_head_and_uint(bytes, p, 5)?;
|
||||
p += consumed;
|
||||
if n_items != 2 {
|
||||
anyhow::bail!("CAR header must have 2 keys, got {n_items}");
|
||||
}
|
||||
|
||||
let mut version: Option<u64> = None;
|
||||
let mut roots: Vec<Cid> = Vec::new();
|
||||
|
||||
for _ in 0..2 {
|
||||
let (key, consumed) = read_head_and_text(bytes, p)?;
|
||||
p += consumed;
|
||||
match key.as_str() {
|
||||
"version" => {
|
||||
let (v, c) = read_head_and_uint(bytes, p, 0)?;
|
||||
p += c;
|
||||
version = Some(v);
|
||||
}
|
||||
"roots" => {
|
||||
let (n_roots, c) = read_head_and_uint(bytes, p, 4)?;
|
||||
p += c;
|
||||
for _ in 0..n_roots {
|
||||
// tag(42)
|
||||
let (_, c) = read_head_and_uint(bytes, p, 6)?;
|
||||
p += c;
|
||||
// bytes
|
||||
let (n, c) = read_head_and_uint(bytes, p, 2)?;
|
||||
p += c;
|
||||
if p + n as usize > bytes.len() {
|
||||
anyhow::bail!("CAR root CID bytes exceed header");
|
||||
}
|
||||
let cid_bytes = &bytes[p..p + n as usize];
|
||||
let cid = Cid::read_bytes(cid_bytes)
|
||||
.map_err(|e| anyhow::anyhow!("invalid root CID bytes: {e}"))?;
|
||||
p += n as usize;
|
||||
roots.push(cid);
|
||||
}
|
||||
}
|
||||
other => anyhow::bail!("unknown CAR header key `{other}`"),
|
||||
}
|
||||
}
|
||||
|
||||
Ok(CarHeader {
|
||||
version: version.unwrap_or(0),
|
||||
roots,
|
||||
})
|
||||
}
|
||||
|
||||
/// Read a CBOR head (single byte for value < 24, otherwise head + varint
|
||||
/// extension) and decode its value. Validates that the major type is
|
||||
/// `expected_major`. Returns the decoded value and the number of bytes
|
||||
/// consumed (head + any extension).
|
||||
#[allow(dead_code)]
|
||||
fn read_head_and_uint(
|
||||
bytes: &[u8],
|
||||
offset: usize,
|
||||
expected_major: u8,
|
||||
) -> Result<(u64, usize)> {
|
||||
if offset >= bytes.len() {
|
||||
anyhow::bail!("CBOR read past end of input");
|
||||
}
|
||||
let first = bytes[offset];
|
||||
let major = first >> 5;
|
||||
if major != expected_major {
|
||||
anyhow::bail!(
|
||||
"expected CBOR major {}, got {}",
|
||||
expected_major,
|
||||
major
|
||||
);
|
||||
}
|
||||
let low = first & 0x1f;
|
||||
let (value, extra) = match low {
|
||||
0..=23 => (low as u64, 0usize),
|
||||
24 => {
|
||||
if offset + 2 > bytes.len() {
|
||||
anyhow::bail!("truncated CBOR uint8");
|
||||
}
|
||||
(bytes[offset + 1] as u64, 1)
|
||||
}
|
||||
25 => {
|
||||
if offset + 3 > bytes.len() {
|
||||
anyhow::bail!("truncated CBOR uint16");
|
||||
}
|
||||
(
|
||||
((bytes[offset + 1] as u64) << 8) | (bytes[offset + 2] as u64),
|
||||
2,
|
||||
)
|
||||
}
|
||||
26 => {
|
||||
if offset + 5 > bytes.len() {
|
||||
anyhow::bail!("truncated CBOR uint32");
|
||||
}
|
||||
let n = ((bytes[offset + 1] as u64) << 24)
|
||||
| ((bytes[offset + 2] as u64) << 16)
|
||||
| ((bytes[offset + 3] as u64) << 8)
|
||||
| (bytes[offset + 4] as u64);
|
||||
(n, 4)
|
||||
}
|
||||
27 => {
|
||||
if offset + 9 > bytes.len() {
|
||||
anyhow::bail!("truncated CBOR uint64");
|
||||
}
|
||||
let mut n = 0u64;
|
||||
for i in 0..8 {
|
||||
n = (n << 8) | (bytes[offset + 1 + i] as u64);
|
||||
}
|
||||
(n, 8)
|
||||
}
|
||||
other => anyhow::bail!("unsupported CBOR uint tag {other}"),
|
||||
};
|
||||
Ok((value, 1 + extra))
|
||||
}
|
||||
|
||||
/// Read a CBOR text string with major type 3, returning the string and the
|
||||
/// total number of bytes consumed.
|
||||
#[allow(dead_code)]
|
||||
fn read_head_and_text(
|
||||
bytes: &[u8],
|
||||
offset: usize,
|
||||
) -> Result<(String, usize)> {
|
||||
let (n, c) = read_head_and_uint(bytes, offset, 3)?;
|
||||
if offset + c + n as usize > bytes.len() {
|
||||
anyhow::bail!("CBOR text string exceeds buffer");
|
||||
}
|
||||
let s = std::str::from_utf8(&bytes[offset + c..offset + c + n as usize])
|
||||
.map_err(|e| anyhow::anyhow!("invalid UTF-8 in CBOR text: {e}"))?;
|
||||
Ok((s.to_string(), c + n as usize))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use at_crypto::cid::cid_for_cbor;
|
||||
|
||||
#[test]
|
||||
fn header_encodes_cids_with_tag_42() {
|
||||
let c1 = cid_for_cbor(b"a").unwrap();
|
||||
let c2 = cid_for_cbor(b"b").unwrap();
|
||||
let bytes = encode_header(&[c1, c2]);
|
||||
// First byte: map(2) = 0xA2
|
||||
assert_eq!(bytes[0], 0xA2, "first byte must be map(2)");
|
||||
// Round-trip via our parser.
|
||||
let h = decode_header(&bytes).unwrap();
|
||||
assert_eq!(h.version, 1);
|
||||
assert_eq!(h.roots, vec![c1, c2]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn car_round_trip_with_one_block() {
|
||||
let cid = cid_for_cbor(b"hello world").unwrap();
|
||||
let mut w = CarWriter::new();
|
||||
w.append(cid, b"hello world");
|
||||
let car = w.finish(&[cid]);
|
||||
let (h, blocks) = parse(&car).unwrap();
|
||||
assert_eq!(h.version, 1);
|
||||
assert_eq!(h.roots, vec![cid]);
|
||||
assert_eq!(blocks.len(), 1);
|
||||
assert_eq!(blocks[0].cid, cid);
|
||||
assert_eq!(blocks[0].data, b"hello world");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn car_round_trip_with_many_blocks_and_no_dupes() {
|
||||
let cids: Vec<Cid> = (0..5)
|
||||
.map(|i| cid_for_cbor(format!("block-{i}").as_bytes()).unwrap())
|
||||
.collect();
|
||||
let mut w = CarWriter::new();
|
||||
for (i, c) in cids.iter().enumerate() {
|
||||
w.append(*c, format!("block-{i}").as_bytes());
|
||||
}
|
||||
// Re-appending the same CID should be a no-op.
|
||||
w.append(cids[0], b"ignored");
|
||||
assert_eq!(w.len(), 5);
|
||||
let car = w.finish(&[cids[2]]);
|
||||
let (h, blocks) = parse(&car).unwrap();
|
||||
assert_eq!(h.roots, vec![cids[2]]);
|
||||
assert_eq!(blocks.len(), 5);
|
||||
for (i, b) in blocks.iter().enumerate() {
|
||||
assert_eq!(b.cid, cids[i]);
|
||||
assert_eq!(b.data, format!("block-{i}").as_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn car_with_empty_roots() {
|
||||
let cid = cid_for_cbor(b"only block").unwrap();
|
||||
let mut w = CarWriter::new();
|
||||
w.append(cid, b"only block");
|
||||
let car = w.finish(&[]);
|
||||
let (h, blocks) = parse(&car).unwrap();
|
||||
assert_eq!(h.version, 1);
|
||||
assert!(h.roots.is_empty());
|
||||
assert_eq!(blocks.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn block_cid_verifies_under_sha256() {
|
||||
// For DAG-CBOR blocks the CID is the SHA-256 of the bytes. Verify the
|
||||
// CID we put in the CAR header matches a re-computed CID over the
|
||||
// block data.
|
||||
let data = b"some record bytes".to_vec();
|
||||
let cid = cid_for_cbor(&data).unwrap();
|
||||
let mut w = CarWriter::new();
|
||||
w.append(cid, &data);
|
||||
let car = w.finish(&[cid]);
|
||||
let (_h, blocks) = parse(&car).unwrap();
|
||||
for b in &blocks {
|
||||
let recomputed = cid_for_cbor(&b.data).unwrap();
|
||||
assert_eq!(b.cid, recomputed);
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user