diff --git a/crates/ctx-cli/src/commands/pack/git.rs b/crates/ctx-cli/src/commands/pack/git.rs index a20bf7d..23cf56f 100644 --- a/crates/ctx-cli/src/commands/pack/git.rs +++ b/crates/ctx-cli/src/commands/pack/git.rs @@ -5,7 +5,9 @@ use super::*; pub(crate) fn git_changed_paths(root: &Path) -> Result, String> { let output = Command::new("git") - .args(["-C", &root.to_string_lossy(), "status", "--porcelain"]) + .arg("-C") + .arg(root) + .args(["status", "--porcelain=v1", "-z", "--untracked-files=all"]) .output(); let output = match output { Ok(output) if output.status.success() => output, @@ -21,18 +23,37 @@ pub(crate) fn git_changed_paths(root: &Path) -> Result std::collections::BTreeSet { let mut changed = std::collections::BTreeSet::new(); - for line in String::from_utf8_lossy(&output.stdout).lines() { - if line.len() < 4 { + let mut fields = output.split(|byte| *byte == 0); + + while let Some(record) = fields.next() { + if record.is_empty() { + continue; + } + if record.len() < 4 || record[2] != b' ' { continue; } - let path = line[3..].trim(); - let path = path.split(" -> ").last().unwrap_or(path); + + let status = &record[..2]; + let path = &record[3..]; if !path.is_empty() { - changed.insert(path.replace('\\', "/")); + // The pack walker also uses to_string_lossy for filesystem paths, + // so applying the same conversion keeps non-UTF-8 names comparable. + changed.insert(String::from_utf8_lossy(path).into_owned()); + } + + // In porcelain v1 -z, rename/copy records put the destination path in + // the main record and the source path in the following NUL field. + if status.iter().any(|byte| matches!(*byte, b'R' | b'C')) { + let _ = fields.next(); } } - Ok(changed) + + changed } pub(crate) fn git_diff_entries( @@ -43,41 +64,32 @@ pub(crate) fn git_diff_entries( let (base, head) = parse_diff_revspec(revspec)?; let before_commit = git_output_in(root, &["rev-parse", "--short=7", base])?; let after_commit = git_output_in(root, &["rev-parse", "--short=7", head])?; - let name_status = git_output_in(root, &["diff", "--name-status", base, head])?; + let name_status = + git_output_bytes_in(root, &["diff", "--name-status", "-z", base, head, "--"])?; let mut entries = Vec::new(); - for line in name_status.lines() { - // --name-status output is tab-separated; paths may contain spaces. - let fields: Vec<&str> = line.split('\t').collect(); - if fields.len() < 2 { - continue; - } - let status = fields[0]; - let (path, before_path) = if status.starts_with('R') && fields.len() >= 3 { - (fields[2], fields[1]) - } else { - (fields[1], fields[1]) - }; + for (status, path, before_path) in parse_git_name_status_z(&name_status) { let added = status.starts_with('A'); let deleted = status.starts_with('D'); - let binary = git_diff_is_binary(root, base, head, path)?; - let patch = git_output_allow_empty(root, &["diff", base, head, "--", path])?; + let binary = git_diff_is_binary(root, base, head, &path)?; + let patch = git_output_allow_empty(root, &["diff", base, head, "--", &path])?; let mut before_content = if added || binary { String::new() } else { - git_show_file(root, base, before_path).unwrap_or_default() + git_show_file(root, base, &before_path).unwrap_or_default() }; let mut after_content = if deleted || binary { String::new() } else { - git_show_file(root, head, path).unwrap_or_default() + git_show_file(root, head, &path).unwrap_or_default() }; if api_only { before_content = - extract_public_api_light(path, &before_content).unwrap_or(before_content); - after_content = extract_public_api_light(path, &after_content).unwrap_or(after_content); + extract_public_api_light(&path, &before_content).unwrap_or(before_content); + after_content = + extract_public_api_light(&path, &after_content).unwrap_or(after_content); } entries.push(ctx_pack::DiffEntry { - path: path.to_string(), + path, before_content, after_content, before_commit: before_commit.clone(), @@ -91,6 +103,50 @@ pub(crate) fn git_diff_entries( Ok(entries) } +fn parse_git_name_status_z(output: &[u8]) -> Vec<(String, String, String)> { + let mut fields = output + .split(|byte| *byte == 0) + .filter(|field| !field.is_empty()); + let mut entries = Vec::new(); + + while let Some(status_bytes) = fields.next() { + let status = String::from_utf8_lossy(status_bytes).into_owned(); + let Some(before_bytes) = fields.next() else { + break; + }; + let before = String::from_utf8_lossy(before_bytes).into_owned(); + + if matches!(status.as_bytes().first(), Some(b'R' | b'C')) { + let Some(after_bytes) = fields.next() else { + break; + }; + let after = String::from_utf8_lossy(after_bytes).into_owned(); + entries.push((status, after, before)); + } else { + entries.push((status, before.clone(), before)); + } + } + + entries +} + +fn git_output_bytes_in(root: &Path, args: &[&str]) -> Result, String> { + let output = Command::new("git") + .arg("-C") + .arg(root) + .args(args) + .output() + .map_err(|err| format!("git {}: {err}", args.join(" ")))?; + if !output.status.success() { + return Err(format!( + "git {}: {}", + args.join(" "), + String::from_utf8_lossy(&output.stderr).trim() + )); + } + Ok(output.stdout) +} + pub(crate) fn parse_diff_revspec(revspec: &str) -> Result<(&str, &str), String> { let Some((base, head)) = revspec.split_once("..") else { return Err("diff revspec must be BASE..HEAD".to_string()); @@ -145,3 +201,47 @@ pub(crate) fn git_output_allow_empty(root: &Path, args: &[&str]) -> Result Err(err), } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + #[test] + fn name_status_z_parser_preserves_special_paths_and_copy_rename_pairs() { + let raw = b"M\0line\nbreak.rs\0R100\0old name.rs\0new -> name.rs\0C075\0source.rs\0copy\\name.rs\0"; + let entries = parse_git_name_status_z(raw); + + assert_eq!( + entries, + vec![ + ( + "M".to_string(), + "line\nbreak.rs".to_string(), + "line\nbreak.rs".to_string(), + ), + ( + "R100".to_string(), + "new -> name.rs".to_string(), + "old name.rs".to_string(), + ), + ( + "C075".to_string(), + r"copy\name.rs".to_string(), + "source.rs".to_string(), + ), + ] + ); + } + + fn porcelain_z_parser_preserves_special_paths_and_rename_target() { + let raw = b"?? dir/a b.rs\0 M literal\\name.rs\0R dst -> literal.rs\0src old.rs\0?? comma,name.rs\0"; + let paths = parse_git_changed_paths(raw); + + assert!(paths.contains("dir/a b.rs")); + assert!(paths.contains(r"literal\name.rs")); + assert!(paths.contains("dst -> literal.rs")); + assert!(paths.contains("comma,name.rs")); + assert!(!paths.contains("src old.rs")); + } +} diff --git a/crates/ctx-cli/src/commands/pack/timefilter.rs b/crates/ctx-cli/src/commands/pack/timefilter.rs index 35e86a6..f338f1f 100644 --- a/crates/ctx-cli/src/commands/pack/timefilter.rs +++ b/crates/ctx-cli/src/commands/pack/timefilter.rs @@ -126,18 +126,46 @@ pub(crate) fn subtract_filter_duration( } pub(crate) fn parse_yyyy_mm_dd_utc(input: &str) -> Option { - let mut parts = input.split('-'); - let year = parts.next()?.parse::().ok()?; - let month = parts.next()?.parse::().ok()?; - let day = parts.next()?.parse::().ok()?; - if parts.next().is_some() || !(1..=12).contains(&month) || !(1..=31).contains(&day) { + let bytes = input.as_bytes(); + if bytes.len() != 10 + || bytes[4] != b'-' + || bytes[7] != b'-' + || bytes + .iter() + .enumerate() + .any(|(index, byte)| index != 4 && index != 7 && !byte.is_ascii_digit()) + { return None; } + + let year = input[0..4].parse::().ok()?; + let month = input[5..7].parse::().ok()?; + let day = input[8..10].parse::().ok()?; + let max_day = days_in_month(year, month)?; + if day == 0 || day > max_day { + return None; + } + let days = days_from_civil(year, month, day); if days < 0 { return None; } - Some(UNIX_EPOCH + Duration::from_secs(days as u64 * 24 * 60 * 60)) + let seconds = (days as u64).checked_mul(24 * 60 * 60)?; + UNIX_EPOCH.checked_add(Duration::from_secs(seconds)) +} + +fn is_leap_year(year: i64) -> bool { + (year % 4 == 0 && year % 100 != 0) || year % 400 == 0 +} + +fn days_in_month(year: i64, month: u32) -> Option { + match month { + 1 | 3 | 5 | 7 | 8 | 10 | 12 => Some(31), + 4 | 6 | 9 | 11 => Some(30), + 2 if is_leap_year(year) => Some(29), + 2 => Some(28), + _ => None, + } } pub(crate) fn days_from_civil(year: i64, month: u32, day: u32) -> i64 { @@ -150,3 +178,45 @@ pub(crate) fn days_from_civil(year: i64, month: u32, day: u32) -> i64 { let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy; era * 146097 + doe - 719468 } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn absolute_date_parser_validates_calendar_dates() { + assert!(parse_yyyy_mm_dd_utc("2024-02-29").is_some()); + assert!(parse_yyyy_mm_dd_utc("2026-01-31").is_some()); + + for invalid in [ + "2023-02-29", + "2024-02-30", + "2026-04-31", + "2026-00-10", + "2026-13-10", + "2026-01-00", + ] { + assert!( + parse_yyyy_mm_dd_utc(invalid).is_none(), + "{invalid} must be rejected" + ); + } + } + + #[test] + fn absolute_date_parser_requires_yyyy_mm_dd_shape() { + for invalid in [ + "2026-1-01", + "2026-01-1", + "26-01-01", + "20260101", + "2026/01/01", + "9223372036854775807-01-01", + ] { + assert!( + parse_yyyy_mm_dd_utc(invalid).is_none(), + "{invalid} must be rejected" + ); + } + } +} diff --git a/crates/ctx-contract/src/parse_refs.rs b/crates/ctx-contract/src/parse_refs.rs index be6ff72..2d0addc 100644 --- a/crates/ctx-contract/src/parse_refs.rs +++ b/crates/ctx-contract/src/parse_refs.rs @@ -86,31 +86,47 @@ pub fn extract_references(response: &[u8]) -> Vec { refs.push(r); }; - let reader = BufReader::new(Cursor::new(response)); + let mut reader = BufReader::new(Cursor::new(response)); + let mut raw_line = Vec::new(); let mut line_no: i32 = 0; - for line_res in reader.lines() { + loop { + raw_line.clear(); + let read = match reader.read_until(b'\n', &mut raw_line) { + Ok(read) => read, + Err(_) => break, + }; + if read == 0 { + break; + } + // PARITY (Phase 1 L-01): mirror bufio.Scanner. We increment // `line_no` at the top of the iteration *before* any early // continue/break so the numbering of accepted lines matches // Go's `for scanner.Scan() { lineNo++; ... }`. line_no += 1; - // BufReader::lines yields io::Result. For UTF-8 errors - // or other IO failures we silently skip the line, the closest - // analogue to bufio.Scanner's drop-on-error behaviour. - let line = match line_res { - Ok(s) => s, - Err(_) => continue, - }; + + // BufRead::lines() rejects the entire line on invalid UTF-8, while + // Go's Scanner.Text() preserves the bytes in a string. Read bytes + // directly and decode lossily so ASCII references later in a line + // remain discoverable even when unrelated bytes are malformed. + if raw_line.last() == Some(&b'\n') { + raw_line.pop(); + if raw_line.last() == Some(&b'\r') { + raw_line.pop(); + } + } + // PARITY (Phase 1 L-01): bufio.Scanner with `Buffer(..., 1MiB)` // returns false the *first* time a line exceeds the cap and // terminates Scan() entirely — subsequent lines are silently // dropped. We mirror that with `break`, not `continue`. - if line.len() > MAX_LINE { + if raw_line.len() > MAX_LINE { break; } + let line = String::from_utf8_lossy(&raw_line); // Diff headers — must run before generic path matcher. - if let Some(caps) = DIFF_HEADER_RE.captures(&line) { + if let Some(caps) = DIFF_HEADER_RE.captures(line.as_ref()) { if let Some(m) = caps.get(1) { add( &mut refs, @@ -133,7 +149,7 @@ pub fn extract_references(response: &[u8]) -> Vec { } // Path / line-range references. - for caps in PATH_REF_RE.captures_iter(&line) { + for caps in PATH_REF_RE.captures_iter(line.as_ref()) { let path = caps.name("path").map(|m| m.as_str()).unwrap_or(""); let start_opt = caps .name("start") @@ -176,7 +192,7 @@ pub fn extract_references(response: &[u8]) -> Vec { } // Symbol references inside backticks. - for caps in SYMBOL_RE.captures_iter(&line) { + for caps in SYMBOL_RE.captures_iter(line.as_ref()) { let sym = caps.get(1).map(|m| m.as_str()).unwrap_or(""); if looks_like_path(sym) { continue; @@ -243,4 +259,21 @@ mod tests { assert_eq!(refs[0].kind, "diff-header"); assert_eq!(refs[0].path, "internal/foo.go"); } + + #[test] + fn invalid_utf8_does_not_drop_ascii_references_on_the_line() { + let response = b"before.rs\nnoise:\xff see internal/latin.rs:L7-L9\nafter.rs\n"; + let refs = extract_references(response); + let paths: Vec<&str> = refs.iter().map(|r| r.path.as_str()).collect(); + + assert!(paths.contains(&"before.rs")); + assert!(paths.contains(&"internal/latin.rs")); + assert!(paths.contains(&"after.rs")); + let latin = refs + .iter() + .find(|r| r.path == "internal/latin.rs") + .expect("latin path reference"); + assert_eq!(latin.source_line, 2); + assert_eq!((latin.line_start, latin.line_end), (7, 9)); + } } diff --git a/crates/ctx-replay/src/prune.rs b/crates/ctx-replay/src/prune.rs index 38fc4bb..9cc39bf 100644 --- a/crates/ctx-replay/src/prune.rs +++ b/crates/ctx-replay/src/prune.rs @@ -58,27 +58,27 @@ pub fn parse_duration(s: &str) -> Result { let num_str = &rest[num_start..unit_start]; has_number = false; - match unit { - "d" => { - let d = atoi(num_str)?; - total += d * 24 * 3_600 * 1_000_000_000; - } - "w" => { - let d = atoi(num_str)?; - total += d * 7 * 24 * 3_600 * 1_000_000_000; - } - _ => { - let part = parse_go_duration_atom(num_str, unit) - .map_err(|e| format!("replay: invalid duration {rest:?}: {e}"))?; - total += part; - } - } + let part = match unit { + "d" => atoi(num_str)? + .checked_mul(24 * 3_600 * 1_000_000_000) + .ok_or_else(|| format!("replay: duration {rest:?} out of range"))?, + "w" => atoi(num_str)? + .checked_mul(7 * 24 * 3_600 * 1_000_000_000) + .ok_or_else(|| format!("replay: duration {rest:?} out of range"))?, + _ => parse_go_duration_atom(num_str, unit) + .map_err(|e| format!("replay: invalid duration {rest:?}: {e}"))?, + }; + total = total + .checked_add(part) + .ok_or_else(|| format!("replay: duration {rest:?} out of range"))?; } if has_number { return Err(format!("replay: duration {rest:?} missing unit")); } if neg { - total = -total; + total = total + .checked_neg() + .ok_or_else(|| format!("replay: duration {rest:?} out of range"))?; } Ok(total) } @@ -92,7 +92,11 @@ fn atoi(s: &str) -> Result { if !c.is_ascii_digit() { return Err(format!("replay: invalid number {s:?}")); } - n = n * 10 + (c as i64 - '0' as i64); + let digit = c as i64 - '0' as i64; + n = n + .checked_mul(10) + .and_then(|value| value.checked_add(digit)) + .ok_or_else(|| format!("replay: number {s:?} out of range"))?; } Ok(n) } @@ -111,7 +115,8 @@ fn parse_go_duration_atom(num: &str, unit: &str) -> Result { "h" => 3_600 * 1_000_000_000, other => return Err(format!("unknown unit {other:?}")), }; - Ok(n * mult) + n.checked_mul(mult) + .ok_or_else(|| format!("duration atom {num}{unit} out of range")) } /// Mirrors `replay.Prune`. @@ -123,7 +128,7 @@ pub fn prune(store: &Store, now: &str, older_nanos: i64) -> Result Result Option { +pub(crate) fn rfc3339_to_unix_nanos(s: &str) -> Option { let s = s.trim(); let len = s.len(); if len < 20 { @@ -172,38 +177,54 @@ fn rfc3339_to_unix_nanos(s: &str) -> Option { let mut frac_nanos: i64 = 0; if idx < len && b[idx] == b'.' { idx += 1; + const MAX_FRACTION_DIGITS: usize = 9; let start = idx; while idx < len && b[idx].is_ascii_digit() { idx += 1; } + if idx == start { + return None; + } let frac = &b[start..idx]; - // pad/truncate to 9 digits for nanoseconds. - let mut buf = [b'0'; 9]; - for (i, &c) in frac.iter().take(9).enumerate() { + // Nanosecond precision matches Go's time.Time. Extra fractional + // digits are accepted but truncated, matching the previous parser. + let mut buf = [b'0'; MAX_FRACTION_DIGITS]; + for (i, &c) in frac.iter().take(MAX_FRACTION_DIGITS).enumerate() { buf[i] = c; } frac_nanos = parse_int(&buf)?; } + let offset_secs: i64 = if idx >= len { return None; } else if b[idx] == b'Z' || b[idx] == b'z' { + idx += 1; 0 } else if b[idx] == b'+' || b[idx] == b'-' { let sign = if b[idx] == b'+' { 1 } else { -1 }; - if idx + 5 >= len { + // RFC3339 requires ±HH:MM and no trailing data. + if idx + 6 > len || b[idx + 3] != b':' { return None; } let oh = parse_int(&b[idx + 1..idx + 3])?; - // optional colon - let mm_start = if b[idx + 3] == b':' { idx + 4 } else { idx + 3 }; - let om = parse_int(&b[mm_start..mm_start + 2])?; - sign * (oh * 3600 + om * 60) + let om = parse_int(&b[idx + 4..idx + 6])?; + if oh > 23 || om > 59 { + return None; + } + idx += 6; + sign * (oh * 3_600 + om * 60) } else { return None; }; + if idx != len { + return None; + } let civil = civil_to_unix(y, mo, d, hh, mm, ss)?; - Some((civil - offset_secs) * 1_000_000_000 + frac_nanos) + civil + .checked_sub(offset_secs)? + .checked_mul(1_000_000_000)? + .checked_add(frac_nanos) } fn parse_int(bytes: &[u8]) -> Option { @@ -212,19 +233,35 @@ fn parse_int(bytes: &[u8]) -> Option { if !b.is_ascii_digit() { return None; } - n = n * 10 + (b - b'0') as i64; + n = n.checked_mul(10)?.checked_add((b - b'0') as i64)?; } Some(n) } +fn is_leap_year(year: i64) -> bool { + (year % 4 == 0 && year % 100 != 0) || year % 400 == 0 +} + +fn days_in_month(year: i64, month: i64) -> Option { + match month { + 1 | 3 | 5 | 7 | 8 | 10 | 12 => Some(31), + 4 | 6 | 9 | 11 => Some(30), + 2 if is_leap_year(year) => Some(29), + 2 => Some(28), + _ => None, + } +} + /// Converts a civil (UTC) date-time to a Unix timestamp in seconds. /// /// Uses the Hinnant-style days-from-civil algorithm. fn civil_to_unix(y: i64, m: i64, d: i64, hh: i64, mm: i64, ss: i64) -> Option { - if !(1..=12).contains(&m) { - return None; - } - if !(1..=31).contains(&d) { + let max_day = days_in_month(y, m)?; + if !(1..=max_day).contains(&d) + || !(0..=23).contains(&hh) + || !(0..=59).contains(&mm) + || !(0..=59).contains(&ss) + { return None; } let y = if m <= 2 { y - 1 } else { y }; @@ -233,7 +270,10 @@ fn civil_to_unix(y: i64, m: i64, d: i64, hh: i64, mm: i64, ss: i64) -> Option 2 { m - 3 } else { m + 9 }) + 2) / 5 + d - 1; let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy; let days = era * 146_097 + doe - 719_468; - Some(days * 86_400 + hh * 3600 + mm * 60 + ss) + days.checked_mul(86_400)? + .checked_add(hh.checked_mul(3_600)?)? + .checked_add(mm.checked_mul(60)?)? + .checked_add(ss) } #[cfg(test)] @@ -272,6 +312,13 @@ mod tests { assert!(parse_duration("5").is_err()); } + #[test] + fn parse_duration_overflow_is_rejected() { + assert!(parse_duration("9223372036854775807h").is_err()); + assert!(parse_duration("92233720368547758070ns").is_err()); + assert!(parse_duration("9223372036854775807ns1ns").is_err()); + } + #[test] fn rfc3339_round_trip() { let ns = rfc3339_to_unix_nanos("2026-05-29T12:00:00Z").unwrap(); @@ -287,4 +334,41 @@ mod tests { let b = rfc3339_to_unix_nanos("2026-05-29T13:00:00+01:00").unwrap(); assert_eq!(a, b); } + + #[test] + fn rfc3339_validates_calendar_and_clock() { + assert!(rfc3339_to_unix_nanos("2024-02-29T23:59:59Z").is_some()); + + for invalid in [ + "2026-02-29T12:00:00Z", + "2024-02-30T12:00:00Z", + "2026-04-31T12:00:00Z", + "2026-05-29T24:00:00Z", + "2026-05-29T12:60:00Z", + "2026-05-29T12:00:60Z", + "2026-05-29T12:00:00+24:00", + "2026-05-29T12:00:00+00:60", + ] { + assert!( + rfc3339_to_unix_nanos(invalid).is_none(), + "{invalid} must be rejected" + ); + } + } + + #[test] + fn rfc3339_rejects_malformed_suffixes_and_range_overflow() { + for invalid in [ + "2026-05-29T12:00:00.Z", + "2026-05-29T12:00:00Zjunk", + "2026-05-29T12:00:00+0100", + "2026-05-29T12:00:00+01:00junk", + "9999-01-01T00:00:00Z", + ] { + assert!( + rfc3339_to_unix_nanos(invalid).is_none(), + "{invalid} must be rejected" + ); + } + } } diff --git a/crates/ctx-replay/src/session.rs b/crates/ctx-replay/src/session.rs index a9dae3a..acb655f 100644 --- a/crates/ctx-replay/src/session.rs +++ b/crates/ctx-replay/src/session.rs @@ -47,6 +47,7 @@ use std::sync::Mutex; use serde::Deserialize; use crate::diff::{compute, compute_selection_diff, sort_selection_diff, DiffOptions}; +use crate::prune::rfc3339_to_unix_nanos; use crate::store::{open_store, Store, StoreError}; use crate::types::{DiffSummary, Manifest, SelectionSummary}; @@ -216,16 +217,12 @@ impl ReplaySession { return Err(QueryError::BadArgs); } let manifests = self.list_manifests()?; - // Reuse the prune module's RFC3339 → unix-nanos via parse_duration - // helpers — these are crate-private, so we inline the tiny call to - // store::Store::list_filtered indirectly by mirroring the prune - // function but read-only. - let now_nanos = rfc3339_to_nanos(&args.now).ok_or(QueryError::BadArgs)?; + let now_nanos = rfc3339_to_unix_nanos(&args.now).ok_or(QueryError::BadArgs)?; let cutoff = now_nanos.saturating_sub(args.older_nanos); let mut candidates: Vec = Vec::new(); let mut kept: i64 = 0; for m in &manifests { - let ts = rfc3339_to_nanos(&m.created_at).unwrap_or(i64::MAX); + let ts = rfc3339_to_unix_nanos(&m.created_at).unwrap_or(i64::MAX); if ts < cutoff { candidates.push(m.id.clone()); } else { @@ -317,101 +314,6 @@ fn map_store_err(e: StoreError) -> QueryError { } } -/// Minimal RFC3339 → unix-nanos parser, duplicated from prune.rs because -/// that helper is private. Same algorithm (Hinnant days-from-civil), -/// covering Z and ±HH:MM offsets. -fn rfc3339_to_nanos(s: &str) -> Option { - let s = s.trim(); - let len = s.len(); - if len < 20 { - return None; - } - let b = s.as_bytes(); - let y = parse_int(&b[0..4])?; - if b[4] != b'-' { - return None; - } - let mo = parse_int(&b[5..7])?; - if b[7] != b'-' { - return None; - } - let d = parse_int(&b[8..10])?; - if b[10] != b'T' && b[10] != b't' && b[10] != b' ' { - return None; - } - let hh = parse_int(&b[11..13])?; - if b[13] != b':' { - return None; - } - let mm = parse_int(&b[14..16])?; - if b[16] != b':' { - return None; - } - let ss = parse_int(&b[17..19])?; - - let mut idx = 19usize; - let mut frac_nanos: i64 = 0; - if idx < len && b[idx] == b'.' { - idx += 1; - let start = idx; - while idx < len && b[idx].is_ascii_digit() { - idx += 1; - } - let frac = &b[start..idx]; - let mut buf = [b'0'; 9]; - for (i, &c) in frac.iter().take(9).enumerate() { - buf[i] = c; - } - frac_nanos = parse_int(&buf)?; - } - let offset_secs: i64 = if idx >= len { - return None; - } else if b[idx] == b'Z' || b[idx] == b'z' { - 0 - } else if b[idx] == b'+' || b[idx] == b'-' { - let sign = if b[idx] == b'+' { 1 } else { -1 }; - if idx + 5 >= len { - return None; - } - let oh = parse_int(&b[idx + 1..idx + 3])?; - let mm_start = if b[idx + 3] == b':' { idx + 4 } else { idx + 3 }; - let om = parse_int(&b[mm_start..mm_start + 2])?; - sign * (oh * 3600 + om * 60) - } else { - return None; - }; - - let civil = civil_to_unix(y, mo, d, hh, mm, ss)?; - Some((civil - offset_secs) * 1_000_000_000 + frac_nanos) -} - -fn parse_int(bytes: &[u8]) -> Option { - let mut n: i64 = 0; - for &b in bytes { - if !b.is_ascii_digit() { - return None; - } - n = n * 10 + (b - b'0') as i64; - } - Some(n) -} - -fn civil_to_unix(y: i64, m: i64, d: i64, hh: i64, mm: i64, ss: i64) -> Option { - if !(1..=12).contains(&m) { - return None; - } - if !(1..=31).contains(&d) { - return None; - } - let y = if m <= 2 { y - 1 } else { y }; - let era = if y >= 0 { y } else { y - 399 } / 400; - let yoe = y - era * 400; - let doy = (153 * (if m > 2 { m - 3 } else { m + 9 }) + 2) / 5 + d - 1; - let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy; - let days = era * 146_097 + doe - 719_468; - Some(days * 86_400 + hh * 3600 + mm * 60 + ss) -} - /// Error variants for `ReplaySession::query`. Mapped to FFI return codes /// by the ffi.rs glue. #[derive(Debug)] @@ -622,5 +524,19 @@ mod tests { s.query("load", r#"{"id":""}"#), Err(QueryError::BadArgs) )); + assert!(matches!( + s.query( + "prune_candidates", + r#"{"now":"2026-02-31T12:00:00Z","older_nanos":0}"#, + ), + Err(QueryError::BadArgs) + )); + assert!(matches!( + s.query( + "prune_candidates", + r#"{"now":"2026-05-29T12:00:00Zjunk","older_nanos":0}"#, + ), + Err(QueryError::BadArgs) + )); } } diff --git a/crates/ctx-web/src/handlers/file.rs b/crates/ctx-web/src/handlers/file.rs index efa60d0..1b881cb 100644 --- a/crates/ctx-web/src/handlers/file.rs +++ b/crates/ctx-web/src/handlers/file.rs @@ -309,7 +309,16 @@ fn cache_put( fn git_status_for_file(root: &str, rel_slash: &str) -> String { let output = Command::new("git") - .args(["-C", root, "status", "--porcelain", "--", rel_slash]) + .arg("-C") + .arg(root) + .args([ + "status", + "--porcelain=v1", + "-z", + "--untracked-files=all", + "--", + rel_slash, + ]) .output(); let Ok(output) = output else { return String::new(); @@ -318,17 +327,9 @@ fn git_status_for_file(root: &str, rel_slash: &str) -> String { return String::new(); } - let stdout = String::from_utf8_lossy(&output.stdout); - for line in stdout.lines() { - if line.len() < 4 { - continue; - } - let path = crate::handlers::tree::normalize_git_status_path(&line[3..]); - if path == rel_slash { - return crate::handlers::tree::normalize_git_status(&line[..2]); - } - } - String::new() + crate::handlers::tree::parse_git_status_map(&output.stdout) + .remove(rel_slash) + .unwrap_or_default() } fn truncate_file_data(data: &[u8]) -> (&[u8], bool) { diff --git a/crates/ctx-web/src/handlers/tree.rs b/crates/ctx-web/src/handlers/tree.rs index 3546210..ce03217 100644 --- a/crates/ctx-web/src/handlers/tree.rs +++ b/crates/ctx-web/src/handlers/tree.rs @@ -106,18 +106,46 @@ fn subtract_filter_duration( } fn parse_yyyy_mm_dd_utc(input: &str) -> Option { - let mut parts = input.split('-'); - let year = parts.next()?.parse::().ok()?; - let month = parts.next()?.parse::().ok()?; - let day = parts.next()?.parse::().ok()?; - if parts.next().is_some() || !(1..=12).contains(&month) || !(1..=31).contains(&day) { + let bytes = input.as_bytes(); + if bytes.len() != 10 + || bytes[4] != b'-' + || bytes[7] != b'-' + || bytes + .iter() + .enumerate() + .any(|(index, byte)| index != 4 && index != 7 && !byte.is_ascii_digit()) + { return None; } + + let year = input[0..4].parse::().ok()?; + let month = input[5..7].parse::().ok()?; + let day = input[8..10].parse::().ok()?; + let max_day = days_in_month(year, month)?; + if day == 0 || day > max_day { + return None; + } + let days = days_from_civil(year, month, day); if days < 0 { return None; } - Some(UNIX_EPOCH + Duration::from_secs(days as u64 * 24 * 60 * 60)) + let seconds = (days as u64).checked_mul(24 * 60 * 60)?; + UNIX_EPOCH.checked_add(Duration::from_secs(seconds)) +} + +fn is_leap_year(year: i64) -> bool { + (year % 4 == 0 && year % 100 != 0) || year % 400 == 0 +} + +fn days_in_month(year: i64, month: u32) -> Option { + match month { + 1 | 3 | 5 | 7 | 8 | 10 | 12 => Some(31), + 4 | 6 | 9 | 11 => Some(30), + 2 if is_leap_year(year) => Some(29), + 2 => Some(28), + _ => None, + } } fn days_from_civil(year: i64, month: u32, day: u32) -> i64 { @@ -630,7 +658,9 @@ struct GitStatusMap { impl GitStatusMap { fn load(root: &str) -> Self { let output = Command::new("git") - .args(["-C", root, "status", "--porcelain"]) + .arg("-C") + .arg(root) + .args(["status", "--porcelain=v1", "-z", "--untracked-files=normal"]) .output(); let Ok(output) = output else { return Self::default(); @@ -639,23 +669,9 @@ impl GitStatusMap { return Self::default(); } - let mut by_path = BTreeMap::new(); - let stdout = String::from_utf8_lossy(&output.stdout); - for line in stdout.lines() { - if line.len() < 4 { - continue; - } - let code = normalize_git_status(&line[..2]); - if code.is_empty() { - continue; - } - let raw_path = &line[3..]; - let path = normalize_git_status_path(raw_path); - if !path.is_empty() { - by_path.insert(path, code); - } + Self { + by_path: parse_git_status_map(&output.stdout), } - Self { by_path } } fn status_for(&self, rel: &str, is_dir: bool) -> String { @@ -725,12 +741,33 @@ pub(crate) fn normalize_git_status(status: &str) -> String { String::new() } -pub(crate) fn normalize_git_status_path(raw: &str) -> String { - let mut path = raw.trim(); - if let Some((_, new_path)) = path.split_once(" -> ") { - path = new_path; +pub(crate) fn parse_git_status_map(output: &[u8]) -> BTreeMap { + let mut by_path = BTreeMap::new(); + let mut fields = output.split(|byte| *byte == 0); + + while let Some(record) = fields.next() { + if record.is_empty() { + continue; + } + if record.len() < 4 || record[2] != b' ' { + continue; + } + + let status = std::str::from_utf8(&record[..2]).unwrap_or(""); + let code = normalize_git_status(status); + let path = &record[3..]; + if !code.is_empty() && !path.is_empty() { + by_path.insert(String::from_utf8_lossy(path).into_owned(), code); + } + + // Porcelain v1 -z writes rename/copy destinations first, followed by + // the original path as a second NUL-delimited field. + if status.bytes().any(|byte| matches!(byte, b'R' | b'C')) { + let _ = fields.next(); + } } - path.trim_matches('"').replace('\\', "/") + + by_path } fn git_status_rank(status: &str) -> i32 { @@ -775,15 +812,20 @@ mod tests { use super::*; #[test] - fn git_status_path_normalizes_rename_target() { + fn git_status_z_parser_preserves_special_paths_and_rename_target() { + let raw = b"?? notes/a b.md\0 M literal\\name.rs\0R dst -> literal.rs\0src old.rs\0"; + let status = parse_git_status_map(raw); + + assert_eq!(status.get("notes/a b.md").map(String::as_str), Some("?")); assert_eq!( - normalize_git_status_path("old/name.rs -> src/name.rs"), - "src/name.rs" + status.get(r"literal\name.rs").map(String::as_str), + Some("M") ); assert_eq!( - normalize_git_status_path("\"web/src/App.svelte\""), - "web/src/App.svelte" + status.get("dst -> literal.rs").map(String::as_str), + Some("R") ); + assert!(!status.contains_key("src old.rs")); } #[test] @@ -858,6 +900,11 @@ mod tests { assert!(parse_pack_time_filter("", now).is_err()); assert!(parse_pack_time_filter("0d", now).is_err()); assert!(parse_pack_time_filter("-1d", now).is_err()); + assert!(parse_pack_time_filter("2023-02-29", now).is_err()); + assert!(parse_pack_time_filter("2024-02-30", now).is_err()); + assert!(parse_pack_time_filter("2026-04-31", now).is_err()); + assert!(parse_pack_time_filter("2026-1-01", now).is_err()); + assert!(parse_pack_time_filter("9223372036854775807-01-01", now).is_err()); } // ----------------------------------------------------------------------- diff --git a/web/src/lib/router.svelte.ts b/web/src/lib/router.svelte.ts index 6a018d4..26005c4 100644 --- a/web/src/lib/router.svelte.ts +++ b/web/src/lib/router.svelte.ts @@ -49,18 +49,36 @@ export interface Route { gitMode?: GitReviewMode; } +function safeDecodeURIComponent(raw: string): string { + try { + return decodeURIComponent(raw); + } catch { + // Hashes can be supplied by hand or by external links. A malformed + // percent escape should remain literal instead of crashing the SPA. + return raw; + } +} + +// The open query value is a comma-delimited list whose individual paths are +// URI-encoded. Read its raw value before URLSearchParams decodes it so an +// encoded comma (%2C) inside a filename is not mistaken for a list separator. +function rawQueryParam(rawQuery: string, name: string): string | null { + for (const field of rawQuery.split('&')) { + const eq = field.indexOf('='); + const rawName = eq === -1 ? field : field.slice(0, eq); + if (safeDecodeURIComponent(rawName.replace(/\+/g, ' ')) !== name) continue; + return eq === -1 ? '' : field.slice(eq + 1); + } + return null; +} + function parseOpenParam(raw: string | null): string[] { if (!raw) return []; const out: string[] = []; const seen = new Set(); for (const token of raw.split(',')) { if (!token) continue; - let decoded: string; - try { - decoded = decodeURIComponent(token); - } catch { - decoded = token; - } + const decoded = safeDecodeURIComponent(token.replace(/\+/g, ' ')); if (!decoded || seen.has(decoded)) continue; seen.add(decoded); out.push(decoded); @@ -119,24 +137,18 @@ function parse(rawHash: string): Route { const modeRaw = queryParams.get('mode'); const mode: FileViewMode | undefined = modeRaw === 'diff' || modeRaw === 'history' ? modeRaw : undefined; - const rightRaw = queryParams.get('right'); - let rightPath = ''; - if (rightRaw) { - try { - rightPath = decodeURIComponent(rightRaw); - } catch { - rightPath = rightRaw; - } - } + // URLSearchParams already percent-decodes query values once. Decoding + // right a second time corrupts literal "%xx" sequences in filenames. + const rightPath = queryParams.get('right') ?? ''; const since = queryParams.get('since') ?? undefined; const until = queryParams.get('until') ?? undefined; const useMtime = queryParams.get('use_mtime') === 'true' ? true : undefined; return { name: 'file', - path: decodeURIComponent(p.slice('file/'.length)), + path: safeDecodeURIComponent(p.slice('file/'.length)), query: '', lineHint, - openPaths: parseOpenParam(queryParams.get('open')), + openPaths: parseOpenParam(rawQueryParam(afterQ, 'open')), rightPath, mode, since, @@ -156,7 +168,7 @@ function parse(rawHash: string): Route { const useMtime = queryParams.get('use_mtime') === 'true' ? true : undefined; return { name: 'dir', - path: decodeURIComponent(p.slice('dir/'.length)), + path: safeDecodeURIComponent(p.slice('dir/'.length)), query: '', openPaths: [], rightPath: '', @@ -186,7 +198,7 @@ function parse(rawHash: string): Route { // path segment is the selected commit's full hash (opaque, no slashes). return { name: 'gitlog', - path: decodeURIComponent(p.slice('gitlog/'.length)), + path: safeDecodeURIComponent(p.slice('gitlog/'.length)), query: '', openPaths: [], rightPath: '',