use byte_unit::{Byte, UnitType}; use clap::{CommandFactory, Parser, ValueEnum}; use clap_complete::{generate, Shell}; use colored::*; use comfy_table::presets::UTF8_FULL; use comfy_table::{Attribute, Cell, CellAlignment, Color, ContentArrangement, Table}; use csv::WriterBuilder; use std::cmp::Ordering; use std::io::Write; use std::path::{Path, PathBuf}; use std::str::FromStr; pub mod scan; pub use scan::{ scan_tree, EntryType, FileIdentity, Node, ScanControl, ScanError, ScanIssue, ScanOptions, ScanReport, }; #[derive(Parser, Debug, Clone)] #[command(author, version, about, long_about = None)] #[command(disable_version_flag = true)] pub struct Args { /// Print version #[arg(short = 'v', long, action = clap::ArgAction::Version)] pub version: Option, /// Directory to analyze #[arg(default_value = ".")] pub path: PathBuf, /// Minimum size filter (e.g. "10MB", "1gb") #[arg(short = 'm', long)] pub min_size: Option, /// Limit the number of results per table #[arg(short = 'n', long)] pub number: Option, /// Sort by columns (e.g. "type:asc,size:dsc"). /// defaults to size:dsc. /// Columns: name, type, size (disk usage), apparent (logical), blocks. #[arg(long)] pub sort: Option, /// Show apparent size (logical file length) in addition to disk usage #[arg(short = 'a', long)] pub apparent: bool, /// Depth to traverse (0 = only immediate children) #[arg(short, long, default_value = "0")] pub depth: usize, /// Display full absolute paths #[arg(short = 'f', long, conflicts_with = "path_relative")] pub path_full: bool, /// Display paths relative to current directory (default) #[arg(long)] pub path_relative: bool, /// Number of threads to use (defaults to available logical CPUs) #[arg(short = 'j', long = "threads")] pub threads: Option, /// Stay on the target's filesystem instead of descending into other mounts #[arg(short = 'x', long)] pub one_file_system: bool, /// Generate shell completions #[arg(long, value_enum)] pub completions: Option, /// Respect .gitignore and .ignore files #[arg(short = 'i', long)] pub ignore: bool, /// Output format #[arg(long, value_enum, default_value_t = OutputFormat::Text)] pub format: OutputFormat, /// Save output to file (optional filename, defaults to timestamped name) #[arg(long, num_args(0..=1), default_missing_value = "_AUTO_")] pub save: Option, /// Precision for size output (decimal places) #[arg(long, default_value = "2")] pub precision: usize, /// Unit system to use #[arg(long, value_enum, default_value_t = UnitSystem::Binary)] pub units: UnitSystem, /// Show comparison between total and non-ignored files #[arg(short = 'c', long)] pub compare: bool, /// Generate man page to the specified output directory #[arg(long, hide = true)] pub generate_man_page: Option, } #[derive(ValueEnum, Clone, Debug, PartialEq, Eq)] pub enum OutputFormat { Text, Csv, Json, } #[derive(ValueEnum, Clone, Debug, PartialEq, Eq)] pub enum UnitSystem { Binary, Decimal, } #[derive(Debug, Clone, PartialEq, Eq)] pub enum SortColumn { Name, Type, Disk, Apparent, Blocks, } impl FromStr for SortColumn { type Err = String; fn from_str(s: &str) -> Result { match s.to_lowercase().as_str() { "name" | "n" => Ok(SortColumn::Name), "type" | "t" => Ok(SortColumn::Type), "size" | "s" | "disk" | "d" => Ok(SortColumn::Disk), "apparent" | "a" => Ok(SortColumn::Apparent), "blocks" | "b" => Ok(SortColumn::Blocks), _ => Err(format!("Unknown column: {s}")), } } } #[derive(Debug, Clone, PartialEq, Eq)] pub enum SortDirection { Asc, Dsc, } impl FromStr for SortDirection { type Err = String; fn from_str(s: &str) -> Result { match s.to_lowercase().as_str() { "asc" | "a" => Ok(SortDirection::Asc), "dsc" | "d" | "desc" => Ok(SortDirection::Dsc), _ => Err(format!("Unknown direction: {s}")), } } } /// Returns whether the scan completed without missing entries. The caller /// chooses the exit status; partial reports remain available to the user. pub fn run(args: Args, mut writer: &mut dyn Write) -> bool { if let Some(shell) = args.completions { let mut cmd = Args::command(); let name = cmd.get_name().to_string(); generate(shell, &mut cmd, name, &mut std::io::stdout()); return true; } if let Some(out_dir) = args.generate_man_page { let cmd = Args::command(); let man = clap_mangen::Man::new(cmd); let mut buffer: Vec = Default::default(); man.render(&mut buffer).expect("Failed to render man page"); std::fs::create_dir_all(&out_dir).expect("Failed to create man page directory"); let file_path = out_dir.join("sized.1"); std::fs::write(&file_path, buffer).expect("Failed to write man page"); println!("Man page generated at {}", file_path.display()); return true; } let target_path = args.path.clone(); let min_bytes = if let Some(size_str) = &args.min_size { match Byte::parse_str(size_str, true) { Ok(byte) => byte.as_u64(), Err(e) => { eprintln!("Error parsing size '{size_str}': {e}"); std::process::exit(1); } } } else { 0 }; let sort_criteria = parse_sort_arg(args.sort.as_deref().unwrap_or("size:dsc")); if args.format == OutputFormat::Csv { if args.apparent { writeln!(writer, "Path,Type,Apparent(Bytes),DiskUsage(Bytes),Blocks") .expect("Failed to write CSV header"); } else { writeln!(writer, "Path,Type,DiskUsage(Bytes),Blocks") .expect("Failed to write CSV header"); } } let report = match scan_tree( &target_path, &ScanOptions { respect_ignore: args.ignore, compare_ignore: args.compare, stay_on_filesystem: args.one_file_system, threads: args.threads, }, &ScanControl::default(), ) { Ok(report) => report, Err(e) => { eprintln!("Error: {e}"); return false; } }; for issue in &report.issues { eprintln!( "Warning: incomplete scan at '{}': {}", issue.path.display(), issue.message ); } let root_node = report.root; if root_node.size_bytes < min_bytes { if args.format == OutputFormat::Text { writeln!(writer, "Root directory is smaller than minimum size.").ok(); } return root_node.complete; } process_node_recursive(&mut writer, &root_node, 0, &args, min_bytes, &sort_criteria); root_node.complete } pub fn build_tree(path: &Path, ignore: bool, compare: bool) -> Node { // Compatibility helper; callers needing diagnostics should use scan_tree. scan_tree( path, &ScanOptions { respect_ignore: ignore, compare_ignore: compare, ..ScanOptions::default() }, &ScanControl::default(), ) .map(|report| report.root) .unwrap_or_else(|_| Node::unavailable(path)) } fn process_node_recursive( writer: &mut dyn Write, node: &Node, current_depth: usize, args: &Args, min_bytes: u64, sort_criteria: &[(SortColumn, SortDirection)], ) { let mut display_children: Vec<&Node> = node .children .iter() .filter(|c| c.size_bytes >= min_bytes) .collect(); display_children.sort_by(|a, b| compare_nodes(a, b, sort_criteria)); if let Some(n) = args.number { if n < display_children.len() { display_children.truncate(n); } } let show_relative = !args.path_full; let relative_path = if show_relative { let cwd = std::env::current_dir().unwrap_or_default(); node.path .strip_prefix(&cwd) .unwrap_or(&node.path) .display() .to_string() } else { node.path .canonicalize() .unwrap_or(node.path.clone()) .display() .to_string() }; print_output( writer, node, &display_children, args, &relative_path, current_depth, ); if current_depth < args.depth { for child in display_children { if child.entry_type == EntryType::Dir { process_node_recursive( writer, child, current_depth + 1, args, min_bytes, sort_criteria, ); } } } } pub fn compare_nodes(a: &Node, b: &Node, criteria: &[(SortColumn, SortDirection)]) -> Ordering { for (col, dir) in criteria { let order = match col { SortColumn::Name => a.path.file_name().cmp(&b.path.file_name()), SortColumn::Type => a.entry_type.cmp(&b.entry_type), SortColumn::Disk => a.blocks.cmp(&b.blocks), SortColumn::Apparent => a.size_bytes.cmp(&b.size_bytes), SortColumn::Blocks => a.blocks.cmp(&b.blocks), }; if order != Ordering::Equal { return match dir { SortDirection::Asc => order, SortDirection::Dsc => order.reverse(), }; } } Ordering::Equal } pub fn parse_sort_arg(s: &str) -> Vec<(SortColumn, SortDirection)> { s.split(',') .filter_map(|part| { let parts: Vec<&str> = part.split(':').collect(); if parts.is_empty() { return None; } let col_str = parts[0]; let col = SortColumn::from_str(col_str).ok()?; let dir = if parts.len() > 1 { SortDirection::from_str(parts[1]).unwrap_or(SortDirection::Dsc) } else { match col { SortColumn::Name | SortColumn::Type => SortDirection::Asc, _ => SortDirection::Dsc, } }; Some((col, dir)) }) .collect() } fn print_output( writer: &mut dyn Write, parent_node: &Node, children: &[&Node], args: &Args, relative_path: &str, depth: usize, ) { match args.format { OutputFormat::Text => print_text(writer, parent_node, children, args, relative_path, depth), OutputFormat::Csv => print_csv(writer, children, args), OutputFormat::Json => print_json(writer, parent_node, children, args), } } fn print_text( writer: &mut dyn Write, parent_node: &Node, children: &[&Node], args: &Args, relative_path: &str, depth: usize, ) { if depth > 0 { writeln!(writer, "{}", "-".repeat(60).dimmed()).ok(); } writeln!(writer, "\n{}", relative_path.bold().blue()).ok(); let unit_type = match args.units { UnitSystem::Binary => UnitType::Binary, UnitSystem::Decimal => UnitType::Decimal, }; if parent_node.accessible { let total_usage = Byte::from_u64(parent_node.blocks * 512).get_appropriate_unit(unit_type); let mut summary = if args.apparent { let total_apparent = Byte::from_u64(parent_node.size_bytes).get_appropriate_unit(unit_type); format!( "Total Apparent: {} / Disk Usage: {} ({} blocks)", format!( "{:.precision$} {}", total_apparent.get_value(), total_apparent.get_unit(), precision = args.precision ) .bold() .green(), format!( "{:.precision$} {}", total_usage.get_value(), total_usage.get_unit(), precision = args.precision ) .bold() .green(), parent_node.blocks ) } else { format!( "Disk Usage: {} ({} blocks)", format!( "{:.precision$} {}", total_usage.get_value(), total_usage.get_unit(), precision = args.precision ) .bold() .green(), parent_node.blocks ) }; if args.compare { let total_usage_filtered = Byte::from_u64(parent_node.blocks_filtered * 512).get_appropriate_unit(unit_type); if args.apparent { let total_apparent_filtered = Byte::from_u64(parent_node.size_bytes_filtered).get_appropriate_unit(unit_type); summary.push_str(&format!( "\nFiltered Apparent: {} / Filtered Disk Usage: {} ({} blocks)", format!( "{:.precision$} {}", total_apparent_filtered.get_value(), total_apparent_filtered.get_unit(), precision = args.precision ) .bold() .cyan(), format!( "{:.precision$} {}", total_usage_filtered.get_value(), total_usage_filtered.get_unit(), precision = args.precision ) .bold() .cyan(), parent_node.blocks_filtered )); } else { summary.push_str(&format!( "\nFiltered Disk Usage: {} ({} blocks)", format!( "{:.precision$} {}", total_usage_filtered.get_value(), total_usage_filtered.get_unit(), precision = args.precision ) .bold() .cyan(), parent_node.blocks_filtered )); } } writeln!(writer, "{summary}").ok(); } else { writeln!(writer, "Total size: {}", "Access Denied".bold().red()).ok(); } if children.is_empty() { if parent_node.accessible { writeln!(writer, "(No children)").ok(); } return; } let mut table = Table::new(); table .load_preset(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); let mut headers = vec![ Cell::new("Name").add_attribute(Attribute::Bold), Cell::new("Type").add_attribute(Attribute::Bold), ]; if args.apparent { headers.push( Cell::new("Apparent") .add_attribute(Attribute::Bold) .set_alignment(CellAlignment::Right), ); } headers.push( Cell::new("Disk Usage") .add_attribute(Attribute::Bold) .set_alignment(CellAlignment::Right), ); if args.compare { if args.apparent { headers.push( Cell::new("App. Filtered") .add_attribute(Attribute::Bold) .set_alignment(CellAlignment::Right), ); } headers.push( Cell::new("Disk Filtered") .add_attribute(Attribute::Bold) .set_alignment(CellAlignment::Right), ); } headers.push( Cell::new("Blocks") .add_attribute(Attribute::Bold) .set_alignment(CellAlignment::Right), ); table.set_header(headers); for child in children { let name = child.path.file_name().unwrap_or_default().to_string_lossy(); let color = if child.entry_type == EntryType::Dir { Color::Blue } else { Color::Cyan }; if child.accessible { let mut row = vec![ Cell::new(name).fg(color), Cell::new(&child.entry_type).fg(Color::Yellow), ]; if args.apparent { let child_byte = Byte::from_u64(child.size_bytes).get_appropriate_unit(unit_type); row.push( Cell::new(format!( "{:.precision$} {}", child_byte.get_value(), child_byte.get_unit(), precision = args.precision )) .fg(Color::Green) .set_alignment(CellAlignment::Right), ); } let disk_usage_byte = Byte::from_u64(child.blocks * 512).get_appropriate_unit(unit_type); row.push( Cell::new(format!( "{:.precision$} {}", disk_usage_byte.get_value(), disk_usage_byte.get_unit(), precision = args.precision )) .fg(Color::Green) .set_alignment(CellAlignment::Right), ); if args.compare { if args.apparent { let child_byte_filtered = Byte::from_u64(child.size_bytes_filtered).get_appropriate_unit(unit_type); row.push( Cell::new(format!( "{:.precision$} {}", child_byte_filtered.get_value(), child_byte_filtered.get_unit(), precision = args.precision )) .fg(Color::Cyan) .set_alignment(CellAlignment::Right), ); } let disk_usage_byte_filtered = Byte::from_u64(child.blocks_filtered * 512).get_appropriate_unit(unit_type); row.push( Cell::new(format!( "{:.precision$} {}", disk_usage_byte_filtered.get_value(), disk_usage_byte_filtered.get_unit(), precision = args.precision )) .fg(Color::Cyan) .set_alignment(CellAlignment::Right), ); } row.push( Cell::new(child.blocks) .fg(Color::Green) .set_alignment(CellAlignment::Right), ); table.add_row(row); } else { let mut row = vec![ Cell::new(name).fg(color), Cell::new(&child.entry_type).fg(Color::Yellow), Cell::new("N/A") .fg(Color::Red) .set_alignment(CellAlignment::Right), Cell::new("Unavailable") .fg(Color::Red) .set_alignment(CellAlignment::Right), ]; if args.compare { row.push( Cell::new("-") .fg(Color::Red) .set_alignment(CellAlignment::Right), ); row.push( Cell::new("-") .fg(Color::Red) .set_alignment(CellAlignment::Right), ); } row.push( Cell::new("-") .fg(Color::Red) .set_alignment(CellAlignment::Right), ); table.add_row(row); } } writeln!(writer, "{table}").ok(); } fn print_csv(writer: &mut dyn Write, children: &[&Node], args: &Args) { let mut wtr = WriterBuilder::new().has_headers(false).from_writer(writer); for child in children { let mut record = vec![ child.path.to_string_lossy().to_string(), child.entry_type.to_string(), ]; if args.apparent { record.push(child.size_bytes.to_string()); } record.push((child.blocks * 512).to_string()); record.push(child.blocks.to_string()); wtr.write_record(&record).ok(); } wtr.flush().ok(); } fn print_json(writer: &mut dyn Write, parent_node: &Node, children: &[&Node], args: &Args) { let entries_view: Vec = children .iter() .map(|c| { let mut val = serde_json::json!({ "path": &c.path, "entry_type": c.entry_type.to_string(), "accessible": c.accessible, "complete": c.complete, "hard_link_duplicate": c.hard_link_duplicate, "skipped_mount": c.skipped_mount, "disk_usage": c.blocks * 512, "blocks": c.blocks, }); if args.apparent { val.as_object_mut() .unwrap() .insert("apparent_size".to_string(), serde_json::json!(c.size_bytes)); } if args.compare { val.as_object_mut().unwrap().insert( "disk_usage_filtered".to_string(), serde_json::json!(c.blocks_filtered * 512), ); val.as_object_mut().unwrap().insert( "blocks_filtered".to_string(), serde_json::json!(c.blocks_filtered), ); if args.apparent { val.as_object_mut().unwrap().insert( "apparent_size_filtered".to_string(), serde_json::json!(c.size_bytes_filtered), ); } } val }) .collect(); let mut report = serde_json::json!({ "path": &parent_node.path, "total_disk_usage": parent_node.blocks * 512, "total_blocks": parent_node.blocks, "entries": entries_view, "accessible": parent_node.accessible, "complete": parent_node.complete, }); if args.apparent { report.as_object_mut().unwrap().insert( "total_apparent".to_string(), serde_json::json!(parent_node.size_bytes), ); } if args.compare { report.as_object_mut().unwrap().insert( "total_disk_usage_filtered".to_string(), serde_json::json!(parent_node.blocks_filtered * 512), ); report.as_object_mut().unwrap().insert( "total_blocks_filtered".to_string(), serde_json::json!(parent_node.blocks_filtered), ); if args.apparent { report.as_object_mut().unwrap().insert( "total_apparent_filtered".to_string(), serde_json::json!(parent_node.size_bytes_filtered), ); } } serde_json::to_writer(&mut *writer, &report).ok(); writer.write_all(b"\n").ok(); } #[cfg(test)] mod tests { use super::*; use std::fs::File; use tempfile::TempDir; #[test] fn test_parse_sort_arg() { let criteria = parse_sort_arg("size:asc,name"); assert_eq!(criteria.len(), 2); assert_eq!(criteria[0], (SortColumn::Disk, SortDirection::Asc)); assert_eq!(criteria[1], (SortColumn::Name, SortDirection::Asc)); // Default for Name is Asc } #[test] fn test_sort_defaults() { let criteria = parse_sort_arg("size"); assert_eq!(criteria[0], (SortColumn::Disk, SortDirection::Dsc)); // Default for Size is Dsc } #[test] fn test_build_tree() { let temp_dir = TempDir::new().unwrap(); let file_path = temp_dir.path().join("test_file.txt"); File::create(&file_path) .unwrap() .write_all(b"Hello") .unwrap(); let node = build_tree(temp_dir.path(), false, false); assert_eq!(node.entry_type, EntryType::Dir); assert!(!node.children.is_empty()); assert_eq!(node.children[0].path, file_path); // Size should be at least 5 bytes assert!(node.children[0].size_bytes >= 5); } }