Initial GPL-3.0 Release (v1.0.0)

This commit is contained in:
2026-01-22 21:29:34 -05:00
commit 9c8d1d43fd
23 changed files with 4087 additions and 0 deletions
+901
View File
@@ -0,0 +1,901 @@
use byte_unit::{Byte, UnitType};
use clap::{CommandFactory, Parser, ValueEnum};
use clap_complete::{generate, Shell};
use colored::*;
use comfy_table::presets::UTF8_FULL;
use comfy_table::{Attribute, Cell, CellAlignment, Color, ContentArrangement, Table};
use csv::WriterBuilder;
use ignore::WalkBuilder;
use rayon::prelude::*;
use serde::Serialize;
use std::cmp::Ordering;
use std::io::Write;
use std::os::unix::fs::MetadataExt;
use std::path::{Path, PathBuf};
use std::str::FromStr;
#[derive(Parser, Debug, Clone)]
#[command(author, version, about, long_about = None)]
#[command(disable_version_flag = true)]
pub struct Args {
/// Print version
#[arg(short = 'v', long, action = clap::ArgAction::Version)]
pub version: Option<bool>,
/// Directory to analyze
#[arg(default_value = ".")]
pub path: PathBuf,
/// Minimum size filter (e.g. "10MB", "1gb")
#[arg(short = 'm', long)]
pub min_size: Option<String>,
/// Limit the number of results per table
#[arg(short = 'n', long)]
pub number: Option<usize>,
/// Sort by columns (e.g. "type:asc,size:dsc").
/// defaults to size:dsc.
/// Columns: name, type, size (disk usage), apparent (logical), blocks.
#[arg(long)]
pub sort: Option<String>,
/// Show apparent size (logical file length) in addition to disk usage
#[arg(short = 'a', long)]
pub apparent: bool,
/// Depth to traverse (0 = only immediate children)
#[arg(short, long, default_value = "0")]
pub depth: usize,
/// Display full absolute paths
#[arg(short = 'f', long, conflicts_with = "path_relative")]
pub path_full: bool,
/// Display paths relative to current directory (default)
#[arg(long)]
pub path_relative: bool,
/// Number of threads to use (defaults to available logical CPUs)
#[arg(short = 'j', long = "threads")]
pub threads: Option<usize>,
/// Generate shell completions
#[arg(long, value_enum)]
pub completions: Option<Shell>,
/// Respect .gitignore and .ignore files
#[arg(short = 'i', long)]
pub ignore: bool,
/// Output format
#[arg(long, value_enum, default_value_t = OutputFormat::Text)]
pub format: OutputFormat,
/// Save output to file (optional filename, defaults to timestamped name)
#[arg(long, num_args(0..=1), default_missing_value = "_AUTO_")]
pub save: Option<PathBuf>,
/// Precision for size output (decimal places)
#[arg(long, default_value = "2")]
pub precision: usize,
/// Unit system to use
#[arg(long, value_enum, default_value_t = UnitSystem::Binary)]
pub units: UnitSystem,
/// Show comparison between total and non-ignored files
#[arg(short = 'c', long)]
pub compare: bool,
/// Generate man page to the specified output directory
#[arg(long, hide = true)]
pub generate_man_page: Option<std::path::PathBuf>,
}
#[derive(ValueEnum, Clone, Debug, PartialEq, Eq)]
pub enum OutputFormat {
Text,
Csv,
Json,
}
#[derive(ValueEnum, Clone, Debug, PartialEq, Eq)]
pub enum UnitSystem {
Binary,
Decimal,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum SortColumn {
Name,
Type,
Disk,
Apparent,
Blocks,
}
impl FromStr for SortColumn {
type Err = String;
fn from_str(s: &str) -> Result<Self, Self::Err> {
match s.to_lowercase().as_str() {
"name" | "n" => Ok(SortColumn::Name),
"type" | "t" => Ok(SortColumn::Type),
"size" | "s" | "disk" | "d" => Ok(SortColumn::Disk),
"apparent" | "a" => Ok(SortColumn::Apparent),
"blocks" | "b" => Ok(SortColumn::Blocks),
_ => Err(format!("Unknown column: {}", s)),
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum SortDirection {
Asc,
Dsc,
}
impl FromStr for SortDirection {
type Err = String;
fn from_str(s: &str) -> Result<Self, Self::Err> {
match s.to_lowercase().as_str() {
"asc" | "a" => Ok(SortDirection::Asc),
"dsc" | "d" | "desc" => Ok(SortDirection::Dsc),
_ => Err(format!("Unknown direction: {}", s)),
}
}
}
#[derive(Clone, Serialize, Debug)]
pub struct Node {
pub path: PathBuf,
pub size_bytes: u64,
pub blocks: u64,
pub size_bytes_filtered: u64,
pub blocks_filtered: u64,
pub entry_type: EntryType,
pub accessible: bool,
#[serde(skip)]
pub children: Vec<Node>,
}
#[derive(Clone, Serialize, Debug, PartialEq, Eq, PartialOrd, Ord)]
pub enum EntryType {
File,
Dir,
}
impl std::fmt::Display for EntryType {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
EntryType::File => write!(f, "File"),
EntryType::Dir => write!(f, "Dir"),
}
}
}
pub fn run(args: Args, mut writer: &mut dyn Write) {
if let Some(shell) = args.completions {
let mut cmd = Args::command();
let name = cmd.get_name().to_string();
generate(shell, &mut cmd, name, &mut std::io::stdout());
return;
}
if let Some(out_dir) = args.generate_man_page {
let cmd = Args::command();
let man = clap_mangen::Man::new(cmd);
let mut buffer: Vec<u8> = Default::default();
man.render(&mut buffer).expect("Failed to render man page");
std::fs::create_dir_all(&out_dir).expect("Failed to create man page directory");
let file_path = out_dir.join("sized.1");
std::fs::write(&file_path, buffer).expect("Failed to write man page");
println!("Man page generated at {}", file_path.display());
return;
}
if let Some(threads) = args.threads {
rayon::ThreadPoolBuilder::new()
.num_threads(threads)
.build_global()
.ok(); // Ignore error if initialized called multiple times (e.g. in tests)
}
let target_path = args.path.clone();
if !target_path.exists() {
eprintln!("Error: Path '{}' does not exist.", target_path.display());
std::process::exit(1);
}
let min_bytes = if let Some(size_str) = &args.min_size {
match Byte::parse_str(size_str, true) {
Ok(byte) => byte.as_u64(),
Err(e) => {
eprintln!("Error parsing size '{}': {}", size_str, e);
std::process::exit(1);
}
}
} else {
0
};
let sort_criteria = parse_sort_arg(args.sort.as_deref().unwrap_or("size:dsc"));
if args.format == OutputFormat::Csv {
if args.apparent {
writeln!(writer, "Path,Type,Apparent(Bytes),DiskUsage(Bytes),Blocks")
.expect("Failed to write CSV header");
} else {
writeln!(writer, "Path,Type,DiskUsage(Bytes),Blocks")
.expect("Failed to write CSV header");
}
}
let root_node = build_tree(&target_path, args.ignore, args.compare);
if root_node.size_bytes < min_bytes {
if args.format == OutputFormat::Text {
writeln!(writer, "Root directory is smaller than minimum size.").ok();
}
return;
}
process_node_recursive(&mut writer, &root_node, 0, &args, min_bytes, &sort_criteria);
}
pub fn build_tree(path: &Path, ignore: bool, compare: bool) -> Node {
let metadata = path.symlink_metadata();
if let Ok(meta) = metadata {
// Treat symlinks as files (nodes) but do not recurse
if meta.is_file() || meta.is_symlink() {
return Node {
path: path.to_path_buf(),
size_bytes: meta.len(),
blocks: meta.blocks(),
size_bytes_filtered: meta.len(), // Single file is its own filtered size for now
blocks_filtered: meta.blocks(),
entry_type: EntryType::File,
accessible: true,
children: vec![],
};
}
}
// Check if we can read the directory (handle permissions)
if let Err(e) = std::fs::read_dir(path) {
if e.kind() == std::io::ErrorKind::PermissionDenied {
// Return an "empty" directory node marked as inaccessible
let meta = path.symlink_metadata().ok();
let size = meta.as_ref().map(|m| m.len()).unwrap_or(0);
let blocks = meta.as_ref().map(|m| m.blocks()).unwrap_or(0);
return Node {
path: path.to_path_buf(),
size_bytes: size,
blocks,
size_bytes_filtered: size,
blocks_filtered: blocks,
entry_type: EntryType::Dir,
accessible: false,
children: vec![],
};
}
}
// Total walker (always everything if compare is true, else respects 'ignore' arg)
let total_ignore = if compare { false } else { ignore };
let walker_total = WalkBuilder::new(path)
.standard_filters(false)
.hidden(false)
.git_ignore(total_ignore)
.ignore(total_ignore)
.max_depth(Some(1))
.build();
let child_paths_total: Vec<PathBuf> = walker_total
.into_iter()
.filter_map(|e| e.ok())
.filter(|e| e.path() != path)
.map(|e| e.path().to_path_buf())
.collect();
// Filtered set (only if compare is true)
let non_ignored_set: std::collections::HashSet<PathBuf> = if compare {
let walker_filtered = WalkBuilder::new(path)
.standard_filters(false)
.hidden(false)
.git_ignore(true)
.ignore(true)
.max_depth(Some(1))
.build();
walker_filtered
.into_iter()
.filter_map(|e| e.ok())
.filter(|e| e.path() != path)
.map(|e| e.path().to_path_buf())
.collect()
} else {
std::collections::HashSet::new()
};
let children: Vec<Node> = child_paths_total
.par_iter()
.map(|p| build_tree(p, ignore, compare))
.collect();
let mut size_bytes = 0;
let mut blocks = 0;
let mut size_bytes_filtered = 0;
let mut blocks_filtered = 0;
for child in &children {
size_bytes += child.size_bytes;
blocks += child.blocks;
if compare {
if non_ignored_set.contains(&child.path) {
size_bytes_filtered += child.size_bytes_filtered;
blocks_filtered += child.blocks_filtered;
}
} else {
size_bytes_filtered += child.size_bytes_filtered;
blocks_filtered += child.blocks_filtered;
}
}
let (self_size, self_blocks) = path
.symlink_metadata()
.map(|m| (m.len(), m.blocks()))
.unwrap_or((0, 0));
Node {
path: path.to_path_buf(),
size_bytes: size_bytes + self_size,
blocks: blocks + self_blocks,
size_bytes_filtered: size_bytes_filtered + self_size, // self is always part of self
blocks_filtered: blocks_filtered + self_blocks,
entry_type: EntryType::Dir,
accessible: true,
children,
}
}
fn process_node_recursive(
writer: &mut dyn Write,
node: &Node,
current_depth: usize,
args: &Args,
min_bytes: u64,
sort_criteria: &[(SortColumn, SortDirection)],
) {
if node.children.is_empty() && node.entry_type == EntryType::Dir {}
let mut display_children: Vec<&Node> = node
.children
.iter()
.filter(|c| c.size_bytes >= min_bytes)
.collect();
display_children.sort_by(|a, b| compare_nodes(a, b, sort_criteria));
if let Some(n) = args.number {
if n < display_children.len() {
display_children.truncate(n);
}
}
let show_relative = !args.path_full;
let relative_path = if show_relative {
let cwd = std::env::current_dir().unwrap_or_default();
node.path
.strip_prefix(&cwd)
.unwrap_or(&node.path)
.display()
.to_string()
} else {
node.path
.canonicalize()
.unwrap_or(node.path.clone())
.display()
.to_string()
};
print_output(
writer,
&node,
&display_children,
args,
&relative_path,
current_depth,
);
if current_depth < args.depth {
for child in display_children {
if child.entry_type == EntryType::Dir {
process_node_recursive(
writer,
child,
current_depth + 1,
args,
min_bytes,
sort_criteria,
);
}
}
}
}
pub fn compare_nodes(a: &Node, b: &Node, criteria: &[(SortColumn, SortDirection)]) -> Ordering {
for (col, dir) in criteria {
let order = match col {
SortColumn::Name => a.path.file_name().cmp(&b.path.file_name()),
SortColumn::Type => a.entry_type.cmp(&b.entry_type),
SortColumn::Disk => a.blocks.cmp(&b.blocks),
SortColumn::Apparent => a.size_bytes.cmp(&b.size_bytes),
SortColumn::Blocks => a.blocks.cmp(&b.blocks),
};
if order != Ordering::Equal {
return match dir {
SortDirection::Asc => order,
SortDirection::Dsc => order.reverse(),
};
}
}
Ordering::Equal
}
pub fn parse_sort_arg(s: &str) -> Vec<(SortColumn, SortDirection)> {
s.split(',')
.filter_map(|part| {
let parts: Vec<&str> = part.split(':').collect();
if parts.is_empty() {
return None;
}
let col_str = parts[0];
let col = SortColumn::from_str(col_str).ok()?;
let dir = if parts.len() > 1 {
SortDirection::from_str(parts[1]).unwrap_or(SortDirection::Dsc)
} else {
match col {
SortColumn::Name | SortColumn::Type => SortDirection::Asc,
_ => SortDirection::Dsc,
}
};
Some((col, dir))
})
.collect()
}
fn print_output(
writer: &mut dyn Write,
parent_node: &Node,
children: &[&Node],
args: &Args,
relative_path: &str,
depth: usize,
) {
match args.format {
OutputFormat::Text => print_text(writer, parent_node, children, args, relative_path, depth),
OutputFormat::Csv => print_csv(writer, children, args),
OutputFormat::Json => print_json(writer, parent_node, children, args),
}
}
fn print_text(
writer: &mut dyn Write,
parent_node: &Node,
children: &[&Node],
args: &Args,
relative_path: &str,
depth: usize,
) {
if depth > 0 {
writeln!(writer, "{}", "-".repeat(60).dimmed()).ok();
}
writeln!(writer, "\n{}", relative_path.bold().blue()).ok();
let unit_type = match args.units {
UnitSystem::Binary => UnitType::Binary,
UnitSystem::Decimal => UnitType::Decimal,
};
if parent_node.accessible {
let total_usage = Byte::from_u64(parent_node.blocks * 512).get_appropriate_unit(unit_type);
let mut summary = if args.apparent {
let total_apparent =
Byte::from_u64(parent_node.size_bytes).get_appropriate_unit(unit_type);
format!(
"Total Apparent: {} / Disk Usage: {} ({} blocks)",
format!(
"{:.precision$} {}",
total_apparent.get_value(),
total_apparent.get_unit(),
precision = args.precision
)
.bold()
.green(),
format!(
"{:.precision$} {}",
total_usage.get_value(),
total_usage.get_unit(),
precision = args.precision
)
.bold()
.green(),
parent_node.blocks
)
} else {
format!(
"Disk Usage: {} ({} blocks)",
format!(
"{:.precision$} {}",
total_usage.get_value(),
total_usage.get_unit(),
precision = args.precision
)
.bold()
.green(),
parent_node.blocks
)
};
if args.compare {
let total_usage_filtered =
Byte::from_u64(parent_node.blocks_filtered * 512).get_appropriate_unit(unit_type);
if args.apparent {
let total_apparent_filtered =
Byte::from_u64(parent_node.size_bytes_filtered).get_appropriate_unit(unit_type);
summary.push_str(&format!(
"\nFiltered Apparent: {} / Filtered Disk Usage: {} ({} blocks)",
format!(
"{:.precision$} {}",
total_apparent_filtered.get_value(),
total_apparent_filtered.get_unit(),
precision = args.precision
)
.bold()
.cyan(),
format!(
"{:.precision$} {}",
total_usage_filtered.get_value(),
total_usage_filtered.get_unit(),
precision = args.precision
)
.bold()
.cyan(),
parent_node.blocks_filtered
));
} else {
summary.push_str(&format!(
"\nFiltered Disk Usage: {} ({} blocks)",
format!(
"{:.precision$} {}",
total_usage_filtered.get_value(),
total_usage_filtered.get_unit(),
precision = args.precision
)
.bold()
.cyan(),
parent_node.blocks_filtered
));
}
}
writeln!(writer, "{}", summary).ok();
} else {
writeln!(writer, "Total size: {}", "Access Denied".bold().red()).ok();
}
if children.is_empty() {
if parent_node.accessible {
writeln!(writer, "(No children)").ok();
}
return;
}
let mut table = Table::new();
table
.load_preset(UTF8_FULL)
.set_content_arrangement(ContentArrangement::Dynamic);
let mut headers = vec![
Cell::new("Name").add_attribute(Attribute::Bold),
Cell::new("Type").add_attribute(Attribute::Bold),
];
if args.apparent {
headers.push(
Cell::new("Apparent")
.add_attribute(Attribute::Bold)
.set_alignment(CellAlignment::Right),
);
}
headers.push(
Cell::new("Disk Usage")
.add_attribute(Attribute::Bold)
.set_alignment(CellAlignment::Right),
);
if args.compare {
if args.apparent {
headers.push(
Cell::new("App. Filtered")
.add_attribute(Attribute::Bold)
.set_alignment(CellAlignment::Right),
);
}
headers.push(
Cell::new("Disk Filtered")
.add_attribute(Attribute::Bold)
.set_alignment(CellAlignment::Right),
);
}
headers.push(
Cell::new("Blocks")
.add_attribute(Attribute::Bold)
.set_alignment(CellAlignment::Right),
);
table.set_header(headers);
for child in children {
let name = child.path.file_name().unwrap_or_default().to_string_lossy();
let color = if child.entry_type == EntryType::Dir {
Color::Blue
} else {
Color::Cyan
};
if child.accessible {
let mut row = vec![
Cell::new(name).fg(color),
Cell::new(&child.entry_type).fg(Color::Yellow),
];
if args.apparent {
let child_byte = Byte::from_u64(child.size_bytes).get_appropriate_unit(unit_type);
row.push(
Cell::new(format!(
"{:.precision$} {}",
child_byte.get_value(),
child_byte.get_unit(),
precision = args.precision
))
.fg(Color::Green)
.set_alignment(CellAlignment::Right),
);
}
let disk_usage_byte =
Byte::from_u64(child.blocks * 512).get_appropriate_unit(unit_type);
row.push(
Cell::new(format!(
"{:.precision$} {}",
disk_usage_byte.get_value(),
disk_usage_byte.get_unit(),
precision = args.precision
))
.fg(Color::Green)
.set_alignment(CellAlignment::Right),
);
if args.compare {
if args.apparent {
let child_byte_filtered =
Byte::from_u64(child.size_bytes_filtered).get_appropriate_unit(unit_type);
row.push(
Cell::new(format!(
"{:.precision$} {}",
child_byte_filtered.get_value(),
child_byte_filtered.get_unit(),
precision = args.precision
))
.fg(Color::Cyan)
.set_alignment(CellAlignment::Right),
);
}
let disk_usage_byte_filtered =
Byte::from_u64(child.blocks_filtered * 512).get_appropriate_unit(unit_type);
row.push(
Cell::new(format!(
"{:.precision$} {}",
disk_usage_byte_filtered.get_value(),
disk_usage_byte_filtered.get_unit(),
precision = args.precision
))
.fg(Color::Cyan)
.set_alignment(CellAlignment::Right),
);
}
row.push(
Cell::new(child.blocks)
.fg(Color::Green)
.set_alignment(CellAlignment::Right),
);
table.add_row(row);
} else {
let mut row = vec![
Cell::new(name).fg(color),
Cell::new(&child.entry_type).fg(Color::Yellow),
Cell::new("N/A")
.fg(Color::Red)
.set_alignment(CellAlignment::Right),
Cell::new("Access Denied")
.fg(Color::Red)
.set_alignment(CellAlignment::Right),
];
if args.compare {
row.push(
Cell::new("-")
.fg(Color::Red)
.set_alignment(CellAlignment::Right),
);
row.push(
Cell::new("-")
.fg(Color::Red)
.set_alignment(CellAlignment::Right),
);
}
row.push(
Cell::new("-")
.fg(Color::Red)
.set_alignment(CellAlignment::Right),
);
table.add_row(row);
}
}
writeln!(writer, "{}", table).ok();
}
fn print_csv(writer: &mut dyn Write, children: &[&Node], args: &Args) {
let mut wtr = WriterBuilder::new().has_headers(false).from_writer(writer);
for child in children {
let mut record = vec![
child.path.to_string_lossy().to_string(),
child.entry_type.to_string(),
];
if args.apparent {
record.push(child.size_bytes.to_string());
}
record.push((child.blocks * 512).to_string());
record.push(child.blocks.to_string());
wtr.write_record(&record).ok();
}
wtr.flush().ok();
}
fn print_json(writer: &mut dyn Write, parent_node: &Node, children: &[&Node], args: &Args) {
let entries_view: Vec<serde_json::Value> = children
.iter()
.map(|c| {
let mut val = serde_json::json!({
"path": &c.path,
"entry_type": c.entry_type.to_string(),
"accessible": c.accessible,
"disk_usage": c.blocks * 512,
"blocks": c.blocks,
});
if args.apparent {
val.as_object_mut()
.unwrap()
.insert("apparent_size".to_string(), serde_json::json!(c.size_bytes));
}
if args.compare {
val.as_object_mut().unwrap().insert(
"disk_usage_filtered".to_string(),
serde_json::json!(c.blocks_filtered * 512),
);
val.as_object_mut().unwrap().insert(
"blocks_filtered".to_string(),
serde_json::json!(c.blocks_filtered),
);
if args.apparent {
val.as_object_mut().unwrap().insert(
"apparent_size_filtered".to_string(),
serde_json::json!(c.size_bytes_filtered),
);
}
}
val
})
.collect();
let mut report = serde_json::json!({
"path": &parent_node.path,
"total_disk_usage": parent_node.blocks * 512,
"total_blocks": parent_node.blocks,
"entries": entries_view,
"accessible": parent_node.accessible,
});
if args.apparent {
report.as_object_mut().unwrap().insert(
"total_apparent".to_string(),
serde_json::json!(parent_node.size_bytes),
);
}
if args.compare {
report.as_object_mut().unwrap().insert(
"total_disk_usage_filtered".to_string(),
serde_json::json!(parent_node.blocks_filtered * 512),
);
report.as_object_mut().unwrap().insert(
"total_blocks_filtered".to_string(),
serde_json::json!(parent_node.blocks_filtered),
);
if args.apparent {
report.as_object_mut().unwrap().insert(
"total_apparent_filtered".to_string(),
serde_json::json!(parent_node.size_bytes_filtered),
);
}
}
serde_json::to_writer(&mut *writer, &report).ok();
writer.write_all(b"\n").ok();
}
#[cfg(test)]
mod tests {
use super::*;
use std::fs::File;
use tempfile::TempDir;
#[test]
fn test_parse_sort_arg() {
let criteria = parse_sort_arg("size:asc,name");
assert_eq!(criteria.len(), 2);
assert_eq!(criteria[0], (SortColumn::Disk, SortDirection::Asc));
assert_eq!(criteria[1], (SortColumn::Name, SortDirection::Asc)); // Default for Name is Asc
}
#[test]
fn test_sort_defaults() {
let criteria = parse_sort_arg("size");
assert_eq!(criteria[0], (SortColumn::Disk, SortDirection::Dsc)); // Default for Size is Dsc
}
#[test]
fn test_build_tree() {
let temp_dir = TempDir::new().unwrap();
let file_path = temp_dir.path().join("test_file.txt");
File::create(&file_path)
.unwrap()
.write_all(b"Hello")
.unwrap();
let node = build_tree(temp_dir.path(), false, false);
assert_eq!(node.entry_type, EntryType::Dir);
assert!(!node.children.is_empty());
assert_eq!(node.children[0].path, file_path);
// Size should be at least 5 bytes
assert!(node.children[0].size_bytes >= 5);
}
}