32 Commits
0.1.1 ... 0.1.3

Author SHA1 Message Date
sreedev
1a573fbf3f release github flow changes 2023-01-23 03:24:12 -05:00
sreedev
b12b662ba1 workflow updates 2023-01-23 03:19:57 -05:00
sreedev
96337b6b48 artifacts upload options added to github workflows 2023-01-23 02:39:37 -05:00
sreedev
02f70aa55f updated release github actions 2023-01-23 02:28:57 -05:00
Sreedev Kodichath
4a18b3fd91 Merge pull request #33 from sreedevk/development
Version 0.1.3
2023-01-23 02:20:59 -05:00
sreedev
876fdc69ac added version changes 2023-01-23 02:20:42 -05:00
sreedev
8acc3f8c8f added github workflow for building binaries with releases 2023-01-23 02:18:23 -05:00
Sreedev Kodichath
38f6c48452 Merge pull request #35 from sreedevk/dependabot/cargo/tokio-1.23.1
build(deps): bump tokio from 1.23.0 to 1.23.1
2023-01-21 19:31:49 -05:00
sreedev
3e1e9c0e2e Merge branch 'main' into development 2023-01-20 21:01:50 -05:00
Sreedev Kodichath
0674ab55bf Update README.md 2023-01-20 21:01:47 -05:00
dependabot[bot]
ccff95dfc5 build(deps): bump tokio from 1.23.0 to 1.23.1
Bumps [tokio](https://github.com/tokio-rs/tokio) from 1.23.0 to 1.23.1.
- [Release notes](https://github.com/tokio-rs/tokio/releases)
- [Commits](https://github.com/tokio-rs/tokio/compare/tokio-1.23.0...tokio-1.23.1)

---
updated-dependencies:
- dependency-name: tokio
  dependency-type: direct:production
...

Signed-off-by: dependabot[bot] <support@github.com>
2023-01-21 01:57:43 +00:00
Sreedev Kodichath
53d28abb62 Merge pull request #32 from sreedevk/performance/glob_improvements
Performance/glob improvements
2023-01-20 19:17:39 -05:00
Sreedev Kodichath
ea6cf94952 Update README.md 2023-01-19 23:37:13 -05:00
sreedev
7b275d40d0 min opts 2 2023-01-19 23:27:40 -05:00
sreedev
30398c3ca9 min opts 2023-01-19 23:25:16 -05:00
sreedev
c0042fc9f7 version 0.1.2 release changes 2023-01-19 09:40:58 -05:00
Sreedev Kodichath
5aea0eb6f4 Merge pull request #28 from sreedevk/development
Version 0.1.2
2023-01-19 09:39:59 -05:00
Sreedev Kodichath
f0ff1ec325 Merge pull request #27 from beeb/sorted_by_size
Sort table results by ascending file size
2023-01-18 10:36:11 -05:00
Sreedev Kodichath
b267fcdedf Merge pull request #29 from beeb/cli_value_hint
feat(cli): add value hint for --dir argument
2023-01-18 10:28:15 -05:00
Valentin Bersier
84b194efca feat(cli): add value hint for --dir argument 2023-01-18 15:40:10 +01:00
Valentin Bersier
d162ca98ef docs: comment 2023-01-18 15:34:58 +01:00
Valentin Bersier
dfb73ceb5c feat: sort interactive results by descending size 2023-01-18 15:33:23 +01:00
Valentin Bersier
72ca7c7a44 Merge branch 'development' into sorted_by_size 2023-01-18 15:24:47 +01:00
Sreedev Kodichath
2f3cfd3162 Merge pull request #26 from sreedevk/feature/min-filesize
Min Size Filter Feature
2023-01-18 02:40:17 -05:00
sreedev
9b1f591c20 adds minsize option to README.md 2023-01-18 02:39:29 -05:00
sreedev
e24603c645 output enhancements 2023-01-18 02:34:08 -05:00
sreedev
0e624b042d added minsize filter 2023-01-18 02:30:48 -05:00
Valentin Bersier
e61d1f1475 perf: unwrap the size once 2023-01-18 08:07:51 +01:00
Valentin Bersier
8c76c6a3c8 refactor: simplify iterator to get size 2023-01-18 08:00:53 +01:00
Valentin Bersier
df3bea3d4f feat: display results in ascending size order 2023-01-18 07:59:51 +01:00
sreedev
7ace0bde63 scanner optimizations 2023-01-18 01:33:51 -05:00
sreedev
70de402eab performance updates in README.md 2023-01-18 00:45:24 -05:00
10 changed files with 172 additions and 146 deletions

41
.github/workflows/release.yml vendored Normal file
View File

@@ -0,0 +1,41 @@
name: Release
env:
PROJECT_NAME: deduplicator
PROJECT_DESC: "Filter, Sort & Delete Duplicate Files Recursively"
PROJECT_AUTH: "sreedevk"
on:
push:
tags:
- "[0-9]+.*"
jobs:
create-release:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v3
- uses: taiki-e/create-gh-release-action@v1.6.2
with:
changelog: CHANGELOG.md
token: ${{ secrets.GITHUB_TOKEN }}
branch: main
upload-assets:
strategy:
matrix:
os:
- ubuntu-latest
- macos-latest
- windows-latest
runs-on: ${{ matrix.os }}
steps:
- uses: actions/checkout@v3
- uses: taiki-e/upload-rust-binary-action@v1
with:
bin: deduplicator
tar: unix
zip: windows
token: ${{ secrets.GITHUB_TOKEN }}
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}

View File

@@ -1,20 +0,0 @@
name: Rust
on:
push:
branches: [ "main" ]
pull_request:
branches: [ "main" ]
env:
CARGO_TERM_COLOR: always
jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v3
- name: Build
run: cargo build --verbose
- name: Run tests
run: cargo test --verbose

29
Cargo.lock generated
View File

@@ -70,6 +70,12 @@ version = "1.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dfb24e866b15a1af2a1b663f10c6b6b8f397a84aadb828f12e5b289ec23a3a3c"
[[package]]
name = "bytesize"
version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6c58ec36aac5066d5ca17df51b3e70279f5670a72102f5752cb7e7c856adfc70"
[[package]]
name = "cc"
version = "1.0.78"
@@ -299,16 +305,16 @@ dependencies = [
[[package]]
name = "deduplicator"
version = "0.1.1"
version = "0.1.3"
dependencies = [
"anyhow",
"bytesize",
"chrono",
"clap",
"colored",
"dashmap",
"fxhash",
"glob",
"humansize",
"indicatif",
"itertools",
"memmap2",
@@ -435,15 +441,6 @@ dependencies = [
"libc",
]
[[package]]
name = "humansize"
version = "2.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4e682e2bd70ecbcce5209f11a992a4ba001fea8e60acf7860ce007629e6d2756"
dependencies = [
"libm",
]
[[package]]
name = "iana-time-zone"
version = "0.1.53"
@@ -540,12 +537,6 @@ version = "0.2.139"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "201de327520df007757c1f0adce6e827fe8562fbc28bfd9c15571c66ca1f5f79"
[[package]]
name = "libm"
version = "0.2.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "348108ab3fba42ec82ff6e9564fc4ca0247bdccdc68dd8af9764bbc79c3c8ffb"
[[package]]
name = "link-cplusplus"
version = "1.0.8"
@@ -941,9 +932,9 @@ dependencies = [
[[package]]
name = "tokio"
version = "1.23.0"
version = "1.23.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "eab6d665857cc6ca78d6e80303a02cea7a7851e85dfbd77cbdc09bd129f1ef46"
checksum = "38a54aca0c15d014013256222ba0ebed095673f89345dd79119d912eb561b7a8"
dependencies = [
"autocfg",
"bytes",

View File

@@ -1,6 +1,6 @@
[package]
name = "deduplicator"
version = "0.1.1"
version = "0.1.3"
edition = "2021"
description = "find,filter,delete Duplicates"
license = "MIT"
@@ -10,18 +10,18 @@ authors = ["Sreedev Kodichath <sreedevpadmakumar@gmail.com>", "Valentin Bersier
[dependencies]
anyhow = "1.0.68"
bytesize = "1.1.0"
chrono = "0.4.23"
clap = { version = "4.0.32", features = ["derive"] }
colored = "2.0.0"
dashmap = { version = "5.4.0", features = ["rayon"] }
fxhash = "0.2.1"
glob = "0.3.0"
humansize = "2.1.2"
indicatif = { version = "0.17.2", features = ["rayon", "tokio"] }
itertools = "0.10.5"
memmap2 = "0.5.8"
prettytable-rs = "0.10.0"
rayon = "1.6.1"
thiserror = "1.0.38"
tokio = { version = "1.23.0", features = ["full"] }
tokio = { version = "1.23.1", features = ["full"] }
unicode-segmentation = "1.10.0"

View File

@@ -14,11 +14,12 @@ NOTE: This project is still being developed. At the moment, as shown in the scre
Usage: deduplicator [OPTIONS]
Options:
-t, --types <TYPES> Filetypes to deduplicate (default = all)
--dir <DIR> Run Deduplicator on dir different from pwd
-i, --interactive Delete files interactively
-h, --help Print help information
-V, --version Print version information
-t, --types <TYPES> Filetypes to deduplicate (default = all)
--dir <DIR> Run Deduplicator on dir different from pwd
-i, --interactive Delete files interactively
-m, --minsize <MINSIZE> Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T). [default = 0]
-h, --help Print help information
-V, --version Print version information
```
<h2 align="center">Installation</h2>
@@ -35,15 +36,13 @@ cargo install deduplicator
<h2 align="center">Performance</h2>
<p align="center">
Deduplicator uses fxhash (a non-cryptographic hashing algorithm) which is extremely fast. As a result, deduplicator is able to process huge amounts of data in a couple of seconds.</p>
Deduplicator uses fxhash (a non-cryptographic hashing algorithm) which is extremely fast. As a result, deduplicator is able to process huge amounts of data in a <del>couple of seconds.</del> few milliseconds.</p>
<p align="center">
While testing, Deduplicator was able to go through 8.6GB of pdf files and detect duplicates in 2.9 seconds
<del>While testing, Deduplicator was able to go through 8.6GB of pdf files and detect duplicates in 2.9 seconds</del>
As of version 0.1.1, on testing locally, deduplicator was able to process and find duplicates in 120GB of files (Videos, PDFs, Images) in ~300ms
</p>
<h2 align="center">Screenshots</h2>
<img src="https://user-images.githubusercontent.com/36154121/211948081-63c12b94-6251-487b-a49f-ac5418169d5a.gif" />
<img src="https://user-images.githubusercontent.com/36154121/211458077-90092aa3-496c-492f-a061-618059890d5f.png" />
<img src="https://user-images.githubusercontent.com/36154121/213618143-e5182e39-731e-4817-87dd-1a6a0f38a449.gif" />

12
src/filters.rs Normal file
View File

@@ -0,0 +1,12 @@
use crate::file_manager::File;
use crate::params::Params;
pub fn is_file_gt_minsize(app_opts: &Params, file: &File) -> bool {
match app_opts.get_minsize() {
Some(msize) => match file.size {
Some(fsize) => fsize >= msize,
None => true,
},
None => true,
}
}

View File

@@ -3,6 +3,7 @@ mod file_manager;
mod output;
mod params;
mod scanner;
mod filters;
use anyhow::Result;
use app::App;

View File

@@ -5,7 +5,6 @@ use chrono::offset::Utc;
use chrono::DateTime;
use colored::Colorize;
use dashmap::DashMap;
use humansize::{format_size, DECIMAL};
use itertools::Itertools;
use prettytable::{format, row, Table};
use std::io::Write;
@@ -30,10 +29,8 @@ fn format_path(path: &str, opts: &Params) -> Result<String> {
Ok(format!("...{:<32}", display_range))
}
fn file_size(path: &String) -> Result<String> {
let mdata = fs::metadata(path)?;
let formatted_size = format!("{:>12}", format_size(mdata.len(), DECIMAL));
Ok(formatted_size)
fn file_size(file: &File) -> Result<String> {
Ok(format!("{:>12}", bytesize::ByteSize::b(file.size.unwrap())))
}
fn modified_time(path: &String) -> Result<String> {
@@ -43,10 +40,6 @@ fn modified_time(path: &String) -> Result<String> {
Ok(modified_time.format("%Y-%m-%d %H:%M:%S").to_string())
}
fn print_meta_info() {
println!("Deduplicator v{}", std::env!("CARGO_PKG_VERSION"));
}
fn scan_group_instruction() -> Result<String> {
println!("\nEnter the indices of the files you want to delete.");
println!("You can enter multiple files using commas to seperate file indices.");
@@ -118,10 +111,20 @@ fn process_group_action(duplicates: &Vec<File>, dup_index: usize, dup_size: usiz
}
pub fn interactive(duplicates: DashMap<String, Vec<File>>, opts: &Params) {
print_meta_info();
if duplicates.is_empty() {
println!(
"\n{}",
"No duplicates found matching your search criteria.".green()
);
return;
}
duplicates
.clone()
.into_iter()
.sorted_unstable_by_key(|(_, f)| {
-(f.first().and_then(|ff| ff.size).unwrap_or_default() as i64)
}) // sort by descending file size in interactive mode
.enumerate()
.for_each(|(gindex, (_, group))| {
let mut itable = Table::new();
@@ -131,7 +134,7 @@ pub fn interactive(duplicates: DashMap<String, Vec<File>>, opts: &Params) {
itable.add_row(row![
index,
format_path(&file.path, opts).unwrap_or_default().blue(),
file_size(&file.path).unwrap_or_default().red(),
file_size(file).unwrap_or_default().red(),
modified_time(&file.path).unwrap_or_default().yellow()
]);
});
@@ -141,22 +144,31 @@ pub fn interactive(duplicates: DashMap<String, Vec<File>>, opts: &Params) {
}
pub fn print(duplicates: DashMap<String, Vec<File>>, opts: &Params) {
print_meta_info();
if duplicates.is_empty() {
println!(
"\n{}",
"No duplicates found matching your search criteria.".green()
);
return;
}
let mut output_table = Table::new();
output_table.set_titles(row!["hash", "duplicates"]);
duplicates.into_iter().for_each(|(hash, group)| {
let mut inner_table = Table::new();
inner_table.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
group.iter().for_each(|file| {
inner_table.add_row(row![
format_path(&file.path, opts).unwrap_or_default().blue(),
file_size(&file.path).unwrap_or_default().red(),
modified_time(&file.path).unwrap_or_default().yellow()
]);
duplicates
.into_iter()
.sorted_unstable_by_key(|(_, f)| f.first().and_then(|ff| ff.size).unwrap_or_default())
.for_each(|(hash, group)| {
let mut inner_table = Table::new();
inner_table.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
group.iter().for_each(|file| {
inner_table.add_row(row![
format_path(&file.path, opts).unwrap_or_default().blue(),
file_size(file).unwrap_or_default().red(),
modified_time(&file.path).unwrap_or_default().yellow()
]);
});
output_table.add_row(row![hash.green(), inner_table]);
});
output_table.add_row(row![hash.green(), inner_table]);
});
output_table.printstd();
}

View File

@@ -1,5 +1,5 @@
use anyhow::{anyhow, Result};
use clap::Parser;
use clap::{Parser, ValueHint};
use std::{fs, path::PathBuf};
#[derive(Parser, Debug)]
@@ -9,19 +9,32 @@ pub struct Params {
#[arg(short, long)]
pub types: Option<String>,
/// Run Deduplicator on dir different from pwd
#[arg(long)]
#[arg(long, value_hint = ValueHint::DirPath)]
pub dir: Option<PathBuf>,
/// Delete files interactively
#[arg(long, short)]
pub interactive: bool,
/// Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T). [default = 0]
#[arg(long, short)]
pub minsize: Option<String>,
}
impl Params {
pub fn get_minsize(&self) -> Option<u64> {
match &self.minsize {
Some(msize) => match msize.parse::<bytesize::ByteSize>() {
Ok(units) => Some(units.0),
Err(_) => None,
},
None => None,
}
}
pub fn get_directory(&self) -> Result<String> {
let dir_pathbuf: PathBuf = self
.dir
.clone()
.unwrap_or(std::env::current_dir()?)
.as_ref()
.unwrap_or(&std::env::current_dir()?)
.as_os_str()
.into();
@@ -34,17 +47,18 @@ impl Params {
Ok(dir)
}
pub fn get_glob_patterns(&self) -> Vec<PathBuf> {
self.types
.clone()
.unwrap_or_else(|| String::from("*"))
.split(',')
.map(|filetype| format!("*.{}", filetype))
.map(|filetype| {
vec![self.get_directory().unwrap(), String::from("**"), filetype]
.iter()
.collect()
})
.collect()
pub fn get_glob_patterns(&self) -> PathBuf {
match self.types.as_ref() {
Some(filetypes) => vec![
self.get_directory().unwrap(),
String::from("**"),
format!("{{{}}}", filetypes),
]
.iter()
.collect::<PathBuf>(),
None => vec![self.get_directory().unwrap().as_str(), "**", "*"]
.iter()
.collect::<PathBuf>(),
}
}
}

View File

@@ -1,3 +1,4 @@
use crate::{file_manager::File, filters, params::Params};
use anyhow::Result;
use dashmap::DashMap;
use fxhash::hash64 as hasher;
@@ -8,8 +9,6 @@ use rayon::prelude::*;
use std::hash::Hasher;
use std::{fs, path::PathBuf};
use crate::{file_manager::File, params::Params};
#[derive(Clone, Copy)]
enum IndexCritera {
Size,
@@ -28,12 +27,7 @@ pub fn duplicates(app_opts: &Params) -> Result<DashMap<String, Vec<File>>> {
.collect::<Vec<File>>();
if sizewize_duplicate_files.len() > 1 {
let size_wise_duplicate_paths = sizewize_duplicate_files
.into_par_iter()
.map(|file| file.path)
.collect::<Vec<String>>();
let hash_index_store = index_files(size_wise_duplicate_paths, IndexCritera::Hash)?;
let hash_index_store = index_files(sizewize_duplicate_files, IndexCritera::Hash)?;
let duplicate_files = hash_index_store
.into_par_iter()
.filter(|(_, files)| files.len() > 1)
@@ -45,72 +39,54 @@ pub fn duplicates(app_opts: &Params) -> Result<DashMap<String, Vec<File>>> {
}
}
fn scan(app_opts: &Params) -> Result<Vec<String>> {
let glob_patterns: Vec<PathBuf> = app_opts.get_glob_patterns();
let files: Vec<String> = glob_patterns
.par_iter()
fn scan(app_opts: &Params) -> Result<Vec<File>> {
let glob_patterns = app_opts.get_glob_patterns().display().to_string();
let glob_iter = glob(&glob_patterns)?;
let files = glob_iter
.filter(Result::is_ok)
.map(|file| file.unwrap())
.filter(|fpath| fpath.is_file())
.collect::<Vec<PathBuf>>()
.into_par_iter()
.progress_with_style(ProgressStyle::with_template(
"{spinner:.green} [scanning files] [{wide_bar:.cyan/blue}] {pos}/{len} files",
"{spinner:.green} [processing scan results] [{wide_bar:.cyan/blue}] {pos}/{len} files",
)?)
.filter_map(|glob_pattern| glob(glob_pattern.as_os_str().to_str()?).ok())
.flat_map(|file_vec| {
file_vec
.filter_map(|x| Some(x.ok()?.as_os_str().to_str()?.to_string()))
.filter(|glob_result| {
fs::metadata(glob_result)
.map(|f| f.is_file())
.unwrap_or(false)
})
.collect::<Vec<String>>()
.map(|fpath| fpath.display().to_string())
.map(|fpath| File {
path: fpath.clone(),
hash: None,
size: Some(fs::metadata(fpath).unwrap().len()),
})
.filter(|file| filters::is_file_gt_minsize(app_opts, file))
.collect();
Ok(files)
}
fn process_file_size_index(fpath: String) -> Result<File> {
Ok(File {
path: fpath.clone(),
size: Some(fs::metadata(fpath)?.len()),
hash: None,
})
}
fn process_file_hash_index(fpath: String) -> Result<File> {
Ok(File {
path: fpath.clone(),
size: None,
hash: Some(hash_file(&fpath).unwrap_or_default()),
})
}
fn process_file_index(
fpath: String,
mut file: File,
store: &DashMap<String, Vec<File>>,
index_criteria: IndexCritera,
) {
match index_criteria {
IndexCritera::Size => {
let processed_file = process_file_size_index(fpath).unwrap();
store
.entry(processed_file.size.unwrap_or_default().to_string())
.and_modify(|fileset| fileset.push(processed_file.clone()))
.or_insert_with(|| vec![processed_file]);
.entry(file.size.unwrap_or_default().to_string())
.and_modify(|fileset| fileset.push(file.clone()))
.or_insert_with(|| vec![file]);
}
IndexCritera::Hash => {
let processed_file = process_file_hash_index(fpath).unwrap();
let indexhash = processed_file.clone().hash.unwrap_or_default();
file.hash = Some(hash_file(&file.path).unwrap_or_default());
store
.entry(indexhash)
.and_modify(|fileset| fileset.push(processed_file.clone()))
.or_insert_with(|| vec![processed_file]);
.entry(file.clone().hash.unwrap())
.and_modify(|fileset| fileset.push(file.clone()))
.or_insert_with(|| vec![file]);
}
}
}
fn index_files(
files: Vec<String>,
files: Vec<File>,
index_criteria: IndexCritera,
) -> Result<DashMap<String, Vec<File>>> {
let store: DashMap<String, Vec<File>> = DashMap::new();
@@ -124,7 +100,7 @@ fn index_files(
Ok(store)
}
pub fn incremental_hashing(filepath: &str) -> Result<String> {
fn incremental_hashing(filepath: &str) -> Result<String> {
let file = fs::File::open(filepath)?;
let fmap = unsafe { Mmap::map(&file)? };
let mut inchasher = fxhash::FxHasher::default();
@@ -135,12 +111,12 @@ pub fn incremental_hashing(filepath: &str) -> Result<String> {
Ok(format!("{}", inchasher.finish()))
}
pub fn standard_hashing(filepath: &str) -> Result<String> {
fn standard_hashing(filepath: &str) -> Result<String> {
let file = fs::read(filepath)?;
Ok(hasher(&*file).to_string())
}
pub fn hash_file(filepath: &str) -> Result<String> {
fn hash_file(filepath: &str) -> Result<String> {
let filemeta = fs::metadata(filepath)?;
// NOTE: USE INCREMENTAL HASHING ONLY FOR FILES > 100MB