mirror of
https://github.com/sreedevk/deduplicator.git
synced 2026-08-26 18:15:33 +00:00
Compare commits
65 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e41e5d35fb | ||
|
|
884f88d749 | ||
|
|
ab15cdfcd7 | ||
|
|
7c5b106784 | ||
|
|
add53a3be1 | ||
|
|
d99d783327 | ||
|
|
ed55aaa005 | ||
|
|
fe79a07b43 | ||
|
|
47b2fc4a87 | ||
|
|
184efdcbbc | ||
|
|
81e13a14a2 | ||
|
|
5daec2f531 | ||
|
|
fcf69ae1cc | ||
|
|
3a03208f47 | ||
|
|
95ecfa0ccc | ||
|
|
ce115fb8dc | ||
|
|
3785a87cee | ||
|
|
d42166a4fd | ||
|
|
4d85f0bbc9 | ||
|
|
8826a87f32 | ||
|
|
4aa536e89b | ||
|
|
9d67a91423 | ||
|
|
4a6aadb78e | ||
|
|
1b06eb8e85 | ||
|
|
f9e86f8522 | ||
|
|
b4bf5d4fb3 | ||
|
|
028b868ea9 | ||
|
|
a56d194ee3 | ||
|
|
080cd791dc | ||
|
|
b16236763c | ||
|
|
3e407f69c8 | ||
|
|
7d386c9420 | ||
|
|
0dc681d4e7 | ||
|
|
442cb4b519 | ||
|
|
8463e72f2d | ||
|
|
05c95bb67a | ||
|
|
062d44acd9 | ||
|
|
ee6655de9b | ||
|
|
ffa5295598 | ||
|
|
6be8596992 | ||
|
|
a40f251e30 | ||
|
|
02d05172da | ||
|
|
dcc709a666 | ||
|
|
e3d48ec505 | ||
|
|
7bd88f1642 | ||
|
|
a65e22269c | ||
|
|
6e3fe4a37d | ||
|
|
eadffb5fea | ||
|
|
5532ded331 | ||
|
|
7119f019d3 | ||
|
|
493cac1762 | ||
|
|
f0dbf05705 | ||
|
|
ef1e9a1fce | ||
|
|
d06ca7897d | ||
|
|
b49940998b | ||
|
|
ab44d1ed04 | ||
|
|
6b06798e8e | ||
|
|
4d27da99f3 | ||
|
|
91dce29bb4 | ||
|
|
f9b6d57968 | ||
|
|
5c86a080c4 | ||
|
|
9f4d9139b6 | ||
|
|
153109ef57 | ||
|
|
fa12f85b6a | ||
|
|
7d66aeef6e |
2
.cargo/config.toml
Normal file
2
.cargo/config.toml
Normal file
@@ -0,0 +1,2 @@
|
||||
[build]
|
||||
rustflags = ["-C", "target-feature=+aes,+sse2"]
|
||||
152
.github/workflows/release.yml
vendored
152
.github/workflows/release.yml
vendored
@@ -1,31 +1,137 @@
|
||||
# CI that:
|
||||
#
|
||||
# * checks for a Git Tag that looks like a release
|
||||
# * creates a Github Release™ and fills in its text
|
||||
# * builds artifacts with cargo-dist (executable-zips, installers)
|
||||
# * uploads those artifacts to the Github Release™
|
||||
#
|
||||
# Note that the Github Release™ will be created before the artifacts,
|
||||
# so there will be a few minutes where the release has no artifacts
|
||||
# and then they will slowly trickle in, possibly failing. To make
|
||||
# this more pleasant we mark the release as a "draft" until all
|
||||
# artifacts have been successfully uploaded. This allows you to
|
||||
# choose what to do with partial successes and avoids spamming
|
||||
# anyone with notifications before the release is actually ready.
|
||||
name: Release
|
||||
|
||||
env:
|
||||
PROJECT_NAME: deduplicator
|
||||
PROJECT_DESC: "Filter, Sort & Delete Duplicate Files Recursively"
|
||||
PROJECT_AUTH: "sreedevk"
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
# This task will run whenever you push a git tag that looks like a version
|
||||
# like "v1", "v1.2.0", "v0.1.0-prerelease01", "my-app-v1.0.0", etc.
|
||||
# The version will be roughly parsed as ({PACKAGE_NAME}-)?v{VERSION}, where
|
||||
# PACKAGE_NAME must be the name of a Cargo package in your workspace, and VERSION
|
||||
# must be a Cargo-style SemVer Version.
|
||||
#
|
||||
# If PACKAGE_NAME is specified, then we will create a Github Release™ for that
|
||||
# package (erroring out if it doesn't have the given version or isn't cargo-dist-able).
|
||||
#
|
||||
# If PACKAGE_NAME isn't specified, then we will create a Github Release™ for all
|
||||
# (cargo-dist-able) packages in the workspace with that version (this is mode is
|
||||
# intended for workspaces with only one dist-able package, or with all dist-able
|
||||
# packages versioned/released in lockstep).
|
||||
#
|
||||
# If you push multiple tags at once, separate instances of this workflow will
|
||||
# spin up, creating an independent Github Release™ for each one.
|
||||
#
|
||||
# If there's a prerelease-style suffix to the version then the Github Release™
|
||||
# will be marked as a prerelease.
|
||||
on:
|
||||
release:
|
||||
types:
|
||||
- created
|
||||
push:
|
||||
tags:
|
||||
- '*-?v[0-9]+*'
|
||||
|
||||
jobs:
|
||||
upload-assets:
|
||||
strategy:
|
||||
matrix:
|
||||
os:
|
||||
- ubuntu-latest
|
||||
- macos-latest
|
||||
- windows-latest
|
||||
runs-on: ${{ matrix.os }}
|
||||
# Create the Github Release™ so the packages have something to be uploaded to
|
||||
create-release:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
has-releases: ${{ steps.create-release.outputs.has-releases }}
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: taiki-e/upload-rust-binary-action@v1
|
||||
with:
|
||||
bin: deduplicator
|
||||
tar: unix
|
||||
zip: windows
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Install Rust
|
||||
run: rustup update 1.71.0 --no-self-update && rustup default 1.71.0
|
||||
- name: Install cargo-dist
|
||||
run: curl --proto '=https' --tlsv1.2 -LsSf https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.sh | sh
|
||||
- id: create-release
|
||||
run: |
|
||||
cargo dist plan --tag=${{ github.ref_name }} --output-format=json > dist-manifest.json
|
||||
echo "dist plan ran successfully"
|
||||
cat dist-manifest.json
|
||||
|
||||
# Create the Github Release™ based on what cargo-dist thinks it should be
|
||||
ANNOUNCEMENT_TITLE=$(jq --raw-output ".announcement_title" dist-manifest.json)
|
||||
IS_PRERELEASE=$(jq --raw-output ".announcement_is_prerelease" dist-manifest.json)
|
||||
jq --raw-output ".announcement_github_body" dist-manifest.json > new_dist_announcement.md
|
||||
gh release create ${{ github.ref_name }} --draft --prerelease="$IS_PRERELEASE" --title="$ANNOUNCEMENT_TITLE" --notes-file=new_dist_announcement.md
|
||||
echo "created announcement!"
|
||||
|
||||
# Upload the manifest to the Github Release™
|
||||
gh release upload ${{ github.ref_name }} dist-manifest.json
|
||||
echo "uploaded manifest!"
|
||||
|
||||
# Disable all the upload-artifacts tasks if we have no actual releases
|
||||
HAS_RELEASES=$(jq --raw-output ".releases != null" dist-manifest.json)
|
||||
echo "has-releases=$HAS_RELEASES" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# Build and packages all the things
|
||||
upload-artifacts:
|
||||
# Let the initial task tell us to not run (currently very blunt)
|
||||
needs: create-release
|
||||
if: ${{ needs.create-release.outputs.has-releases == 'true' }}
|
||||
strategy:
|
||||
matrix:
|
||||
# For these target platforms
|
||||
include:
|
||||
- os: macos-11
|
||||
dist-args: --artifacts=local --target=aarch64-apple-darwin --target=x86_64-apple-darwin
|
||||
install-dist: curl --proto '=https' --tlsv1.2 -LsSf https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.sh | sh
|
||||
- os: ubuntu-20.04
|
||||
dist-args: --artifacts=local --target=x86_64-unknown-linux-gnu
|
||||
install-dist: curl --proto '=https' --tlsv1.2 -LsSf https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.sh | sh
|
||||
- os: windows-2019
|
||||
dist-args: --artifacts=local --target=x86_64-pc-windows-msvc
|
||||
install-dist: irm https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.ps1 | iex
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- name: Install Rust
|
||||
run: rustup update 1.71.0 --no-self-update && rustup default 1.71.0
|
||||
- name: Install cargo-dist
|
||||
run: ${{ matrix.install-dist }}
|
||||
- name: Run cargo-dist
|
||||
# This logic is a bit janky because it's trying to be a polyglot between
|
||||
# powershell and bash since this will run on windows, macos, and linux!
|
||||
# The two platforms don't agree on how to talk about env vars but they
|
||||
# do agree on 'cat' and '$()' so we use that to marshal values between commands.
|
||||
run: |
|
||||
# Actually do builds and make zips and whatnot
|
||||
cargo dist build --tag=${{ github.ref_name }} --output-format=json ${{ matrix.dist-args }} > dist-manifest.json
|
||||
echo "dist ran successfully"
|
||||
cat dist-manifest.json
|
||||
|
||||
# Parse out what we just built and upload it to the Github Release™
|
||||
jq --raw-output ".artifacts[]?.path | select( . != null )" dist-manifest.json > uploads.txt
|
||||
echo "uploading..."
|
||||
cat uploads.txt
|
||||
gh release upload ${{ github.ref_name }} $(cat uploads.txt)
|
||||
echo "uploaded!"
|
||||
|
||||
# Mark the Github Release™ as a non-draft now that everything has succeeded!
|
||||
publish-release:
|
||||
# Only run after all the other tasks, but it's ok if upload-artifacts was skipped
|
||||
needs: [create-release, upload-artifacts]
|
||||
if: ${{ always() && needs.create-release.result == 'success' && (needs.upload-artifacts.result == 'skipped' || needs.upload-artifacts.result == 'success') }}
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- name: mark release as non-draft
|
||||
run: |
|
||||
gh release edit ${{ github.ref_name }} --draft=false
|
||||
|
||||
5
.gitignore
vendored
5
.gitignore
vendored
@@ -1,3 +1,4 @@
|
||||
/target
|
||||
/test_data
|
||||
.envrc
|
||||
/Cargo.lock
|
||||
/result-bin
|
||||
/.bacon-locations
|
||||
|
||||
992
Cargo.lock
generated
992
Cargo.lock
generated
File diff suppressed because it is too large
Load Diff
53
Cargo.toml
53
Cargo.toml
@@ -1,31 +1,58 @@
|
||||
[package]
|
||||
name = "deduplicator"
|
||||
version = "0.1.4"
|
||||
version = "0.3.0"
|
||||
edition = "2021"
|
||||
description = "find,filter,delete Duplicates"
|
||||
description = "find,filter and delete duplicate files"
|
||||
repository = "https://github.com/sreedevk/deduplicator"
|
||||
license = "MIT"
|
||||
authors = [
|
||||
"Sreedev Kodichath <sreedevpadmakumar@gmail.com>",
|
||||
"Valentin Bersier <vbersier@gmail.com>",
|
||||
"Dhruva Sagar <dhruva.sagar@gmail.com>",
|
||||
"Sreedev Kodichath <sreedevpadmakumar@gmail.com>",
|
||||
"Valentin Bersier <vbersier@gmail.com>",
|
||||
"Dhruva Sagar <dhruva.sagar@gmail.com>",
|
||||
]
|
||||
|
||||
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
|
||||
[[bin]]
|
||||
name = "deduplicator"
|
||||
path = "src/main.rs"
|
||||
|
||||
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
|
||||
[dependencies]
|
||||
anyhow = "1.0.68"
|
||||
bytesize = "1.1.0"
|
||||
chrono = "0.4.23"
|
||||
clap = { version = "4.0.32", features = ["derive"] }
|
||||
colored = "2.0.0"
|
||||
dashmap = { version = "5.4.0", features = ["rayon"] }
|
||||
fxhash = "0.2.1"
|
||||
globwalk = "0.8.1"
|
||||
indicatif = { version = "0.17.2", features = ["rayon", "tokio"] }
|
||||
itertools = "0.10.5"
|
||||
gxhash = { version = "3.4.1", default-features = false }
|
||||
indicatif = { version = "0.17.2", features = ["rayon"] }
|
||||
memmap2 = "0.5.8"
|
||||
pathdiff = "0.2.1"
|
||||
prettytable-rs = "0.10.0"
|
||||
rand = "0.9.1"
|
||||
rayon = "1.6.1"
|
||||
thiserror = "1.0.38"
|
||||
tokio = { version = "1.23.1", features = ["full"] }
|
||||
unicode-segmentation = "1.10.0"
|
||||
threadpool = "1.8.1"
|
||||
|
||||
[profile.release]
|
||||
strip = true
|
||||
opt-level = 3
|
||||
lto = "thin"
|
||||
debug = false
|
||||
codegen-units = 1
|
||||
|
||||
# generated by 'cargo dist init'
|
||||
[profile.dist]
|
||||
inherits = "release"
|
||||
|
||||
[workspace.metadata.dist]
|
||||
rust-toolchain-version = "1.87.0"
|
||||
ci = ["github"]
|
||||
targets = [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"x86_64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc",
|
||||
"aarch64-apple-darwin",
|
||||
]
|
||||
cargo-dist-version = "0.0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.20.0"
|
||||
|
||||
204
README.md
204
README.md
@@ -7,94 +7,156 @@
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
Usage: deduplicator [OPTIONS]
|
||||
find,filter and delete duplicate files
|
||||
|
||||
Usage: deduplicator [OPTIONS] [scan_dir_path]
|
||||
|
||||
Arguments:
|
||||
[scan_dir_path] Run Deduplicator on dir different from pwd (e.g., ~/Pictures )
|
||||
|
||||
Options:
|
||||
-t, --types <TYPES> Filetypes to deduplicate (default = all)
|
||||
--dir <DIR> Run Deduplicator on dir different from pwd
|
||||
-i, --interactive Delete files interactively
|
||||
-m, --minsize <MINSIZE> Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T). [default = 0]
|
||||
-h, --help Print help information
|
||||
-V, --version Print version information
|
||||
-t, --types <TYPES> Filetypes to deduplicate [default = all]
|
||||
-i, --interactive Delete files interactively
|
||||
-m, --min-size <MIN_SIZE> Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T) [default: 1b]
|
||||
-D, --max-depth <MAX_DEPTH> Max Depth to scan while looking for duplicates
|
||||
-d, --min-depth <MIN_DEPTH> Min Depth to scan while looking for duplicates
|
||||
-f, --follow-links Follow links while scanning directories
|
||||
-s, --strict Guarantees that two files are duplicate (performs a full hash)
|
||||
-p, --progress Show Progress spinners & metrics
|
||||
-h, --help Print help
|
||||
-V, --version Print version
|
||||
```
|
||||
### Examples
|
||||
|
||||
```bash
|
||||
# Scan for duplicates recursively from the current dir, only look for png, jpg & pdf file types & interactively delete files
|
||||
deduplicator -t pdf,jpg,png -i
|
||||
|
||||
# Scan for duplicates recursively from the ~/Pictures dir, only look for png, jpeg, jpg & pdf file types & interactively delete files
|
||||
deduplicator ~/Pictures/ -t png,jpeg,jpg,pdf -i
|
||||
|
||||
# Scan for duplicates in the ~/Pictures without recursing into subdirectories
|
||||
deduplicator ~/Pictures --max-depth 0
|
||||
|
||||
# look for duplicates in the ~/.config directory while also recursing into symbolic link paths
|
||||
deduplicator ~/.config --follow-links
|
||||
|
||||
# scan for duplicates that are greater than 100mb in the ~/Media directory
|
||||
deduplicator ~/Media --min-size 100mb
|
||||
```
|
||||
|
||||
## Demo
|
||||

|
||||
|
||||
|
||||
|
||||
## Installation
|
||||
Currently, you can only install deduplicator using cargo package manager.
|
||||
|
||||
### Cargo Install
|
||||
|
||||
#### Stable
|
||||
### Cargo
|
||||
> GxHash relies on aes hardware acceleration, so please set `RUSTFLAGS` to `"-C target-feature=+aes"` or `"-C target-cpu=native"` before
|
||||
> installing.
|
||||
|
||||
#### install from crates.io
|
||||
```bash
|
||||
$ cargo install deduplicator
|
||||
$ RUSTFLAGS="-C target-cpu=native" cargo install deduplicator
|
||||
```
|
||||
|
||||
#### Nightly
|
||||
|
||||
if you'd like to install with nightly features, you can use
|
||||
|
||||
#### install from git
|
||||
```bash
|
||||
$ cargo install --git https://github.com/sreedevk/deduplicator
|
||||
$ RUSTFLAGS="-C target-cpu=native" cargo install deduplicator --git https://github.com/sreedevk/deduplicator
|
||||
```
|
||||
Please note that if you use a version manager to install rust (like asdf), you need to reshim (`asdf reshim rust`).
|
||||
|
||||
### Linux (Pre-built Binary)
|
||||
|
||||
you can download the pre-built binary from the [Releases](https://github.com/sreedevk/deduplicator/releases) page.
|
||||
download the `deduplicator-x86_64-unknown-linux-gnu.tar.gz` for linux. Once you have the tarball file with the executable,
|
||||
you can follow these steps to install:
|
||||
|
||||
```bash
|
||||
$ tar -zxvf deduplicator-x86_64-unknown-linux-gnu.tar.gz
|
||||
$ sudo mv deduplicator /usr/bin/
|
||||
```
|
||||
|
||||
### Mac OS (Pre-built Binary)
|
||||
|
||||
you can download the pre-build binary from the [Releases](https://github.com/sreedevk/deduplicator/releases) page.
|
||||
download the `deduplicator-x86_64-apple-darwin.tar.gz` tarball for mac os. Once you have the tarball file with the executable, you can follow these steps to install:
|
||||
|
||||
```bash
|
||||
$ tar -zxvf deduplicator-x86_64-unknown-linux-gnu.tar.gz
|
||||
$ sudo mv deduplicator /usr/bin/
|
||||
```
|
||||
|
||||
### Windows (Pre-built Binary)
|
||||
|
||||
you can download the pre-build binary from the [Releases](https://github.com/sreedevk/deduplicator/releases) page.
|
||||
download the `deduplicator-x86_64-pc-windows-msvc.zip` zip file for windows. unzip the `zip` file & move the `deduplicator.exe` to a location in the PATH system environment variable.
|
||||
|
||||
Note: If you Run into an msvc error, please install MSCV from [here](https://learn.microsoft.com/en-us/cpp/windows/latest-supported-vc-redist?view=msvc-170)
|
||||
|
||||
## Performance
|
||||
Deduplicator uses size comparison and [GxHash](https://docs.rs/gxhash/latest/gxhash/) to quickly check a large number of files to find duplicates. its also heavily parallelized. The default behavior of deduplicator is to only hash the first page (4K) of the file. This is to ensure that performance is the default priority. You can modify this behavior by using the `--strict` flag which will hash the whole file and ensure that 2 files are indeed duplicates. I'll add benchmarks in future versions.
|
||||
|
||||
Deduplicator uses size comparison and fxhash (a non non-cryptographic hashing algo) to quickly scan through large number of files to find duplicates. its also highly parallel (uses rayon and dashmap). I was able to scan through 120GB of files (Videos, PDFs, Images) in ~300ms. checkout the benchmarks
|
||||
|
||||
## benchmarks
|
||||
|
||||
| Command | Dirsize | Mean [ms] | Min [ms] | Max [ms] | Relative |
|
||||
|:---|:---|---:|---:|---:|---:|
|
||||
| `deduplicator --dir ~/Data/tmp` | (~120G) | 27.5 ± 1.0 | 26.0 | 32.1 | 1.70 ± 0.09 |
|
||||
| `deduplicator --dir ~/Data/books` | (~8.6G) | 21.8 ± 0.7 | 20.5 | 24.4 | 1.35 ± 0.07 |
|
||||
| `deduplicator --dir ~/Data/books --minsize 10M` | (~8.6G) | 16.1 ± 0.6 | 14.9 | 18.8 | 1.00 |
|
||||
| `deduplicator --dir ~/Data/ --types pdf,jpg,png,jpeg` | (~290G) | 1857.4 ± 24.5 | 1817.0 | 1895.5 | 115.07 ± 4.64 |
|
||||
|
||||
* The last entry is lower because of the number of files deduplicator had to go through (~660895 Files). The average size of the files rarely affect the performance of deduplicator.
|
||||
|
||||
These benchmarks were run using [hyperfine](https://github.com/sharkdp/hyperfine). Here are the specs of the machine used to benchmark deduplicator:
|
||||
### Benchmarks
|
||||
I've used hyperfine to run deduplicator on files generated by the rake file at `rakelib/benchmark.rake`. The Benchmarking accuracy can further be improved by isolating runs inside restricted docker containers. I'll include that in the future. For now, here's the hyperfine output on my i7-12800H laptop with 32G of RAM.
|
||||
|
||||
#### Fewer Large Files
|
||||
```
|
||||
OS: Arch Linux x86_64
|
||||
Host: Precision 5540
|
||||
Kernel: 5.15.89-1-lts
|
||||
Uptime: 4 hours, 44 mins
|
||||
Shell: zsh 5.9
|
||||
Terminal: kitty
|
||||
CPU: Intel i9-9880H (16) @ 4.800GHz
|
||||
GPU: NVIDIA Quadro T2000 Mobile / Max-Q
|
||||
GPU: Intel CoffeeLake-H GT2 [UHD Graphics 630]
|
||||
Memory: 31731MiB (~32GiB)
|
||||
# hyperfine -N --warmup 80 './target/release/deduplicator bench_artifacts'
|
||||
Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
Time (mean ± σ): 2.5 ms ± 0.4 ms [User: 2.2 ms, System: 4.8 ms]
|
||||
Range (min … max): 1.9 ms … 6.8 ms 1322 runs
|
||||
|
||||
# dust 'bench_artifacts'
|
||||
37M ┌── file_1_fwds.bin │██ │ 1%
|
||||
201M ├── file_0_fwds.bin │██████████ │ 8%
|
||||
390M ├── file_0_fwdcbss.bin│████████████████████ │ 15%
|
||||
390M ├── file_0_fwscas.bin │████████████████████ │ 15%
|
||||
390M ├── file_0_fwss.bin │████████████████████ │ 15%
|
||||
390M ├── file_1_fwdcbss.bin│████████████████████ │ 15%
|
||||
390M ├── file_1_fwscas.bin │████████████████████ │ 15%
|
||||
390M ├── file_1_fwss.bin │████████████████████ │ 15%
|
||||
2.5G ┌─┴ bench_artifacts │███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████ │ 100%
|
||||
```
|
||||
|
||||
## Screenshots
|
||||
#### Many Small Files
|
||||
```
|
||||
# hyperfine --warmup 20 './target/release/deduplicator bench_artifacts'
|
||||
Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
Time (mean ± σ): 22.3 ms ± 1.7 ms [User: 35.8 ms, System: 46.3 ms]
|
||||
Range (min … max): 18.9 ms … 27.2 ms 112 runs
|
||||
|
||||

|
||||
# dust 'bench_artifacts'
|
||||
3.9M ┌── file_992_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_993_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_994_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_995_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_996_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_997_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_998_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_999_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwss.bin │█ │ 0%
|
||||
11G ┌─┴ bench_artifacts │█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████ │ 100%
|
||||
```
|
||||
|
||||
## proposed
|
||||
- [ ] parallelization
|
||||
- [ ] (scanning + processing sw + processing hw) & formatting & printing
|
||||
- [ ] scanning + processing sw + processing hw + formatting + printing
|
||||
- [ ] max file path size should use the last set of duplicates
|
||||
- [ ] add more unit tests
|
||||
- [ ] test against different filesystems
|
||||
- [ ] test against different file name encodings
|
||||
- [ ] restore json output (was removed in 0.3)
|
||||
- [ ] fix memory leak on very large filesystems
|
||||
- [ ] maybe use a bloom filter
|
||||
- [ ] reduce FileInfo size
|
||||
- [ ] output in a tree format
|
||||
- [ ] tui
|
||||
- [ ] change the default hashing method to include the first & last page of a file (8K)
|
||||
|
||||
## v0.3
|
||||
- [x] parallelization
|
||||
- [x] (scanning) + (processing sw & processing hw & formatting & printing)
|
||||
- [x] reduce cloning values on the heap
|
||||
- [x] add a partial hashing mode (--strict)
|
||||
- [x] add unit tests
|
||||
- [x] add silent mode
|
||||
- [x] update documentation
|
||||
- [x] remove color output
|
||||
- [x] progress bar improvements
|
||||
- [x] use progress bar groups
|
||||
- [x] remove broken json rendering
|
||||
- [x] add benchmarks
|
||||
|
||||
93
rakelib/benchmark.rake
Normal file
93
rakelib/benchmark.rake
Normal file
@@ -0,0 +1,93 @@
|
||||
require 'tempfile'
|
||||
require 'fileutils'
|
||||
require 'securerandom'
|
||||
|
||||
namespace :benchmark do
|
||||
file 'target/release/deduplicator' do
|
||||
sh "cargo build --release"
|
||||
end
|
||||
|
||||
task :few_large_files => 'target/release/deduplicator' do
|
||||
# Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
# Time (mean ± σ): 2.5 ms ± 0.6 ms [User: 2.4 ms, System: 4.8 ms]
|
||||
# Range (min … max): 1.8 ms … 9.7 ms 1474 runs
|
||||
|
||||
root = "bench_artifacts"
|
||||
Dir.mkdir(root)
|
||||
|
||||
# files with same size
|
||||
2.times.map do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 100_000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different sizes
|
||||
2.times.map do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwds.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * (rand * 100_000).ceil))
|
||||
end
|
||||
end
|
||||
|
||||
# files with same content & size
|
||||
2.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwscas.bin"), 'wb') do |f|
|
||||
f.write("\0" * (4096 * 100_000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different content but same size
|
||||
2.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwdcbss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 100_000))
|
||||
end
|
||||
end
|
||||
|
||||
sh("hyperfine -N --warmup 80 './target/release/deduplicator #{root}'")
|
||||
sh("dust '#{root}'")
|
||||
|
||||
FileUtils.rm_rf(root)
|
||||
end
|
||||
|
||||
task :many_small_files => 'target/release/deduplicator' do
|
||||
# Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
# Time (mean ± σ): 10.6 ms ± 1.0 ms [User: 20.0 ms, System: 22.5 ms]
|
||||
# Range (min … max): 8.4 ms … 14.2 ms 235 runs
|
||||
|
||||
root = "bench_artifacts"
|
||||
Dir.mkdir(root)
|
||||
|
||||
# files with same size
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 1000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different sizes
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwds.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * (rand * 100).ceil))
|
||||
end
|
||||
end
|
||||
|
||||
# files with same content & size
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwscas.bin"), 'wb') do |f|
|
||||
f.write("\0" * (4096 * 1000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different content but same size
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwdcbss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 1000))
|
||||
end
|
||||
end
|
||||
|
||||
sh("hyperfine --warmup 20 './target/release/deduplicator #{root}'")
|
||||
sh("dust '#{root}'")
|
||||
|
||||
FileUtils.rm_rf(root)
|
||||
end
|
||||
end
|
||||
18
src/app.rs
18
src/app.rs
@@ -1,18 +0,0 @@
|
||||
use crate::output;
|
||||
use crate::params::Params;
|
||||
use crate::scanner;
|
||||
use anyhow::Result;
|
||||
|
||||
pub struct App;
|
||||
|
||||
impl App {
|
||||
pub fn init(app_args: &Params) -> Result<()> {
|
||||
let duplicates = scanner::duplicates(app_args)?;
|
||||
match app_args.interactive {
|
||||
true => output::interactive(duplicates, app_args),
|
||||
false => output::print(duplicates, app_args),
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -1,20 +0,0 @@
|
||||
use anyhow::Result;
|
||||
use colored::Colorize;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct File {
|
||||
pub path: String,
|
||||
pub size: Option<u64>,
|
||||
pub hash: Option<String>,
|
||||
}
|
||||
|
||||
pub fn delete_files(files: Vec<File>) -> Result<()> {
|
||||
files.into_iter().for_each(|file| {
|
||||
match std::fs::remove_file(file.path.clone()) {
|
||||
Ok(_) => println!("{}: {}", "DELETED".green(), file.path),
|
||||
Err(_) => println!("{}: {}", "FAILED".red(), file.path)
|
||||
}
|
||||
});
|
||||
|
||||
Ok(())
|
||||
}
|
||||
45
src/fileinfo.rs
Normal file
45
src/fileinfo.rs
Normal file
@@ -0,0 +1,45 @@
|
||||
use anyhow::Result;
|
||||
use gxhash::gxhash128;
|
||||
use memmap2::Mmap;
|
||||
use std::{
|
||||
fs,
|
||||
io::Read,
|
||||
path::{Path, PathBuf},
|
||||
time::SystemTime,
|
||||
};
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct FileInfo {
|
||||
pub path: Box<Path>,
|
||||
pub size: u64,
|
||||
pub modified: SystemTime,
|
||||
}
|
||||
|
||||
impl FileInfo {
|
||||
pub fn hash(&self, seed: i64) -> Result<u128> {
|
||||
let file = fs::File::open(&self.path)?;
|
||||
let mapper = unsafe { Mmap::map(&file)? };
|
||||
let final_hash = mapper.chunks(4096).fold(0u128, |acc, chunk: &[u8]| {
|
||||
acc.wrapping_add(gxhash128(chunk, seed))
|
||||
});
|
||||
|
||||
Ok(final_hash)
|
||||
}
|
||||
|
||||
pub fn initial_page_hash(&self, seed: i64) -> Result<u128> {
|
||||
let mut file = fs::File::open(&self.path)?;
|
||||
let mut buffer = [0; 4096];
|
||||
let bytes_read = file.read(&mut buffer)?;
|
||||
|
||||
Ok(gxhash128(&buffer[..bytes_read], seed))
|
||||
}
|
||||
|
||||
pub fn new(path: PathBuf) -> Result<Self> {
|
||||
let filemeta = std::fs::metadata(&path)?;
|
||||
Ok(Self {
|
||||
path: path.into_boxed_path(),
|
||||
size: filemeta.len(),
|
||||
modified: filemeta.modified()?,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -1,12 +0,0 @@
|
||||
use crate::file_manager::File;
|
||||
use crate::params::Params;
|
||||
|
||||
pub fn is_file_gt_minsize(app_opts: &Params, file: &File) -> bool {
|
||||
match app_opts.get_minsize() {
|
||||
Some(msize) => match file.size {
|
||||
Some(fsize) => fsize >= msize,
|
||||
None => true,
|
||||
},
|
||||
None => true,
|
||||
}
|
||||
}
|
||||
101
src/formatter.rs
Normal file
101
src/formatter.rs
Normal file
@@ -0,0 +1,101 @@
|
||||
use crate::{fileinfo::FileInfo, params::Params};
|
||||
use anyhow::Result;
|
||||
use chrono::{DateTime, Utc};
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{ParallelProgressIterator, ProgressBar, ProgressFinish, ProgressStyle};
|
||||
use pathdiff::diff_paths;
|
||||
use prettytable::{format, row, Row, Table};
|
||||
use rayon::prelude::*;
|
||||
use std::{borrow::Cow, path::PathBuf, sync::Arc, time::Duration};
|
||||
|
||||
pub struct Formatter;
|
||||
impl Formatter {
|
||||
pub fn human_path(
|
||||
file: &FileInfo,
|
||||
app_args: &Params,
|
||||
min_path_length: usize,
|
||||
) -> Result<String> {
|
||||
let base_directory: PathBuf = app_args.get_directory()?;
|
||||
let relative_path = diff_paths(&file.path, base_directory).unwrap_or_default();
|
||||
|
||||
let formatted_path = format!(
|
||||
"{:<0width$}",
|
||||
relative_path.to_str().unwrap_or_default().to_string(),
|
||||
width = min_path_length
|
||||
);
|
||||
|
||||
Ok(formatted_path)
|
||||
}
|
||||
|
||||
pub fn human_filesize(file: &FileInfo) -> Result<String> {
|
||||
Ok(format!("{:>12}", bytesize::ByteSize::b(file.size)))
|
||||
}
|
||||
|
||||
pub fn human_mtime(file: &FileInfo) -> Result<String> {
|
||||
let modified_time: DateTime<Utc> = file.modified.into();
|
||||
Ok(modified_time.format("%Y-%m-%d %H:%M:%S").to_string())
|
||||
}
|
||||
|
||||
pub fn gen_sub_tbl(items: Vec<FileInfo>, app_args: &Params, max_path_len: u64) -> Table {
|
||||
let mut inner_table = Table::new();
|
||||
inner_table.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
|
||||
items.iter().for_each(|file| {
|
||||
inner_table.add_row(row![
|
||||
Self::human_path(file, app_args, max_path_len as usize).unwrap_or_default(),
|
||||
Self::human_filesize(file).unwrap_or_default(),
|
||||
Self::human_mtime(file).unwrap_or_default()
|
||||
]);
|
||||
});
|
||||
inner_table
|
||||
}
|
||||
|
||||
pub fn generate_table(
|
||||
raw: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
mpath_len: u64,
|
||||
args: &Params,
|
||||
) -> Result<Table> {
|
||||
let progress_bar = match args.progress {
|
||||
true => ProgressBar::new_spinner(),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("generating output");
|
||||
|
||||
let rows = raw
|
||||
.par_iter_mut()
|
||||
.progress_with(progress_bar)
|
||||
.with_finish(ProgressFinish::WithMessage(Cow::from("output generated")))
|
||||
.filter(|i| i.value().len() > 1)
|
||||
.map(|i| {
|
||||
row![
|
||||
i.key(),
|
||||
Self::gen_sub_tbl(i.value().to_vec(), args, mpath_len)
|
||||
]
|
||||
})
|
||||
.collect::<Vec<Row>>();
|
||||
|
||||
let mut output_table = Table::new();
|
||||
output_table.set_titles(row!["hash", "duplicates"]);
|
||||
output_table.extend(rows);
|
||||
Ok(output_table)
|
||||
}
|
||||
|
||||
pub fn print(
|
||||
raw: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
max_path_len: u64,
|
||||
app_args: &Params,
|
||||
) -> Result<()> {
|
||||
if raw.is_empty() {
|
||||
println!("\n\nNo duplicates found matching your search criteria.\n");
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let output_table = Self::generate_table(raw, max_path_len, app_args)?;
|
||||
output_table.printstd();
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
125
src/interactive.rs
Normal file
125
src/interactive.rs
Normal file
@@ -0,0 +1,125 @@
|
||||
use crate::{fileinfo::FileInfo, formatter::Formatter, params::Params};
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use prettytable::{format, row, Table};
|
||||
use std::{
|
||||
io::{self, Write},
|
||||
sync::Arc,
|
||||
};
|
||||
|
||||
pub struct Interactive;
|
||||
|
||||
impl Interactive {
|
||||
pub fn init(result: Arc<DashMap<u128, Vec<FileInfo>>>, app_args: &Params) -> Result<()> {
|
||||
result
|
||||
.clone()
|
||||
.iter()
|
||||
.filter(|i| i.value().len() > 1)
|
||||
.enumerate()
|
||||
.for_each(|(gindex, i)| {
|
||||
let group = i.value();
|
||||
let mut itable = Table::new();
|
||||
itable.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
|
||||
itable.set_titles(row!["index", "filename", "size", "updated_at"]);
|
||||
|
||||
let max_path_size = group
|
||||
.iter()
|
||||
.map(|f| f.path.iter().count())
|
||||
.max()
|
||||
.unwrap_or_default();
|
||||
|
||||
group.iter().enumerate().for_each(|(index, file)| {
|
||||
itable.add_row(row![
|
||||
index,
|
||||
Formatter::human_path(file, app_args, max_path_size).unwrap_or_default(),
|
||||
Formatter::human_filesize(file).unwrap_or_default(),
|
||||
Formatter::human_mtime(file).unwrap_or_default()
|
||||
]);
|
||||
});
|
||||
|
||||
Self::process_group_action(group, gindex, result.len(), itable);
|
||||
});
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn scan_group_confirmation() -> Result<bool> {
|
||||
print!("\nconfirm? [y/N]: ");
|
||||
std::io::stdout().flush()?;
|
||||
let mut user_input = String::new();
|
||||
io::stdin().read_line(&mut user_input)?;
|
||||
|
||||
match user_input.trim() {
|
||||
"Y" | "y" => Ok(true),
|
||||
_ => Ok(false),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn scan_group_instruction() -> Result<String> {
|
||||
println!("\nEnter the indices of the files you want to delete.");
|
||||
println!("You can enter multiple files using commas to seperate file indices.");
|
||||
println!("example: 1,2");
|
||||
print!("\n> ");
|
||||
std::io::stdout().flush()?;
|
||||
let mut user_input = String::new();
|
||||
io::stdin().read_line(&mut user_input)?;
|
||||
|
||||
Ok(user_input)
|
||||
}
|
||||
|
||||
pub fn process_group_action(
|
||||
duplicates: &Vec<FileInfo>,
|
||||
dup_index: usize,
|
||||
dup_size: usize,
|
||||
table: Table,
|
||||
) {
|
||||
println!("\nDuplicate Set {} of {}\n", dup_index + 1, dup_size);
|
||||
table.printstd();
|
||||
let files_to_delete = Self::scan_group_instruction().unwrap_or_default();
|
||||
let parsed_file_indices = files_to_delete
|
||||
.trim()
|
||||
.split(',')
|
||||
.filter(|element| !element.is_empty())
|
||||
.map(|index| index.parse::<usize>().unwrap_or_default())
|
||||
.collect::<Vec<usize>>();
|
||||
|
||||
if parsed_file_indices
|
||||
.clone()
|
||||
.into_iter()
|
||||
.any(|index| index > (duplicates.len() - 1))
|
||||
{
|
||||
println!("Err: File Index Out of Bounds!");
|
||||
return Self::process_group_action(duplicates, dup_index, dup_size, table);
|
||||
}
|
||||
|
||||
print!("{esc}[2J{esc}[1;1H", esc = 27 as char);
|
||||
|
||||
if parsed_file_indices.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
let files_to_delete = parsed_file_indices
|
||||
.into_iter()
|
||||
.map(|index| duplicates[index].clone());
|
||||
|
||||
println!("\nThe following files will be deleted:");
|
||||
files_to_delete
|
||||
.clone()
|
||||
.enumerate()
|
||||
.for_each(|(index, file)| {
|
||||
println!("{}: {}", index, file.path.display());
|
||||
});
|
||||
|
||||
match Self::scan_group_confirmation().unwrap() {
|
||||
true => {
|
||||
files_to_delete.into_iter().for_each(|file| {
|
||||
match std::fs::remove_file(file.path.clone()) {
|
||||
Ok(_) => println!("DELETED: {}", file.path.display()),
|
||||
Err(_) => println!("FAILED: {}", file.path.display()),
|
||||
}
|
||||
});
|
||||
}
|
||||
false => println!("\nCancelled Delete Operation."),
|
||||
}
|
||||
}
|
||||
}
|
||||
36
src/main.rs
36
src/main.rs
@@ -1,15 +1,35 @@
|
||||
mod app;
|
||||
mod file_manager;
|
||||
mod output;
|
||||
mod fileinfo;
|
||||
mod formatter;
|
||||
mod interactive;
|
||||
mod params;
|
||||
mod processor;
|
||||
mod scanner;
|
||||
mod filters;
|
||||
mod server;
|
||||
|
||||
use self::{formatter::Formatter, interactive::Interactive, server::Server};
|
||||
use anyhow::Result;
|
||||
use app::App;
|
||||
use clap::Parser;
|
||||
use params::Params;
|
||||
use std::sync::atomic::Ordering;
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> Result<()> {
|
||||
App::init(¶ms::Params::parse())
|
||||
fn main() -> Result<()> {
|
||||
let app_args = Params::parse();
|
||||
let server = Server::new(app_args.clone());
|
||||
|
||||
server.start()?;
|
||||
|
||||
match app_args.interactive {
|
||||
false => {
|
||||
Formatter::print(
|
||||
server.hw_duplicate_set,
|
||||
server.max_file_path_len.load(Ordering::Acquire),
|
||||
&app_args,
|
||||
)?;
|
||||
}
|
||||
true => {
|
||||
Interactive::init(server.hw_duplicate_set, &app_args)?;
|
||||
}
|
||||
};
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
183
src/output.rs
183
src/output.rs
@@ -1,183 +0,0 @@
|
||||
use crate::file_manager::{self, File};
|
||||
use crate::params::Params;
|
||||
use anyhow::Result;
|
||||
use chrono::offset::Utc;
|
||||
use chrono::DateTime;
|
||||
use colored::Colorize;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{ProgressBar, ProgressIterator, ProgressStyle};
|
||||
use itertools::Itertools;
|
||||
use prettytable::{format, row, Table};
|
||||
use std::io::Write;
|
||||
use std::{fs, io};
|
||||
use unicode_segmentation::UnicodeSegmentation;
|
||||
|
||||
fn format_path(path: &str, opts: &Params) -> Result<String> {
|
||||
let display_path = path.replace(opts.get_directory()?.to_string_lossy().as_ref(), "");
|
||||
let display_range = if display_path.chars().count() > 32 {
|
||||
display_path
|
||||
.graphemes(true)
|
||||
.collect::<Vec<&str>>()
|
||||
.into_iter()
|
||||
.rev()
|
||||
.take(32)
|
||||
.rev()
|
||||
.collect()
|
||||
} else {
|
||||
display_path
|
||||
};
|
||||
|
||||
Ok(format!("...{display_range:<32}"))
|
||||
}
|
||||
|
||||
fn file_size(file: &File) -> Result<String> {
|
||||
Ok(format!("{:>12}", bytesize::ByteSize::b(file.size.unwrap())))
|
||||
}
|
||||
|
||||
fn modified_time(path: &String) -> Result<String> {
|
||||
let mdata = fs::metadata(path)?;
|
||||
let modified_time: DateTime<Utc> = mdata.modified()?.into();
|
||||
|
||||
Ok(modified_time.format("%Y-%m-%d %H:%M:%S").to_string())
|
||||
}
|
||||
|
||||
fn scan_group_instruction() -> Result<String> {
|
||||
println!("\nEnter the indices of the files you want to delete.");
|
||||
println!("You can enter multiple files using commas to seperate file indices.");
|
||||
println!("example: 1,2");
|
||||
print!("\n> ");
|
||||
std::io::stdout().flush()?;
|
||||
let mut user_input = String::new();
|
||||
io::stdin().read_line(&mut user_input)?;
|
||||
|
||||
Ok(user_input)
|
||||
}
|
||||
|
||||
fn scan_group_confirmation() -> Result<bool> {
|
||||
print!("\nconfirm? [y/N]: ");
|
||||
std::io::stdout().flush()?;
|
||||
let mut user_input = String::new();
|
||||
io::stdin().read_line(&mut user_input)?;
|
||||
|
||||
match user_input.trim() {
|
||||
"Y" | "y" => Ok(true),
|
||||
_ => Ok(false),
|
||||
}
|
||||
}
|
||||
|
||||
fn process_group_action(duplicates: &Vec<File>, dup_index: usize, dup_size: usize, table: Table) {
|
||||
println!("\nDuplicate Set {} of {}\n", dup_index + 1, dup_size);
|
||||
table.printstd();
|
||||
let files_to_delete = scan_group_instruction().unwrap_or_default();
|
||||
let parsed_file_indices = files_to_delete
|
||||
.trim()
|
||||
.split(',')
|
||||
.filter(|element| !element.is_empty())
|
||||
.map(|index| index.parse::<usize>().unwrap_or_default())
|
||||
.collect_vec();
|
||||
|
||||
if parsed_file_indices
|
||||
.clone()
|
||||
.into_iter()
|
||||
.any(|index| index > (duplicates.len() - 1))
|
||||
{
|
||||
println!("{}", "Err: File Index Out of Bounds!".red());
|
||||
return process_group_action(duplicates, dup_index, dup_size, table);
|
||||
}
|
||||
|
||||
print!("{esc}[2J{esc}[1;1H", esc = 27 as char);
|
||||
|
||||
if parsed_file_indices.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
let files_to_delete = parsed_file_indices
|
||||
.into_iter()
|
||||
.map(|index| duplicates[index].clone());
|
||||
|
||||
println!("\n{}", "The following files will be deleted:".red());
|
||||
files_to_delete
|
||||
.clone()
|
||||
.enumerate()
|
||||
.for_each(|(index, file)| {
|
||||
println!("{}: {}", index.to_string().blue(), file.path);
|
||||
});
|
||||
|
||||
match scan_group_confirmation().unwrap() {
|
||||
true => {
|
||||
file_manager::delete_files(files_to_delete.collect_vec()).ok();
|
||||
}
|
||||
false => println!("{}", "\nCancelled Delete Operation.".red()),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn interactive(duplicates: DashMap<String, Vec<File>>, opts: &Params) {
|
||||
if duplicates.is_empty() {
|
||||
println!(
|
||||
"\n{}",
|
||||
"No duplicates found matching your search criteria.".green()
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
duplicates
|
||||
.clone()
|
||||
.into_iter()
|
||||
.sorted_unstable_by_key(|(_, f)| {
|
||||
-(f.first().and_then(|ff| ff.size).unwrap_or_default() as i64)
|
||||
}) // sort by descending file size in interactive mode
|
||||
.enumerate()
|
||||
.for_each(|(gindex, (_, group))| {
|
||||
let mut itable = Table::new();
|
||||
itable.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
|
||||
itable.set_titles(row!["index", "filename", "size", "updated_at"]);
|
||||
group.iter().enumerate().for_each(|(index, file)| {
|
||||
itable.add_row(row![
|
||||
index,
|
||||
format_path(&file.path, opts).unwrap_or_default().blue(),
|
||||
file_size(file).unwrap_or_default().red(),
|
||||
modified_time(&file.path).unwrap_or_default().yellow()
|
||||
]);
|
||||
});
|
||||
|
||||
process_group_action(&group, gindex, duplicates.len(), itable);
|
||||
});
|
||||
}
|
||||
|
||||
pub fn print(duplicates: DashMap<String, Vec<File>>, opts: &Params) {
|
||||
if duplicates.is_empty() {
|
||||
println!(
|
||||
"\n{}",
|
||||
"No duplicates found matching your search criteria.".green()
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
let mut output_table = Table::new();
|
||||
let progress_bar = ProgressBar::new(duplicates.len() as u64);
|
||||
let progress_style = ProgressStyle::default_bar()
|
||||
.template("{spinner:.green} [generating output] [{wide_bar:.cyan/blue}] {pos}/{len} files")
|
||||
.unwrap();
|
||||
|
||||
progress_bar.set_style(progress_style);
|
||||
output_table.set_titles(row!["hash", "duplicates"]);
|
||||
|
||||
duplicates
|
||||
.into_iter()
|
||||
.sorted_unstable_by_key(|(_, f)| f.first().and_then(|ff| ff.size).unwrap_or_default())
|
||||
.progress_with(progress_bar)
|
||||
.for_each(|(hash, group)| {
|
||||
let mut inner_table = Table::new();
|
||||
inner_table.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
|
||||
group.iter().for_each(|file| {
|
||||
inner_table.add_row(row![
|
||||
format_path(&file.path, opts).unwrap_or_default().blue(),
|
||||
file_size(file).unwrap_or_default().red(),
|
||||
modified_time(&file.path).unwrap_or_default().yellow()
|
||||
]);
|
||||
});
|
||||
output_table.add_row(row![hash.green(), inner_table]);
|
||||
});
|
||||
|
||||
output_table.printstd();
|
||||
}
|
||||
@@ -1,29 +1,43 @@
|
||||
use std::{fs, path::PathBuf};
|
||||
|
||||
use anyhow::{anyhow, Result};
|
||||
use anyhow::Result;
|
||||
use clap::{Parser, ValueHint};
|
||||
use globwalk::{GlobWalker, GlobWalkerBuilder};
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
#[derive(Parser, Debug, Default, Clone)]
|
||||
#[command(author, version, about, long_about = None)]
|
||||
pub struct Params {
|
||||
/// Filetypes to deduplicate (default = all)
|
||||
/// Filetypes to deduplicate [default = all]
|
||||
#[arg(short, long)]
|
||||
pub types: Option<String>,
|
||||
/// Run Deduplicator on dir different from pwd
|
||||
#[arg(long, value_hint = ValueHint::DirPath)]
|
||||
/// Run Deduplicator on dir different from pwd (e.g., ~/Pictures )
|
||||
#[arg(value_hint = ValueHint::DirPath, value_name = "scan_dir_path")]
|
||||
pub dir: Option<PathBuf>,
|
||||
/// Delete files interactively
|
||||
#[arg(long, short)]
|
||||
pub interactive: bool,
|
||||
/// Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T). [default = 0]
|
||||
/// Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T).
|
||||
#[arg(long, short = 'm', default_value = "1b")]
|
||||
pub min_size: Option<String>,
|
||||
/// Max Depth to scan while looking for duplicates
|
||||
#[arg(long, short = 'D')]
|
||||
pub max_depth: Option<usize>,
|
||||
/// Min Depth to scan while looking for duplicates
|
||||
#[arg(long, short = 'd')]
|
||||
pub min_depth: Option<usize>,
|
||||
/// Follow links while scanning directories
|
||||
#[arg(long, short)]
|
||||
pub minsize: Option<String>,
|
||||
pub follow_links: bool,
|
||||
/// Guarantees that two files are duplicate (performs a full hash)
|
||||
#[arg(long, short = 's', default_value = "false")]
|
||||
pub strict: bool,
|
||||
/// Show Progress spinners & metrics
|
||||
#[arg(long, short = 'p', default_value = "false")]
|
||||
pub progress: bool,
|
||||
}
|
||||
|
||||
impl Params {
|
||||
pub fn get_minsize(&self) -> Option<u64> {
|
||||
match &self.minsize {
|
||||
pub fn get_min_size(&self) -> Option<u64> {
|
||||
match &self.min_size {
|
||||
Some(msize) => match msize.parse::<bytesize::ByteSize>() {
|
||||
Ok(units) => Some(units.0),
|
||||
Err(_) => None,
|
||||
@@ -39,14 +53,7 @@ impl Params {
|
||||
Ok(dir)
|
||||
}
|
||||
|
||||
pub fn get_glob_walker(&self) -> Result<GlobWalker> {
|
||||
let pattern: String = match self.types.as_ref() {
|
||||
Some(filetypes) => format!("**/*{{{filetypes}}}"),
|
||||
None => "**/*".to_string(),
|
||||
};
|
||||
// TODO: add params for maximum depth and following symlinks, then pass them to this builder
|
||||
GlobWalkerBuilder::from_patterns(self.get_directory()?, &[pattern])
|
||||
.build()
|
||||
.map_err(|e| anyhow!(e))
|
||||
pub fn get_types(&self) -> Option<String> {
|
||||
self.types.clone()
|
||||
}
|
||||
}
|
||||
|
||||
369
src/processor.rs
Normal file
369
src/processor.rs
Normal file
@@ -0,0 +1,369 @@
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{
|
||||
MultiProgress, ParallelProgressIterator, ProgressBar, ProgressFinish, ProgressStyle,
|
||||
};
|
||||
use rayon::prelude::{IntoParallelIterator, ParallelIterator};
|
||||
use std::borrow::Cow;
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::sync::{Arc, Mutex, TryLockError, TryLockResult};
|
||||
use std::time::Duration;
|
||||
|
||||
use crate::fileinfo::FileInfo;
|
||||
use crate::params::Params;
|
||||
|
||||
pub struct Processor {}
|
||||
|
||||
impl Processor {
|
||||
pub fn hashwise(
|
||||
app_args: Arc<Params>,
|
||||
sw_store: Arc<DashMap<u64, Vec<FileInfo>>>,
|
||||
hw_store: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
progress_bar_box: Arc<MultiProgress>,
|
||||
max_file_size: Arc<AtomicU64>,
|
||||
seed: i64,
|
||||
) -> Result<()> {
|
||||
let progress_bar = match app_args.progress {
|
||||
true => progress_bar_box.add(ProgressBar::new_spinner()),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let keys: Vec<u64> = sw_store.clone().iter().map(|i| *i.key()).collect();
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("files grouped by hash.");
|
||||
|
||||
keys.into_par_iter()
|
||||
.progress_with(progress_bar)
|
||||
.with_finish(ProgressFinish::WithMessage(Cow::from(
|
||||
"files grouped by hash.",
|
||||
)))
|
||||
.for_each(|key| {
|
||||
let group: Vec<FileInfo> = sw_store.get(&key).unwrap().to_vec();
|
||||
if group.len() > 1 {
|
||||
group.into_par_iter().for_each(|file| {
|
||||
let fhash = if app_args.strict {
|
||||
file.hash(seed).expect("hashing file failed.")
|
||||
} else {
|
||||
file.initial_page_hash(seed).expect("hashing file failed.")
|
||||
};
|
||||
|
||||
Self::compare_and_update_max_path_len(
|
||||
max_file_size.clone(),
|
||||
file.path.to_string_lossy().len() as u64,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
hw_store
|
||||
.entry(fhash)
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file]);
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn compare_and_update_max_path_len(current: Arc<AtomicU64>, next: u64) -> Result<()> {
|
||||
if current.load(Ordering::Relaxed) < next {
|
||||
current.store(next, Ordering::Release);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn sizewise(
|
||||
app_args: Arc<Params>,
|
||||
scanner_finished: Arc<AtomicBool>,
|
||||
store: Arc<DashMap<u64, Vec<FileInfo>>>,
|
||||
files: Arc<Mutex<Vec<FileInfo>>>,
|
||||
progress_bar_box: Arc<MultiProgress>,
|
||||
) -> Result<()> {
|
||||
let progress_bar = match app_args.progress {
|
||||
true => progress_bar_box.add(ProgressBar::new_spinner()),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("files grouped by size");
|
||||
|
||||
loop {
|
||||
let fileopt: Option<FileInfo> = {
|
||||
match files.try_lock() {
|
||||
Ok(mut flist) => flist.pop(),
|
||||
TryLockResult::Err(TryLockError::WouldBlock) => None,
|
||||
_ => None,
|
||||
}
|
||||
};
|
||||
|
||||
match fileopt {
|
||||
Some(file) => {
|
||||
progress_bar.inc(1);
|
||||
store
|
||||
.entry(file.size)
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file]);
|
||||
continue;
|
||||
}
|
||||
None => match scanner_finished.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
true => {
|
||||
progress_bar.finish_with_message("files grouped by size");
|
||||
break Ok(());
|
||||
}
|
||||
false => continue,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::MultiProgress;
|
||||
use rand::Rng;
|
||||
use std::fs::File;
|
||||
use std::io::Write;
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use tempfile::TempDir;
|
||||
|
||||
use crate::{fileinfo::FileInfo, params::Params};
|
||||
|
||||
use super::Processor;
|
||||
|
||||
fn generate_bytes(size: usize) -> Vec<u8> {
|
||||
let mut rng = rand::rng();
|
||||
(0..size).map(|_| rng.random::<u8>()).collect::<Vec<u8>>()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_init_page_only_strict_mode() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(4096);
|
||||
|
||||
let mut content_x = content.clone();
|
||||
let mut content_y = content.clone();
|
||||
|
||||
content_x.extend(generate_bytes(1720320));
|
||||
content_y.extend(generate_bytes(1720320));
|
||||
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), content_x),
|
||||
(root.path().join("filetwo.bin"), content_y),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let hw_dupstore = Arc::new(DashMap::new());
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
let args = Params {
|
||||
strict: true,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
Processor::hashwise(
|
||||
Arc::new(args),
|
||||
dupstore.clone(),
|
||||
hw_dupstore.clone(),
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 2);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_init_page_only_fast_mode() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(4096);
|
||||
|
||||
let mut content_x = content.clone();
|
||||
let mut content_y = content.clone();
|
||||
|
||||
content_x.extend(generate_bytes(1720320));
|
||||
content_y.extend(generate_bytes(1720320));
|
||||
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), content_x),
|
||||
(root.path().join("filetwo.bin"), content_y),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let hw_dupstore = Arc::new(DashMap::new());
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
Processor::hashwise(
|
||||
Arc::new(Params::default()),
|
||||
dupstore.clone(),
|
||||
hw_dupstore.clone(),
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 1);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_data() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(282624);
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), content.clone()),
|
||||
(root.path().join("filetwo.bin"), content.clone()),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let hw_dupstore = Arc::new(DashMap::new());
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
Processor::hashwise(
|
||||
Arc::new(Params::default()),
|
||||
dupstore.clone(),
|
||||
hw_dupstore.clone(),
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 1);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sizewise_sorting_two_files_of_different_sizes() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), generate_bytes(282624)),
|
||||
(root.path().join("filetwo.bin"), generate_bytes(1720320)),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
assert_eq!(dupstore.len(), 2);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sizewise_sorting_two_files_of_same_size() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), generate_bytes(282624)),
|
||||
(root.path().join("filetwo.bin"), generate_bytes(282624)),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
assert_eq!(dupstore.len(), 1);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
199
src/scanner.rs
199
src/scanner.rs
@@ -1,133 +1,100 @@
|
||||
use crate::{file_manager::File, filters, params::Params};
|
||||
use crate::{fileinfo::FileInfo, params::Params};
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use fxhash::hash64 as hasher;
|
||||
use indicatif::{ParallelProgressIterator, ProgressBar, ProgressIterator, ProgressStyle};
|
||||
use memmap2::Mmap;
|
||||
use rayon::prelude::*;
|
||||
use std::hash::Hasher;
|
||||
use std::time::Duration;
|
||||
use std::{fs, path::PathBuf};
|
||||
use indicatif::{MultiProgress, ProgressBar, ProgressStyle};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::{path::Path, time::Duration};
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
enum IndexCritera {
|
||||
Size,
|
||||
Hash,
|
||||
use globwalk::{GlobWalker, GlobWalkerBuilder};
|
||||
|
||||
pub struct Scanner {
|
||||
pub directory: Box<Path>,
|
||||
pub filetypes: Option<String>,
|
||||
pub min_depth: Option<usize>,
|
||||
pub max_depth: Option<usize>,
|
||||
pub min_size: Option<u64>,
|
||||
pub follow_links: bool,
|
||||
pub progress: bool,
|
||||
}
|
||||
|
||||
pub fn duplicates(app_opts: &Params) -> Result<DashMap<String, Vec<File>>> {
|
||||
let scan_results = scan(app_opts)?;
|
||||
let size_index_store = index_files(scan_results, IndexCritera::Size)?;
|
||||
|
||||
let sizewize_duplicate_files = size_index_store
|
||||
.into_par_iter()
|
||||
.filter(|(_, files)| files.len() > 1)
|
||||
.map(|(_, files)| files)
|
||||
.flatten()
|
||||
.collect::<Vec<File>>();
|
||||
|
||||
if sizewize_duplicate_files.len() > 1 {
|
||||
let hash_index_store = index_files(sizewize_duplicate_files, IndexCritera::Hash)?;
|
||||
let duplicate_files = hash_index_store
|
||||
.into_par_iter()
|
||||
.filter(|(_, files)| files.len() > 1)
|
||||
.collect();
|
||||
|
||||
Ok(duplicate_files)
|
||||
} else {
|
||||
Ok(DashMap::new())
|
||||
}
|
||||
}
|
||||
|
||||
fn scan(app_opts: &Params) -> Result<Vec<File>> {
|
||||
let walker = app_opts.get_glob_walker()?;
|
||||
let progress = ProgressBar::new_spinner();
|
||||
let progress_style =
|
||||
ProgressStyle::with_template("{spinner:.green} [mapping paths] {pos} paths")?;
|
||||
progress.set_style(progress_style);
|
||||
progress.enable_steady_tick(Duration::from_millis(100));
|
||||
|
||||
let files = walker
|
||||
.progress_with(progress)
|
||||
.filter_map(Result::ok)
|
||||
.map(|file| file.into_path())
|
||||
.filter(|fpath| fpath.is_file())
|
||||
.collect::<Vec<PathBuf>>()
|
||||
.into_par_iter()
|
||||
.progress_with_style(ProgressStyle::with_template(
|
||||
"{spinner:.green} [processing mapped paths] [{wide_bar:.cyan/blue}] {pos}/{len} files",
|
||||
)?)
|
||||
.map(|fpath| fpath.display().to_string())
|
||||
.map(|fpath| File {
|
||||
path: fpath.clone(),
|
||||
hash: None,
|
||||
size: Some(fs::metadata(fpath).unwrap().len()),
|
||||
impl Scanner {
|
||||
pub fn new(app_args: Arc<Params>) -> Result<Self> {
|
||||
Ok(Self {
|
||||
directory: app_args.get_directory()?.into_boxed_path(),
|
||||
filetypes: app_args.get_types(),
|
||||
min_depth: app_args.min_depth,
|
||||
max_depth: app_args.max_depth,
|
||||
min_size: app_args.get_min_size(),
|
||||
follow_links: app_args.follow_links,
|
||||
progress: app_args.progress,
|
||||
})
|
||||
.filter(|file| filters::is_file_gt_minsize(app_opts, file))
|
||||
.collect();
|
||||
}
|
||||
|
||||
Ok(files)
|
||||
}
|
||||
fn scan_patterns(&self) -> Result<String> {
|
||||
Ok(match &self.filetypes {
|
||||
Some(ftypes) => format!("**/*{{{ftypes}}}"),
|
||||
None => "**/*".to_string(),
|
||||
})
|
||||
}
|
||||
|
||||
fn process_file_index(
|
||||
mut file: File,
|
||||
store: &DashMap<String, Vec<File>>,
|
||||
index_criteria: IndexCritera,
|
||||
) {
|
||||
match index_criteria {
|
||||
IndexCritera::Size => {
|
||||
store
|
||||
.entry(file.size.unwrap_or_default().to_string())
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file]);
|
||||
}
|
||||
IndexCritera::Hash => {
|
||||
file.hash = Some(hash_file(&file.path).unwrap_or_default());
|
||||
store
|
||||
.entry(file.clone().hash.unwrap())
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file]);
|
||||
fn attach_link_opts(&self, walker: GlobWalkerBuilder) -> Result<GlobWalkerBuilder> {
|
||||
Ok(walker.follow_links(self.follow_links))
|
||||
}
|
||||
|
||||
fn attach_walker_min_depth(&self, walker: GlobWalkerBuilder) -> Result<GlobWalkerBuilder> {
|
||||
match self.min_depth {
|
||||
Some(min_depth) => Ok(walker.min_depth(min_depth)),
|
||||
None => Ok(walker),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn index_files(
|
||||
files: Vec<File>,
|
||||
index_criteria: IndexCritera,
|
||||
) -> Result<DashMap<String, Vec<File>>> {
|
||||
let store: DashMap<String, Vec<File>> = DashMap::new();
|
||||
files
|
||||
.into_par_iter()
|
||||
.progress_with_style(ProgressStyle::with_template(
|
||||
"{spinner:.green} [indexing files] [{wide_bar:.cyan/blue}] {pos}/{len} files",
|
||||
)?)
|
||||
.for_each(|file| process_file_index(file, &store, index_criteria));
|
||||
fn attach_walker_max_depth(&self, walker: GlobWalkerBuilder) -> Result<GlobWalkerBuilder> {
|
||||
match self.max_depth {
|
||||
Some(max_depth) => Ok(walker.max_depth(max_depth)),
|
||||
None => Ok(walker),
|
||||
}
|
||||
}
|
||||
fn build_walker(&self) -> Result<GlobWalker> {
|
||||
let walker = Ok(GlobWalkerBuilder::from_patterns(
|
||||
self.directory.clone(),
|
||||
&[self.scan_patterns()?],
|
||||
))
|
||||
.and_then(|walker| self.attach_walker_min_depth(walker))
|
||||
.and_then(|walker| self.attach_walker_max_depth(walker))
|
||||
.and_then(|walker| self.attach_link_opts(walker))?;
|
||||
|
||||
Ok(store)
|
||||
}
|
||||
Ok(walker.build()?)
|
||||
}
|
||||
|
||||
fn incremental_hashing(filepath: &str) -> Result<String> {
|
||||
let file = fs::File::open(filepath)?;
|
||||
let fmap = unsafe { Mmap::map(&file)? };
|
||||
let mut inchasher = fxhash::FxHasher::default();
|
||||
pub fn scan(
|
||||
&self,
|
||||
files: Arc<Mutex<Vec<FileInfo>>>,
|
||||
progress_bar_box: Arc<MultiProgress>,
|
||||
) -> Result<()> {
|
||||
let progress_bar = match self.progress {
|
||||
true => progress_bar_box.add(ProgressBar::new_spinner()),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
fmap.chunks(1_000_000)
|
||||
.for_each(|mega| inchasher.write(mega));
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("paths mapped");
|
||||
let min_size = self.min_size.unwrap_or_default();
|
||||
|
||||
Ok(format!("{}", inchasher.finish()))
|
||||
}
|
||||
self.build_walker()?
|
||||
.filter_map(Result::ok)
|
||||
.map(|entity| entity.into_path())
|
||||
.inspect(|_path| progress_bar.inc(1))
|
||||
.filter(|path| path.is_file())
|
||||
.map(FileInfo::new)
|
||||
.filter_map(Result::ok)
|
||||
.filter(|file| file.size > min_size)
|
||||
.for_each(|file| {
|
||||
let mut flock = files.lock().unwrap();
|
||||
flock.push(file);
|
||||
});
|
||||
|
||||
fn standard_hashing(filepath: &str) -> Result<String> {
|
||||
let file = fs::read(filepath)?;
|
||||
Ok(hasher(&*file).to_string())
|
||||
}
|
||||
|
||||
fn hash_file(filepath: &str) -> Result<String> {
|
||||
let filemeta = fs::metadata(filepath)?;
|
||||
|
||||
// NOTE: USE INCREMENTAL HASHING ONLY FOR FILES > 100MB
|
||||
match filemeta.len() < 100_000_000 {
|
||||
true => standard_hashing(filepath),
|
||||
false => incremental_hashing(filepath),
|
||||
progress_bar.finish_with_message("paths mapped");
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
99
src/server.rs
Normal file
99
src/server.rs
Normal file
@@ -0,0 +1,99 @@
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use crate::processor::Processor;
|
||||
use crate::scanner::Scanner;
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{MultiProgress, ProgressDrawTarget};
|
||||
use rand::Rng;
|
||||
use threadpool::ThreadPool;
|
||||
|
||||
use crate::fileinfo::FileInfo;
|
||||
use crate::params::Params;
|
||||
|
||||
pub struct Server {
|
||||
filequeue: Arc<Mutex<Vec<FileInfo>>>,
|
||||
sw_duplicate_set: Arc<DashMap<u64, Vec<FileInfo>>>,
|
||||
pub hw_duplicate_set: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
threadpool: ThreadPool,
|
||||
app_args: Arc<Params>,
|
||||
pub max_file_path_len: Arc<AtomicU64>,
|
||||
}
|
||||
|
||||
impl Server {
|
||||
pub fn new(opts: Params) -> Self {
|
||||
Self {
|
||||
filequeue: Arc::new(Mutex::new(Vec::new())),
|
||||
sw_duplicate_set: Arc::new(DashMap::new()),
|
||||
hw_duplicate_set: Arc::new(DashMap::new()),
|
||||
threadpool: ThreadPool::new(4),
|
||||
app_args: Arc::new(opts),
|
||||
max_file_path_len: Arc::new(AtomicU64::new(0)),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn start(&self) -> Result<()> {
|
||||
let progbarbox = Arc::new(MultiProgress::new());
|
||||
let mut rng = rand::rng();
|
||||
let seed: i64 = rng.random();
|
||||
|
||||
if !self.app_args.progress {
|
||||
progbarbox.set_draw_target(ProgressDrawTarget::hidden());
|
||||
}
|
||||
|
||||
let app_args_clone_for_sc = self.app_args.clone();
|
||||
let app_args_clone_for_pr = self.app_args.clone();
|
||||
let file_queue_clone_sc = self.filequeue.clone();
|
||||
let file_queue_clone_pr = self.filequeue.clone();
|
||||
let scanner_finished = Arc::new(AtomicBool::new(false));
|
||||
|
||||
let sfin_sc_tr_cl = scanner_finished.clone();
|
||||
let sfin_pr_tr_cl = scanner_finished.clone();
|
||||
|
||||
let store_dupl_sw_for_sw = self.sw_duplicate_set.clone();
|
||||
let store_dupl_sw_for_hw = self.sw_duplicate_set.clone();
|
||||
let store_dupl_hw = self.hw_duplicate_set.clone();
|
||||
let max_file_path_len_clone = self.max_file_path_len.clone();
|
||||
|
||||
let progbarbox_sc_clone = progbarbox.clone();
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Scanner::new(app_args_clone_for_sc)
|
||||
.unwrap()
|
||||
.scan(file_queue_clone_sc, progbarbox_sc_clone)
|
||||
.unwrap();
|
||||
|
||||
sfin_sc_tr_cl.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
});
|
||||
|
||||
let progbarbox_pr_clone = progbarbox.clone();
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Processor::sizewise(
|
||||
app_args_clone_for_pr.clone(),
|
||||
sfin_pr_tr_cl,
|
||||
store_dupl_sw_for_sw,
|
||||
file_queue_clone_pr,
|
||||
progbarbox_pr_clone.clone(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
Processor::hashwise(
|
||||
app_args_clone_for_pr,
|
||||
store_dupl_sw_for_hw,
|
||||
store_dupl_hw,
|
||||
progbarbox_pr_clone,
|
||||
max_file_path_len_clone,
|
||||
seed,
|
||||
)
|
||||
.unwrap();
|
||||
});
|
||||
|
||||
progbarbox.clear()?;
|
||||
|
||||
self.threadpool.join();
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user