mirror of
https://github.com/sreedevk/deduplicator.git
synced 2026-08-26 10:05:32 +00:00
Compare commits
150 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d06fe93171 | ||
|
|
6a24ca5ecf | ||
|
|
6f23a51a53 | ||
|
|
4894d1b4db | ||
|
|
c0286b17cc | ||
|
|
e78dfc4211 | ||
|
|
66f49a9aee | ||
|
|
711eb36fb9 | ||
|
|
130a8f99ba | ||
|
|
d97af6b8f6 | ||
|
|
90198502a3 | ||
|
|
e1b81b9981 | ||
|
|
48e7c62036 | ||
|
|
58e33cc8da | ||
|
|
f9e86f8522 | ||
|
|
b4bf5d4fb3 | ||
|
|
028b868ea9 | ||
|
|
a56d194ee3 | ||
|
|
080cd791dc | ||
|
|
b16236763c | ||
|
|
3e407f69c8 | ||
|
|
7d386c9420 | ||
|
|
0dc681d4e7 | ||
|
|
442cb4b519 | ||
|
|
8463e72f2d | ||
|
|
05c95bb67a | ||
|
|
062d44acd9 | ||
|
|
ee6655de9b | ||
|
|
ffa5295598 | ||
|
|
6be8596992 | ||
|
|
a40f251e30 | ||
|
|
02d05172da | ||
|
|
dcc709a666 | ||
|
|
e3d48ec505 | ||
|
|
7bd88f1642 | ||
|
|
a65e22269c | ||
|
|
6e3fe4a37d | ||
|
|
eadffb5fea | ||
|
|
5532ded331 | ||
|
|
7119f019d3 | ||
|
|
493cac1762 | ||
|
|
f0dbf05705 | ||
|
|
ef1e9a1fce | ||
|
|
d06ca7897d | ||
|
|
b49940998b | ||
|
|
ab44d1ed04 | ||
|
|
6b06798e8e | ||
|
|
4d27da99f3 | ||
|
|
91dce29bb4 | ||
|
|
f9b6d57968 | ||
|
|
5c86a080c4 | ||
|
|
9f4d9139b6 | ||
|
|
153109ef57 | ||
|
|
fa12f85b6a | ||
|
|
7d66aeef6e | ||
|
|
163e8d85c0 | ||
|
|
b1c508e9e7 | ||
|
|
4125b18d67 | ||
|
|
10329015f7 | ||
|
|
de6d2e325e | ||
|
|
850c274adc | ||
|
|
31d35aae51 | ||
|
|
087f36ac52 | ||
|
|
6f8b1d55df | ||
|
|
c6318a310b | ||
|
|
76c69ef8a0 | ||
|
|
fd4e882aa8 | ||
|
|
fd1d214d62 | ||
|
|
1a573fbf3f | ||
|
|
b12b662ba1 | ||
|
|
96337b6b48 | ||
|
|
02f70aa55f | ||
|
|
4a18b3fd91 | ||
|
|
876fdc69ac | ||
|
|
8acc3f8c8f | ||
|
|
38f6c48452 | ||
|
|
3e1e9c0e2e | ||
|
|
0674ab55bf | ||
|
|
ccff95dfc5 | ||
|
|
53d28abb62 | ||
|
|
ea6cf94952 | ||
|
|
7b275d40d0 | ||
|
|
30398c3ca9 | ||
|
|
c0042fc9f7 | ||
|
|
5aea0eb6f4 | ||
|
|
f0ff1ec325 | ||
|
|
b267fcdedf | ||
|
|
84b194efca | ||
|
|
d162ca98ef | ||
|
|
dfb73ceb5c | ||
|
|
72ca7c7a44 | ||
|
|
2f3cfd3162 | ||
|
|
9b1f591c20 | ||
|
|
e24603c645 | ||
|
|
0e624b042d | ||
|
|
e61d1f1475 | ||
|
|
8c76c6a3c8 | ||
|
|
df3bea3d4f | ||
|
|
7ace0bde63 | ||
|
|
70de402eab | ||
|
|
01dd93a0ea | ||
|
|
8ac78fb856 | ||
|
|
37a4c4d52d | ||
|
|
1ea5705474 | ||
|
|
ea748c60d8 | ||
|
|
51f5f8e61a | ||
|
|
175c7579c4 | ||
|
|
81c96ce3b8 | ||
|
|
0031891b8b | ||
|
|
ae87e4e830 | ||
|
|
284e168453 | ||
|
|
a6511f2cf3 | ||
|
|
64d0106765 | ||
|
|
c75b2eb1c9 | ||
|
|
b736853dfe | ||
|
|
92290480a8 | ||
|
|
b7f775e04c | ||
|
|
12295d7847 | ||
|
|
d81f499db3 | ||
|
|
9e1360aeb4 | ||
|
|
38ea37711f | ||
|
|
2294471b50 | ||
|
|
dbf504fc17 | ||
|
|
06efb2ed6f | ||
|
|
95af6c4a70 | ||
|
|
dd4b051378 | ||
|
|
533f81f724 | ||
|
|
471e60fa6c | ||
|
|
3fd4869869 | ||
|
|
99b87cf7fa | ||
|
|
b826dbe118 | ||
|
|
2c75a017ce | ||
|
|
b5bb58b3ed | ||
|
|
6e412fc59c | ||
|
|
a0083cb571 | ||
|
|
0f33b4d6b5 | ||
|
|
96fe667d3f | ||
|
|
c32e64450f | ||
|
|
85acbded46 | ||
|
|
f12cecc9ee | ||
|
|
05f513735e | ||
|
|
552f6c73f2 | ||
|
|
138b66038f | ||
|
|
d8e1de169d | ||
|
|
bc170f139e | ||
|
|
c76ad81a55 | ||
|
|
254e61cabe | ||
|
|
a76163f24f | ||
|
|
27fef21be0 | ||
|
|
8b00faf075 |
30
.github/ISSUE_TEMPLATE/bug_report.md
vendored
Normal file
30
.github/ISSUE_TEMPLATE/bug_report.md
vendored
Normal file
@@ -0,0 +1,30 @@
|
||||
---
|
||||
name: Bug report
|
||||
about: Create a report to help us improve
|
||||
title: "[Bug] Title"
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
** Runtime Info **
|
||||
App Arguments: [e.g. `-i --nocache`]
|
||||
Install Type: [e.g. `cargo install`]
|
||||
App Version: [e.g. v0.0.7]
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Screenshots**
|
||||
If applicable, add screenshots to help explain your problem.
|
||||
|
||||
**Platform Details (please complete the following information):**
|
||||
- OS: [e.g. Arch Linux]
|
||||
- Terminal Emulator: [e.g Alacritty]
|
||||
- Shell [e.g. Zshell]
|
||||
|
||||
**Additional context**
|
||||
Add any other context about the problem here.
|
||||
20
.github/ISSUE_TEMPLATE/feature-request.md
vendored
Normal file
20
.github/ISSUE_TEMPLATE/feature-request.md
vendored
Normal file
@@ -0,0 +1,20 @@
|
||||
---
|
||||
name: Feature request
|
||||
about: Suggest an idea for this project
|
||||
title: "[Feature] Title"
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Is your feature request related to a problem? Please describe.**
|
||||
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
||||
|
||||
**Describe the solution you'd like**
|
||||
A clear and concise description of what you want to happen.
|
||||
|
||||
**Describe alternatives you've considered**
|
||||
A clear and concise description of any alternative solutions or features you've considered.
|
||||
|
||||
**Additional context**
|
||||
Add any other context or screenshots about the feature request here.
|
||||
121
.github/workflows/release.yml
vendored
Normal file
121
.github/workflows/release.yml
vendored
Normal file
@@ -0,0 +1,121 @@
|
||||
name: Deduplicator Release Build CI Pipeline
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*.*.*'
|
||||
|
||||
jobs:
|
||||
lint_test:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
RUSTFLAGS: "-C target-feature=+aes,+sse2"
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
components: clippy
|
||||
|
||||
- name: Run cargo check
|
||||
run: cargo check
|
||||
|
||||
- name: Run cargo clippy
|
||||
run: cargo clippy -- -D warnings
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test -- --test-threads=1
|
||||
|
||||
build:
|
||||
needs: lint_test
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- target: aarch64-unknown-linux-gnu
|
||||
os: ubuntu-latest
|
||||
artifact_name: linux-aarch64
|
||||
rustflags: "-C target-feature=+aes,+neon"
|
||||
- target: x86_64-unknown-linux-gnu
|
||||
os: ubuntu-latest
|
||||
artifact_name: linux-amd64
|
||||
rustflags: "-C target-feature=+aes,+sse2"
|
||||
- target: x86_64-apple-darwin
|
||||
os: macos-14
|
||||
artifact_name: macos-amd64
|
||||
rustflags: "-C target-feature=+aes,+sse2"
|
||||
- target: aarch64-apple-darwin
|
||||
os: macos-14
|
||||
artifact_name: macos-arm64
|
||||
rustflags: "-C target-feature=+aes,+neon"
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Add target
|
||||
run: rustup target add ${{ matrix.target }}
|
||||
|
||||
- name: Install cross-compilation toolchain (Linux ARM only)
|
||||
if: matrix.target == 'aarch64-unknown-linux-gnu'
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y gcc-aarch64-linux-gnu
|
||||
|
||||
- name: Build release binary
|
||||
env:
|
||||
RUSTFLAGS: ${{ matrix.rustflags }}
|
||||
CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER: ${{ matrix.target == 'aarch64-unknown-linux-gnu' && 'aarch64-linux-gnu-gcc' || '' }}
|
||||
run: |
|
||||
if [ "${{ matrix.target }}" = "aarch64-unknown-linux-gnu" ]; then
|
||||
export CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER=aarch64-linux-gnu-gcc
|
||||
fi
|
||||
cargo build --release --target ${{ matrix.target }}
|
||||
|
||||
- name: Package binary
|
||||
run: |
|
||||
if [[ "${{ matrix.os }}" == macos* ]]; then
|
||||
BINARY_NAME=$(find target/${{ matrix.target }}/release -maxdepth 1 -type f -perm +111 -print0 | xargs -0 basename | head -1)
|
||||
else
|
||||
BINARY_NAME=$(find target/${{ matrix.target }}/release -maxdepth 1 -type f -executable -print0 | xargs -0 basename | head -1)
|
||||
fi
|
||||
|
||||
echo "Packaging binary: $BINARY_NAME"
|
||||
mkdir -p release
|
||||
cp target/${{ matrix.target }}/release/$BINARY_NAME release/$BINARY_NAME
|
||||
tar -C release -czf ${{ matrix.artifact_name }}.tar.gz $BINARY_NAME
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: ${{ matrix.artifact_name }}-binary
|
||||
path: ${{ matrix.artifact_name }}.tar.gz
|
||||
|
||||
create_release:
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
path: artifacts
|
||||
|
||||
- name: Create GitHub Release
|
||||
id: create_release
|
||||
uses: softprops/action-gh-release@v1
|
||||
with:
|
||||
tag_name: ${{ github.ref_name }}
|
||||
name: Release ${{ github.ref_name }}
|
||||
body: "Automated release for version ${{ github.ref_name }}"
|
||||
files: |
|
||||
artifacts/*/*.tar.gz
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
4
.gitignore
vendored
4
.gitignore
vendored
@@ -1,2 +1,4 @@
|
||||
/target
|
||||
/test_data
|
||||
/Cargo.lock
|
||||
/result-bin
|
||||
/.bacon-locations
|
||||
|
||||
28
CONTRIBUTING.md
Normal file
28
CONTRIBUTING.md
Normal file
@@ -0,0 +1,28 @@
|
||||
## How to contribute to Deduplicator
|
||||
|
||||
#### **Did you find a bug?**
|
||||
|
||||
* **Ensure the bug was not already reported** by searching on GitHub under [Issues](https://github.com/sreedevk/deduplicator/issues).
|
||||
|
||||
* If you're unable to find an open issue addressing the problem, [open a new one](https://github.com/sreedevk/deduplicator/issues/new). Be sure to include a **title and clear description**, as much relevant information as possible, and a **code sample** or an **executable test case** demonstrating the expected behavior that is not occurring.
|
||||
|
||||
* If possible, use the [bug report template](https://github.com/sreedevk/deduplicator/blob/main/.github/ISSUE_TEMPLATE/bug_report.md) to create the issue.
|
||||
|
||||
#### **Would you like to write a fix for the bug?**
|
||||
* Assign the Issue to yourself (if unassigned) before you start working in order to avoid any conficts.
|
||||
* Open a new GitHub pull request with the patch.
|
||||
* Ensure the PR description clearly describes the problem and solution. Include the relevant issue number.
|
||||
* Make sure that the PR points to the development branch.
|
||||
|
||||
#### **Did you fix whitespace, format code, or make a purely cosmetic patch?**
|
||||
|
||||
Changes that are cosmetic in nature and do not add anything substantial to the stability, functionality, or testability of Deduplicator will generally not be accepted/
|
||||
|
||||
#### **Do you intend to add a new feature or change an existing one?**
|
||||
|
||||
* First open an issue with the sugggestion using the [feature request template](https://github.com/sreedevk/deduplicator/blob/main/.github/ISSUE_TEMPLATE/feature-request.md)
|
||||
* Do not create a PR before one of the core contributors has conveyed acceptance for a feature request.
|
||||
|
||||
#### **Do you have questions about the source code?**
|
||||
|
||||
* If you have a question, raise an issue in the repository with a "question" label.
|
||||
1383
Cargo.lock
generated
1383
Cargo.lock
generated
File diff suppressed because it is too large
Load Diff
61
Cargo.toml
61
Cargo.toml
@@ -1,25 +1,58 @@
|
||||
[package]
|
||||
name = "deduplicator"
|
||||
version = "0.0.4"
|
||||
version = "0.3.1"
|
||||
edition = "2021"
|
||||
description = "find,filter,delete Duplicates"
|
||||
description = "find,filter and delete duplicate files"
|
||||
repository = "https://github.com/sreedevk/deduplicator"
|
||||
license = "MIT"
|
||||
authors = ["Sreedev Kodichath <sreedevpadmakumar@gmail.com>", "Valentin Bersier <vbersier@gmail.com>"]
|
||||
authors = [
|
||||
"Sreedev Kodichath <sreedevpadmakumar@gmail.com>",
|
||||
"Valentin Bersier <vbersier@gmail.com>",
|
||||
"Dhruva Sagar <dhruva.sagar@gmail.com>",
|
||||
]
|
||||
|
||||
[[bin]]
|
||||
name = "deduplicator"
|
||||
path = "src/main.rs"
|
||||
|
||||
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
|
||||
|
||||
[dependencies]
|
||||
anyhow = "1.0.68"
|
||||
bytesize = "2.0.1"
|
||||
chrono = "0.4.23"
|
||||
clap = { version = "4.0.32", features = ["derive"] }
|
||||
colored = "2.0.0"
|
||||
crossterm = "0.25.0"
|
||||
fxhash = "0.2.1"
|
||||
glob = "0.3.0"
|
||||
humansize = "2.1.2"
|
||||
itertools = "0.10.5"
|
||||
dashmap = { version = "6.1.0", features = ["rayon"] }
|
||||
globwalk = "0.9.1"
|
||||
gxhash = { version = "3.4.1", default-features = false }
|
||||
indicatif = { version = "0.18.0", features = ["rayon"] }
|
||||
memmap2 = "0.9.7"
|
||||
pathdiff = "0.2.1"
|
||||
prettytable-rs = "0.10.0"
|
||||
rand = "0.9.1"
|
||||
rayon = "1.6.1"
|
||||
sqlite = "0.30.3"
|
||||
thiserror = "1.0.38"
|
||||
tokio = { version = "1.23.0", features = ["full"] }
|
||||
tui = "0.19.0"
|
||||
threadpool = "1.8.1"
|
||||
|
||||
[profile.release]
|
||||
strip = true
|
||||
opt-level = 3
|
||||
lto = "thin"
|
||||
debug = false
|
||||
codegen-units = 1
|
||||
|
||||
# generated by 'cargo dist init'
|
||||
[profile.dist]
|
||||
inherits = "release"
|
||||
|
||||
[workspace.metadata.dist]
|
||||
rust-toolchain-version = "1.87.0"
|
||||
ci = ["github"]
|
||||
targets = [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"x86_64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc",
|
||||
"aarch64-apple-darwin",
|
||||
]
|
||||
cargo-dist-version = "0.0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.20.0"
|
||||
|
||||
201
README.md
201
README.md
@@ -4,42 +4,193 @@
|
||||
Find, Sort, Filter & Delete duplicate files
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
NOTE: This project is still being developed. At the moment, as shown in the screenshot below, deduplicator is able to scan through and list duplicates with and without caching. Contributions are welcome.
|
||||
</p>
|
||||
|
||||
<h2 align="center">Usage</h2>
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
Usage: deduplicator [OPTIONS]
|
||||
find,filter and delete duplicate files
|
||||
|
||||
Usage: deduplicator [OPTIONS] [scan_dir_path]
|
||||
|
||||
Arguments:
|
||||
[scan_dir_path] Run Deduplicator on dir different from pwd (e.g., ~/Pictures )
|
||||
|
||||
Options:
|
||||
-t, --types <TYPES> Filetypes to deduplicate (default = all)
|
||||
--dir <DIR> Run Deduplicator on dir different from pwd
|
||||
-n, --nocache Don't use cache for indexing files (default = true)
|
||||
-h, --help Print help information
|
||||
-V, --version Print version information
|
||||
-T, --exclude-types <EXCLUDE_TYPES> Exclude Filetypes [default = none]
|
||||
-t, --types <TYPES> Filetypes to deduplicate [default = all]
|
||||
-i, --interactive Delete files interactively
|
||||
-m, --min-size <MIN_SIZE> Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T) [default: 1b]
|
||||
-D, --max-depth <MAX_DEPTH> Max Depth to scan while looking for duplicates
|
||||
-d, --min-depth <MIN_DEPTH> Min Depth to scan while looking for duplicates
|
||||
-f, --follow-links Follow links while scanning directories
|
||||
-s, --strict Guarantees that two files are duplicate (performs a full hash)
|
||||
-p, --progress Show Progress spinners & metrics
|
||||
-h, --help Print help
|
||||
-V, --version Print version
|
||||
```
|
||||
### Examples
|
||||
|
||||
```bash
|
||||
# Scan for duplicates recursively from the current dir, only look for png, jpg & pdf file types & interactively delete files
|
||||
deduplicator -t pdf,jpg,png -i
|
||||
|
||||
# Scan for duplicates recursively from current dir, exclude png and jpg file types
|
||||
deduplicator -T jpg,png
|
||||
|
||||
# Scan for duplicates recursively from the ~/Pictures dir, only look for png, jpeg, jpg & pdf file types & interactively delete files
|
||||
deduplicator ~/Pictures/ -t png,jpeg,jpg,pdf -i
|
||||
|
||||
# Scan for duplicates in the ~/Pictures without recursing into subdirectories
|
||||
deduplicator ~/Pictures --max-depth 0
|
||||
|
||||
# look for duplicates in the ~/.config directory while also recursing into symbolic link paths
|
||||
deduplicator ~/.config --follow-links
|
||||
|
||||
# scan for duplicates that are greater than 100mb in the ~/Media directory
|
||||
deduplicator ~/Media --min-size 100mb
|
||||
```
|
||||
|
||||
<h2 align="center">Installation</h2>
|
||||
## Demo
|
||||

|
||||
|
||||
<p align="center">Currently, deduplicator is only installable via rust's cargo package manager</p>
|
||||
|
||||
|
||||
## Installation
|
||||
Currently, you can only install deduplicator using cargo package manager.
|
||||
|
||||
### Cargo
|
||||
> GxHash relies on aes hardware acceleration, so please set `RUSTFLAGS` to `"-C target-feature=+aes"` or `"-C target-cpu=native"` before
|
||||
> installing.
|
||||
|
||||
#### install from crates.io (stable)
|
||||
|
||||
```bash
|
||||
$ RUSTFLAGS="-C target-cpu=native" cargo install deduplicator
|
||||
|
||||
# or
|
||||
|
||||
$ RUSTFLAGS="-C target-feature=+aes,+sse2" cargo install deduplicator
|
||||
```
|
||||
cargo install deduplicator
|
||||
|
||||
#### install from git (nightly)
|
||||
```bash
|
||||
$ RUSTFLAGS="-C target-cpu=native" cargo install deduplicator --git https://github.com/sreedevk/deduplicator
|
||||
|
||||
# or
|
||||
|
||||
$ RUSTFLAGS="-C target-feature=+aes,+sse2" cargo install --git https://github.com/sreedevk/deduplicator
|
||||
```
|
||||
<p align="center">
|
||||
note that if you use a version manager to install rust (like asdf), you need to reshim (`asdf reshim rust`).
|
||||
</p>
|
||||
|
||||
<h2 align="center">Performance</h2>
|
||||
## Performance
|
||||
Deduplicator uses size comparison and [GxHash](https://docs.rs/gxhash/latest/gxhash/) to quickly check a large number of files to find duplicates. its also heavily parallelized. The default behavior of deduplicator is to only hash the first page (4K) of the file. This is to ensure that performance is the default priority. You can modify this behavior by using the `--strict` flag which will hash the whole file and ensure that 2 files are indeed duplicates. I'll add benchmarks in future versions.
|
||||
|
||||
<p align="center">
|
||||
Deduplicator uses fxhash (a non-cryptographic hashing algorithm) which is extremely fast. As a result, deduplicator is able to process huge amounts of data in a couple of seconds.</p>
|
||||
### Benchmarks
|
||||
I've used hyperfine to run deduplicator on files generated by the rake file at `rakelib/benchmark.rake`. The Benchmarking accuracy can further be improved by isolating runs inside restricted docker containers. I'll include that in the future. For now, here's the hyperfine output on my i7-12800H laptop with 32G of RAM.
|
||||
|
||||
<p align="center">
|
||||
While testing, Deduplicator was able to go through 8.6GB of pdf files and detect duplicates in 2.9 seconds
|
||||
</p>
|
||||
<h2 align="center">Screenshots</h2>
|
||||
#### Fewer Large Files
|
||||
```
|
||||
# hyperfine -N --warmup 80 './target/release/deduplicator bench_artifacts'
|
||||
Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
Time (mean ± σ): 2.2 ms ± 0.4 ms [User: 2.2 ms, System: 4.4 ms]
|
||||
Range (min … max): 1.3 ms … 7.1 ms 1522 runs
|
||||
|
||||

|
||||
dust 'bench_artifacts'
|
||||
|
||||
54M ┌── file_0_fwds.bin │████ │ 2%
|
||||
122M ├── file_1_fwds.bin │████████ │ 5%
|
||||
390M ├── file_0_fwdcbss.bin│██████████████████████████ │ 15%
|
||||
390M ├── file_0_fwscas.bin │██████████████████████████ │ 15%
|
||||
390M ├── file_0_fwss.bin │██████████████████████████ │ 15%
|
||||
390M ├── file_1_fwdcbss.bin│██████████████████████████ │ 15%
|
||||
390M ├── file_1_fwscas.bin │██████████████████████████ │ 15%
|
||||
390M ├── file_1_fwss.bin │██████████████████████████ │ 15%
|
||||
2.5G ┌─┴ bench_artifacts │██████████████████████████████████████████████████████████████████ │ 100%
|
||||
```
|
||||
|
||||
#### Many Small Files
|
||||
```
|
||||
# hyperfine --warmup 20 './target/release/deduplicator bench_artifacts'
|
||||
Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
Time (mean ± σ): 40.1 ms ± 2.3 ms [User: 251.0 ms, System: 277.3 ms]
|
||||
Range (min … max): 35.0 ms … 45.9 ms 72 runs
|
||||
|
||||
dust 'bench_artifacts'
|
||||
3.9M ┌── file_992_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_992_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_993_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_994_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_995_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_996_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_997_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_998_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_999_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwss.bin │█ │ 0%
|
||||
11G ┌─┴ bench_artifacts │████████████████████████████████████████████████████████████████ │ 100%
|
||||
```
|
||||
|
||||
## proposed
|
||||
- [ ] parallelization
|
||||
- [ ] scanning + processing sw + processing hw + formatting + printing
|
||||
- [ ] max file path size should use the last set of duplicates
|
||||
- [ ] add more unit tests
|
||||
- [ ] test against different filesystems
|
||||
- [ ] test against different file name encodings
|
||||
- [ ] restore json output (was removed in 0.3 due to quality issues)
|
||||
- [ ] fix memory leak on very large filesystems
|
||||
- [ ] maybe use a bloom filter
|
||||
- [ ] reduce FileInfo size
|
||||
- [ ] tui
|
||||
- [ ] change the default hashing method to include the first & last page of a file (8K)
|
||||
- [ ] provide option to localize duplicate detection to arbitrary levels relative to current directory
|
||||
- [ ] bulk operations
|
||||
- [ ] --keep-latest
|
||||
- [ ] --keep-oldest
|
||||
- [ ] --keep-last-modified
|
||||
- [ ] --keep-first-modified
|
||||
|
||||
- [ ] fix: partial hash collision - a file full of null bytes ("\0") and an empty file. This is a known trade off in gxhash.
|
||||
- [ ] include initial pages and final pages of the file
|
||||
- [ ] append the offset between the last initial page hashed and the first final page hashed in the content passed to the hasher.
|
||||
|
||||
## v0.3.1
|
||||
- [x] parallelization
|
||||
- [x] (scanning + processing sw + processing hw) & formatting & printing
|
||||
- [x] remove formatting step and write directly to stdout
|
||||
- [x] simplify output to improve performance
|
||||
- [x] increase the number of pages hashed in partial hashing
|
||||
- [x] updated dependencies
|
||||
- [x] fix: full file hash collision between a file full of null bytes ("\0") and an empty file. This is a known trade off in gxhash.
|
||||
- [x] appending the file size at the end of content before hashing.
|
||||
- [x] created an automated release pipeline
|
||||
|
||||
## v0.3.0
|
||||
- [x] parallelization
|
||||
- [x] (scanning) + (processing sw & processing hw & formatting & printing)
|
||||
- [x] reduce cloning values on the heap
|
||||
- [x] add a partial hashing mode (--strict)
|
||||
- [x] add unit tests
|
||||
- [x] add silent mode
|
||||
- [x] update documentation
|
||||
- [x] remove color output
|
||||
- [x] progress bar improvements
|
||||
- [x] use progress bar groups
|
||||
- [x] remove broken json rendering
|
||||
- [x] add benchmarks
|
||||
|
||||
102
rakelib/benchmark.rake
Normal file
102
rakelib/benchmark.rake
Normal file
@@ -0,0 +1,102 @@
|
||||
require 'tempfile'
|
||||
require 'fileutils'
|
||||
require 'securerandom'
|
||||
|
||||
namespace :benchmark do
|
||||
file 'target/release/deduplicator' do
|
||||
sh "cargo build --release"
|
||||
end
|
||||
|
||||
task :few_large_files => 'target/release/deduplicator' do
|
||||
# Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
# Time (mean ± σ): 2.5 ms ± 0.6 ms [User: 2.4 ms, System: 4.8 ms]
|
||||
# Range (min … max): 1.8 ms … 9.7 ms 1474 runs
|
||||
|
||||
root = "bench_artifacts"
|
||||
FileUtils.rm_rf(root)
|
||||
Dir.mkdir(root)
|
||||
|
||||
# files with same size
|
||||
puts "generating files of same size ..."
|
||||
2.times.map do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 100_000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different sizes
|
||||
puts "generating files of different sizes ..."
|
||||
2.times.map do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwds.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * (rand * 100_000).ceil))
|
||||
end
|
||||
end
|
||||
|
||||
# files with same content & size
|
||||
puts "generating files of same content and sizes ..."
|
||||
2.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwscas.bin"), 'wb') do |f|
|
||||
f.write("\0" * (4096 * 100_000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different content but same size
|
||||
puts "generating files of different content but same sizes ..."
|
||||
2.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwdcbss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 100_000))
|
||||
end
|
||||
end
|
||||
|
||||
sh("hyperfine -N --warmup 80 './target/release/deduplicator #{root}'")
|
||||
sh("dust '#{root}'")
|
||||
|
||||
FileUtils.rm_rf(root)
|
||||
end
|
||||
|
||||
task :many_small_files => 'target/release/deduplicator' do
|
||||
# Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
# Time (mean ± σ): 10.6 ms ± 1.0 ms [User: 20.0 ms, System: 22.5 ms]
|
||||
# Range (min … max): 8.4 ms … 14.2 ms 235 runs
|
||||
|
||||
root = "bench_artifacts"
|
||||
Dir.mkdir(root)
|
||||
|
||||
# files with same size
|
||||
puts "generating 1000 files of the same size ... "
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 1000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different sizes
|
||||
puts "generating 1000 files of different sizes ... "
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwds.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * (rand * 100).ceil))
|
||||
end
|
||||
end
|
||||
|
||||
# files with same content & size
|
||||
puts "generating files of same content and sizes ..."
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwscas.bin"), 'wb') do |f|
|
||||
f.write("\0" * (4096 * 1000))
|
||||
end
|
||||
end
|
||||
|
||||
# files with different content but same size
|
||||
puts "generating files of different content but same sizes ..."
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwdcbss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 1000))
|
||||
end
|
||||
end
|
||||
|
||||
sh("hyperfine --warmup 20 './target/release/deduplicator #{root}'")
|
||||
sh("dust '#{root}'")
|
||||
|
||||
FileUtils.rm_rf(root)
|
||||
end
|
||||
end
|
||||
@@ -1,28 +0,0 @@
|
||||
use std::time::Duration;
|
||||
|
||||
use crossterm::event::{self, KeyCode, KeyEvent};
|
||||
use anyhow::Result;
|
||||
use super::events;
|
||||
|
||||
pub struct EventHandler;
|
||||
|
||||
impl EventHandler {
|
||||
pub fn init() -> Result<events::Event> {
|
||||
if crossterm::event::poll(Duration::from_millis(10))? {
|
||||
match event::read()? {
|
||||
event::Event::Key(keycode) => Self::handle_keypress(keycode),
|
||||
_ => Ok(events::Event::Noop),
|
||||
}
|
||||
} else {
|
||||
Ok(events::Event::Noop)
|
||||
}
|
||||
}
|
||||
|
||||
fn handle_keypress(keyevent: KeyEvent) -> Result<events::Event> {
|
||||
match keyevent.code {
|
||||
KeyCode::Char('q') => Ok(events::Event::Exit),
|
||||
_ => Ok(events::Event::Noop)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +0,0 @@
|
||||
pub enum Event {
|
||||
Exit,
|
||||
Noop
|
||||
}
|
||||
@@ -1,80 +0,0 @@
|
||||
mod event_handler;
|
||||
mod events;
|
||||
mod ui;
|
||||
mod formatter;
|
||||
|
||||
use crate::database;
|
||||
use crate::output;
|
||||
use crate::params::Params;
|
||||
use crate::scanner;
|
||||
use anyhow::{anyhow, Result};
|
||||
use crossterm::{event, execute, terminal};
|
||||
use event_handler::EventHandler;
|
||||
use std::io;
|
||||
use std::thread;
|
||||
use std::time::Duration;
|
||||
use tui::{
|
||||
backend::CrosstermBackend,
|
||||
widgets::{Block, Borders, Widget},
|
||||
Terminal,
|
||||
};
|
||||
use ui::Ui;
|
||||
|
||||
pub struct App;
|
||||
|
||||
impl App {
|
||||
pub fn init(app_args: &Params) -> Result<()> {
|
||||
// let mut term = Self::init_terminal()?;
|
||||
|
||||
let connection = database::get_connection(&app_args)?;
|
||||
let duplicates = scanner::duplicates(&app_args, &connection)?;
|
||||
|
||||
// Self::init_render_loop(&mut term)?;
|
||||
// Self::cleanup(&mut term)?;
|
||||
|
||||
output::print(duplicates, &app_args); /* TODO: APP TUI INIT FUNCTION */
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn cleanup(term: &mut Terminal<CrosstermBackend<io::Stdout>>) -> Result<()> {
|
||||
terminal::disable_raw_mode()?;
|
||||
execute!(
|
||||
term.backend_mut(),
|
||||
terminal::LeaveAlternateScreen,
|
||||
event::DisableMouseCapture
|
||||
)?;
|
||||
|
||||
term.show_cursor()?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn render_cycle(term: &mut Terminal<CrosstermBackend<io::Stdout>>) -> Result<()> {
|
||||
match EventHandler::init()? {
|
||||
events::Event::Noop => Ui::render_frame(term),
|
||||
events::Event::Exit => Err(anyhow!("Exit")),
|
||||
}
|
||||
}
|
||||
|
||||
fn init_render_loop(term: &mut Terminal<CrosstermBackend<io::Stdout>>) -> Result<()> {
|
||||
loop {
|
||||
match Self::render_cycle(term) {
|
||||
Ok(_) => continue,
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn init_terminal() -> Result<Terminal<CrosstermBackend<io::Stdout>>> {
|
||||
terminal::enable_raw_mode()?;
|
||||
let mut stdout = io::stdout();
|
||||
execute!(
|
||||
stdout,
|
||||
terminal::EnterAlternateScreen,
|
||||
event::EnableMouseCapture
|
||||
)?;
|
||||
let backend = CrosstermBackend::new(stdout);
|
||||
Ok(Terminal::new(backend)?)
|
||||
}
|
||||
}
|
||||
@@ -1,53 +0,0 @@
|
||||
use anyhow::Result;
|
||||
use std::io;
|
||||
use tui::{
|
||||
backend::{Backend, CrosstermBackend},
|
||||
layout::{Constraint, Direction, Layout, Rect},
|
||||
style::{Modifier, Style},
|
||||
text::{Span, Spans},
|
||||
widgets::{Block, Borders, List, ListItem, Widget},
|
||||
Frame, Terminal,
|
||||
};
|
||||
|
||||
pub struct Ui;
|
||||
|
||||
impl Ui {
|
||||
fn generate_file_list() -> impl Widget {
|
||||
let tasks: Vec<ListItem> = vec!["Sreedev"; 100]
|
||||
.into_iter()
|
||||
.map(|item| ListItem::new(vec![Spans::from(Span::raw(item))]))
|
||||
.collect();
|
||||
|
||||
List::new(tasks)
|
||||
.block(Block::default().borders(Borders::ALL).title("List"))
|
||||
.highlight_style(Style::default().add_modifier(Modifier::BOLD))
|
||||
.highlight_symbol("> ")
|
||||
}
|
||||
|
||||
fn generate_info_bar() -> impl Widget {
|
||||
Block::default().title("Description").borders(Borders::ALL)
|
||||
}
|
||||
|
||||
fn generate_file_desc() -> impl Widget {
|
||||
Block::default().title("Description").borders(Borders::ALL)
|
||||
}
|
||||
|
||||
pub fn render_frame(term: &mut Terminal<CrosstermBackend<io::Stdout>>) -> Result<()> {
|
||||
term.draw(|f| {
|
||||
let windows = Layout::default()
|
||||
.direction(Direction::Vertical)
|
||||
.constraints([Constraint::Ratio(2, 16), Constraint::Ratio(14, 16)].as_ref())
|
||||
.split(f.size());
|
||||
|
||||
let subwindows = Layout::default()
|
||||
.direction(Direction::Horizontal)
|
||||
.constraints([Constraint::Ratio(1, 4), Constraint::Ratio(3, 4)].as_ref())
|
||||
.split(windows[1]);
|
||||
|
||||
f.render_widget(Self::generate_info_bar(), windows[0]);
|
||||
f.render_widget(Self::generate_file_list(), subwindows[0]);
|
||||
f.render_widget(Self::generate_file_desc(), subwindows[1]);
|
||||
})?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -1,83 +0,0 @@
|
||||
use anyhow::Result;
|
||||
use crate::params::Params;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct File {
|
||||
pub path: String,
|
||||
pub hash: String,
|
||||
}
|
||||
|
||||
pub fn get_connection(args: &Params) -> Result<sqlite::Connection, sqlite::Error> {
|
||||
let connection_url = match args.nocache {
|
||||
false => "/tmp/deduplicator.db",
|
||||
true => ":memory:"
|
||||
};
|
||||
|
||||
sqlite::open(connection_url).and_then(|conn| {
|
||||
setup(&conn).ok();
|
||||
Ok(conn)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn setup(connection: &sqlite::Connection) -> Result<()> {
|
||||
let query = "CREATE TABLE files (file_identifier STRING, hash STRING)";
|
||||
connection.execute(query).ok();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn put(file: &File, connection: &sqlite::Connection) -> Result<()> {
|
||||
let query = format!(
|
||||
"INSERT INTO files (file_identifier, hash) VALUES (\"{}\", \"{}\")",
|
||||
file.path, file.hash
|
||||
);
|
||||
let result = connection.execute(query)?;
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
pub fn indexed_paths(connection: &sqlite::Connection) -> Result<Vec<File>> {
|
||||
let query = format!(
|
||||
"SELECT * FROM files"
|
||||
);
|
||||
|
||||
let result: Vec<File> = connection
|
||||
.prepare(query)?
|
||||
.into_iter()
|
||||
.map(|row_result| row_result.unwrap())
|
||||
.map(|row| {
|
||||
let path = row.read::<&str, _>("file_identifier").to_string();
|
||||
let hash = row.read::<i64, _>("hash").to_string();
|
||||
File { path, hash }
|
||||
})
|
||||
.collect();
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
pub fn duplicate_hashes(connection: &sqlite::Connection, path: &String) -> Result<Vec<File>> {
|
||||
let query = format!(
|
||||
"
|
||||
SELECT a.* FROM files a
|
||||
JOIN (SELECT file_identifier, hash, COUNT(*)
|
||||
FROM files
|
||||
GROUP BY hash
|
||||
HAVING count(*) > 1 ) b
|
||||
ON a.hash = b.hash
|
||||
WHERE a.file_identifier LIKE \"{}%\"
|
||||
ORDER BY a.file_identifier
|
||||
", path
|
||||
);
|
||||
|
||||
let result: Vec<File> = connection
|
||||
.prepare(query)?
|
||||
.into_iter()
|
||||
.map(|row_result| row_result.unwrap())
|
||||
.map(|row| {
|
||||
let path = row.read::<&str, _>("file_identifier").to_string();
|
||||
let hash = row.read::<i64, _>("hash").to_string();
|
||||
File { path, hash }
|
||||
})
|
||||
.collect();
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
104
src/fileinfo.rs
Normal file
104
src/fileinfo.rs
Normal file
@@ -0,0 +1,104 @@
|
||||
use anyhow::Result;
|
||||
use gxhash::gxhash128;
|
||||
use memmap2::Mmap;
|
||||
use std::{
|
||||
fs,
|
||||
io::Read,
|
||||
path::{Path, PathBuf},
|
||||
sync::{Arc, Mutex},
|
||||
time::SystemTime,
|
||||
};
|
||||
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum FileState {
|
||||
Unprocessed,
|
||||
SwProcessed,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct FileInfo {
|
||||
pub path: Box<Path>,
|
||||
pub size: u64,
|
||||
pub modified: SystemTime,
|
||||
pub state: Arc<Mutex<FileState>>,
|
||||
}
|
||||
|
||||
impl FileInfo {
|
||||
pub fn hash(&self, seed: i64) -> Result<u128> {
|
||||
if self.size == 0 {
|
||||
return Ok(0u128);
|
||||
};
|
||||
|
||||
let file = fs::File::open(&self.path)?;
|
||||
let mapper = unsafe { Mmap::map(&file)? };
|
||||
let content_hash = mapper
|
||||
.chunks(4096)
|
||||
.fold(0u128, |acc, chunk: &[u8]| acc ^ gxhash128(chunk, seed));
|
||||
|
||||
// NOTE: avoids collision bw an empty file & a file full of null bytes.
|
||||
Ok(content_hash ^ gxhash128(&self.size.to_ne_bytes(), seed))
|
||||
}
|
||||
|
||||
pub fn initpages_hash(&self, seed: i64) -> Result<u128> {
|
||||
let mut file = fs::File::open(&self.path)?;
|
||||
let mut buffer = [0; 16384];
|
||||
let bytes_read = file.read(&mut buffer)?;
|
||||
|
||||
Ok(gxhash128(&buffer[..bytes_read], seed))
|
||||
}
|
||||
|
||||
pub fn new(path: PathBuf) -> Result<Self> {
|
||||
let filemeta = std::fs::metadata(&path)?;
|
||||
Ok(Self {
|
||||
path: path.into_boxed_path(),
|
||||
size: filemeta.len(),
|
||||
modified: filemeta.modified()?,
|
||||
state: Arc::new(Mutex::new(FileState::Unprocessed)),
|
||||
})
|
||||
}
|
||||
|
||||
pub fn sw_processed(&self) {
|
||||
let mut self_state = self.state.lock().unwrap();
|
||||
*self_state = FileState::SwProcessed;
|
||||
}
|
||||
|
||||
pub fn is_sw_processed(&self) -> bool {
|
||||
let self_state = self.state.lock().unwrap();
|
||||
*self_state == FileState::SwProcessed
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod test {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
use std::fs::File;
|
||||
use std::io::Write;
|
||||
use anyhow::Result;
|
||||
|
||||
fn generate_null_bytes(size: usize) -> Vec<u8> {
|
||||
(0..size).map(|_| 0).collect::<Vec<u8>>()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hash_differentiates_between_a_file_of_null_bytes_vs_an_empty_file() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let empty_file_name = root.path().join("empty_file.bin");
|
||||
|
||||
File::create_new(&empty_file_name)?;
|
||||
|
||||
let file_with_null_bytes_name = root.path().join("file_with_null_bytes.bin");
|
||||
let mut file_with_null_bytes = File::create_new(&file_with_null_bytes_name)?;
|
||||
|
||||
file_with_null_bytes.write_all(&generate_null_bytes(1000 * 4096))?;
|
||||
|
||||
let empty_file_info = FileInfo::new(empty_file_name)?;
|
||||
let file_with_empty_bytes_info = FileInfo::new(file_with_null_bytes_name)?;
|
||||
|
||||
let seed: i64 = 246910456374;
|
||||
|
||||
assert_ne!(empty_file_info.hash(seed)?, file_with_empty_bytes_info.hash(seed)?);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
64
src/formatter.rs
Normal file
64
src/formatter.rs
Normal file
@@ -0,0 +1,64 @@
|
||||
use crate::{fileinfo::FileInfo, params::Params};
|
||||
use anyhow::Result;
|
||||
use chrono::{DateTime, Utc};
|
||||
use dashmap::DashMap;
|
||||
use pathdiff::diff_paths;
|
||||
use rayon::prelude::*;
|
||||
use std::{path::PathBuf, sync::Arc};
|
||||
|
||||
const YELLOW: &str = "\x1b[33m";
|
||||
const RESET: &str = "\x1b[0m";
|
||||
|
||||
pub struct Formatter;
|
||||
impl Formatter {
|
||||
pub fn human_path(file: &FileInfo, aargs: &Params, min_path_length: usize) -> Result<String> {
|
||||
let base_directory: PathBuf = aargs.get_directory()?;
|
||||
let relative_path = diff_paths(&file.path, base_directory).unwrap_or_default();
|
||||
|
||||
let formatted_path = format!(
|
||||
"{:<0width$}",
|
||||
relative_path.to_str().unwrap_or_default().to_string(),
|
||||
width = min_path_length
|
||||
);
|
||||
|
||||
Ok(formatted_path)
|
||||
}
|
||||
|
||||
pub fn human_filesize(file: &FileInfo) -> Result<String> {
|
||||
Ok(format!("{:>12}", bytesize::ByteSize::b(file.size)))
|
||||
}
|
||||
|
||||
pub fn human_mtime(file: &FileInfo) -> Result<String> {
|
||||
let modified_time: DateTime<Utc> = file.modified.into();
|
||||
Ok(modified_time.format("%Y-%m-%d %H:%M:%S").to_string())
|
||||
}
|
||||
|
||||
pub fn print(raw: Arc<DashMap<u128, Vec<FileInfo>>>, max_path_len: u64, aargs: &Params) {
|
||||
print!("{}", "\n".repeat(if aargs.progress { 2 } else { 1 })); // spacing
|
||||
|
||||
if raw.is_empty() {
|
||||
println!("No duplicates found matching your search criteria.");
|
||||
} else {
|
||||
raw.par_iter().for_each(|sref| {
|
||||
let mut ostring = format!("{}{:32x}{}\n", YELLOW, sref.key(), RESET);
|
||||
let subfields = sref
|
||||
.value()
|
||||
.par_iter()
|
||||
.map(|finfo| {
|
||||
format!(
|
||||
"├─ {}\t{}\t{}\n",
|
||||
Self::human_path(finfo, aargs, max_path_len as usize)
|
||||
.expect("path formatting failed."),
|
||||
Self::human_filesize(finfo).expect("filesize formatting failed."),
|
||||
Self::human_mtime(finfo).expect("modified time formatting failed.")
|
||||
)
|
||||
})
|
||||
.collect::<String>();
|
||||
|
||||
ostring.push_str(&subfields);
|
||||
|
||||
println!("{ostring}");
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
125
src/interactive.rs
Normal file
125
src/interactive.rs
Normal file
@@ -0,0 +1,125 @@
|
||||
use crate::{fileinfo::FileInfo, formatter::Formatter, params::Params};
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use prettytable::{format, row, Table};
|
||||
use std::{
|
||||
io::{self, Write},
|
||||
sync::Arc,
|
||||
};
|
||||
|
||||
pub struct Interactive;
|
||||
|
||||
impl Interactive {
|
||||
pub fn init(result: Arc<DashMap<u128, Vec<FileInfo>>>, app_args: &Params) -> Result<()> {
|
||||
result
|
||||
.clone()
|
||||
.iter()
|
||||
.filter(|i| i.value().len() > 1)
|
||||
.enumerate()
|
||||
.for_each(|(gindex, i)| {
|
||||
let group = i.value();
|
||||
let mut itable = Table::new();
|
||||
itable.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
|
||||
itable.set_titles(row!["index", "filename", "size", "updated_at"]);
|
||||
|
||||
let max_path_size = group
|
||||
.iter()
|
||||
.map(|f| f.path.iter().count())
|
||||
.max()
|
||||
.unwrap_or_default();
|
||||
|
||||
group.iter().enumerate().for_each(|(index, file)| {
|
||||
itable.add_row(row![
|
||||
index,
|
||||
Formatter::human_path(file, app_args, max_path_size).unwrap_or_default(),
|
||||
Formatter::human_filesize(file).unwrap_or_default(),
|
||||
Formatter::human_mtime(file).unwrap_or_default()
|
||||
]);
|
||||
});
|
||||
|
||||
Self::process_group_action(group, gindex, result.len(), itable);
|
||||
});
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn scan_group_confirmation() -> Result<bool> {
|
||||
print!("\nconfirm? [y/N]: ");
|
||||
std::io::stdout().flush()?;
|
||||
let mut user_input = String::new();
|
||||
io::stdin().read_line(&mut user_input)?;
|
||||
|
||||
match user_input.trim() {
|
||||
"Y" | "y" => Ok(true),
|
||||
_ => Ok(false),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn scan_group_instruction() -> Result<String> {
|
||||
println!("\nEnter the indices of the files you want to delete.");
|
||||
println!("You can enter multiple files using commas to seperate file indices.");
|
||||
println!("example: 1,2");
|
||||
print!("\n> ");
|
||||
std::io::stdout().flush()?;
|
||||
let mut user_input = String::new();
|
||||
io::stdin().read_line(&mut user_input)?;
|
||||
|
||||
Ok(user_input)
|
||||
}
|
||||
|
||||
pub fn process_group_action(
|
||||
duplicates: &Vec<FileInfo>,
|
||||
dup_index: usize,
|
||||
dup_size: usize,
|
||||
table: Table,
|
||||
) {
|
||||
println!("\nDuplicate Set {} of {}\n", dup_index + 1, dup_size);
|
||||
table.printstd();
|
||||
let files_to_delete = Self::scan_group_instruction().unwrap_or_default();
|
||||
let parsed_file_indices = files_to_delete
|
||||
.trim()
|
||||
.split(',')
|
||||
.filter(|element| !element.is_empty())
|
||||
.map(|index| index.parse::<usize>().unwrap_or_default())
|
||||
.collect::<Vec<usize>>();
|
||||
|
||||
if parsed_file_indices
|
||||
.clone()
|
||||
.into_iter()
|
||||
.any(|index| index > (duplicates.len() - 1))
|
||||
{
|
||||
println!("Err: File Index Out of Bounds!");
|
||||
return Self::process_group_action(duplicates, dup_index, dup_size, table);
|
||||
}
|
||||
|
||||
print!("{esc}[2J{esc}[1;1H", esc = 27 as char);
|
||||
|
||||
if parsed_file_indices.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
let files_to_delete = parsed_file_indices
|
||||
.into_iter()
|
||||
.map(|index| duplicates[index].clone());
|
||||
|
||||
println!("\nThe following files will be deleted:");
|
||||
files_to_delete
|
||||
.clone()
|
||||
.enumerate()
|
||||
.for_each(|(index, file)| {
|
||||
println!("{}: {}", index, file.path.display());
|
||||
});
|
||||
|
||||
match Self::scan_group_confirmation().unwrap() {
|
||||
true => {
|
||||
files_to_delete.into_iter().for_each(|file| {
|
||||
match std::fs::remove_file(file.path.clone()) {
|
||||
Ok(_) => println!("DELETED: {}", file.path.display()),
|
||||
Err(_) => println!("FAILED: {}", file.path.display()),
|
||||
}
|
||||
});
|
||||
}
|
||||
false => println!("\nCancelled Delete Operation."),
|
||||
}
|
||||
}
|
||||
}
|
||||
35
src/main.rs
35
src/main.rs
@@ -1,14 +1,35 @@
|
||||
mod fileinfo;
|
||||
mod formatter;
|
||||
mod interactive;
|
||||
mod params;
|
||||
mod database;
|
||||
mod output;
|
||||
mod processor;
|
||||
mod scanner;
|
||||
mod app;
|
||||
mod server;
|
||||
|
||||
use self::{formatter::Formatter, interactive::Interactive, server::Server};
|
||||
use anyhow::Result;
|
||||
use clap::Parser;
|
||||
use app::App;
|
||||
use params::Params;
|
||||
use std::sync::atomic::Ordering;
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> Result<()> {
|
||||
App::init(¶ms::Params::parse())
|
||||
fn main() -> Result<()> {
|
||||
let app_args = Params::parse();
|
||||
let server = Server::new(app_args.clone());
|
||||
|
||||
server.start()?;
|
||||
|
||||
match app_args.interactive {
|
||||
false => {
|
||||
Formatter::print(
|
||||
server.hw_duplicate_set,
|
||||
server.max_file_path_len.load(Ordering::Acquire),
|
||||
&app_args,
|
||||
);
|
||||
}
|
||||
true => {
|
||||
Interactive::init(server.hw_duplicate_set, &app_args)?;
|
||||
}
|
||||
};
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -1,73 +0,0 @@
|
||||
use std::{collections::HashMap, fs};
|
||||
|
||||
use anyhow::Result;
|
||||
use chrono::offset::Utc;
|
||||
use chrono::DateTime;
|
||||
use colored::Colorize;
|
||||
use humansize::{format_size, DECIMAL};
|
||||
|
||||
use crate::database::File;
|
||||
use crate::params::Params;
|
||||
|
||||
fn format_path(path: &str, opts: &Params) -> Result<String> {
|
||||
let display_path = path.replace(&opts.get_directory()?, "");
|
||||
let text_vec = display_path.chars().collect::<Vec<_>>();
|
||||
|
||||
let display_range = if text_vec.len() > 32 {
|
||||
text_vec[(display_path.len() - 32)..]
|
||||
.iter()
|
||||
.collect::<String>()
|
||||
} else {
|
||||
display_path
|
||||
};
|
||||
|
||||
Ok(format!("...{}", display_range))
|
||||
}
|
||||
|
||||
fn file_size(path: &String) -> Result<String> {
|
||||
let mdata = fs::metadata(path)?;
|
||||
let formatted_size = format_size(mdata.len(), DECIMAL);
|
||||
Ok(formatted_size)
|
||||
}
|
||||
|
||||
fn modified_time(path: &String) -> Result<String> {
|
||||
let mdata = fs::metadata(path)?;
|
||||
let modified_time: DateTime<Utc> = mdata.modified()?.into();
|
||||
|
||||
Ok(modified_time.format("%Y-%m-%d %H:%M:%S").to_string())
|
||||
}
|
||||
|
||||
fn print_divider() {
|
||||
println!("-------------------+-------------------------------------+------------------+----------------------------------+");
|
||||
}
|
||||
|
||||
pub fn print(duplicates: Vec<File>, opts: &Params) {
|
||||
print_divider();
|
||||
println!(
|
||||
"| {0: <16} | {1: <35} | {2: <16} | {3: <32} |",
|
||||
"hash", "filename", "size", "updated_at"
|
||||
);
|
||||
print_divider();
|
||||
|
||||
let mut dup_index: HashMap<String, Vec<File>> = HashMap::new();
|
||||
|
||||
duplicates.into_iter().for_each(|file| {
|
||||
dup_index
|
||||
.entry(file.hash.clone())
|
||||
.and_modify(|value| value.push(file.clone()))
|
||||
.or_insert_with(|| vec![file]);
|
||||
});
|
||||
|
||||
dup_index.iter().for_each(|(_, group)| {
|
||||
group.iter().for_each(|file| {
|
||||
println!(
|
||||
"| {0: <16} | {1: <35} | {2: <16} | {3: <32} |",
|
||||
file.hash.red(),
|
||||
format_path(&file.path, opts).unwrap_or_default().yellow(),
|
||||
file_size(&file.path).unwrap_or_default().blue(),
|
||||
modified_time(&file.path).unwrap_or_default().blue()
|
||||
);
|
||||
});
|
||||
print_divider();
|
||||
});
|
||||
}
|
||||
115
src/params.rs
115
src/params.rs
@@ -1,40 +1,103 @@
|
||||
use std::path::PathBuf;
|
||||
use clap::Parser;
|
||||
use anyhow::Result;
|
||||
use std::fs;
|
||||
use std::{fs, path::PathBuf};
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
use anyhow::Result;
|
||||
use clap::{Parser, ValueHint};
|
||||
use std::collections::HashSet;
|
||||
|
||||
#[derive(Parser, Debug, Default, Clone)]
|
||||
#[command(author, version, about, long_about = None)]
|
||||
pub struct Params {
|
||||
/// Filetypes to deduplicate (default = all)
|
||||
/// Exclude Filetypes [default = none]
|
||||
#[arg(short = 'T', long)]
|
||||
pub exclude_types: Option<String>,
|
||||
/// Filetypes to deduplicate [default = all]
|
||||
#[arg(short, long)]
|
||||
pub types: Option<String>,
|
||||
/// Run Deduplicator on dir different from pwd
|
||||
#[arg(long)]
|
||||
/// Run Deduplicator on dir different from pwd (e.g., ~/Pictures )
|
||||
#[arg(value_hint = ValueHint::DirPath, value_name = "scan_dir_path")]
|
||||
pub dir: Option<PathBuf>,
|
||||
/// Don't use cache for indexing files (default = true)
|
||||
/// Delete files interactively
|
||||
#[arg(long, short)]
|
||||
pub nocache: bool,
|
||||
pub interactive: bool,
|
||||
/// Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T).
|
||||
#[arg(long, short = 'm', default_value = "1b")]
|
||||
pub min_size: Option<String>,
|
||||
/// Max Depth to scan while looking for duplicates
|
||||
#[arg(long, short = 'D')]
|
||||
pub max_depth: Option<usize>,
|
||||
/// Min Depth to scan while looking for duplicates
|
||||
#[arg(long, short = 'd')]
|
||||
pub min_depth: Option<usize>,
|
||||
/// Follow links while scanning directories
|
||||
#[arg(long, short)]
|
||||
pub follow_links: bool,
|
||||
/// Guarantees that two files are duplicate (performs a full hash)
|
||||
#[arg(long, short = 's', default_value = "false")]
|
||||
pub strict: bool,
|
||||
/// Show Progress spinners & metrics
|
||||
#[arg(long, short = 'p', default_value = "false")]
|
||||
pub progress: bool,
|
||||
}
|
||||
|
||||
impl Params {
|
||||
pub fn get_directory(&self) -> Result<String> {
|
||||
let dir_string: String = self
|
||||
.dir
|
||||
.clone()
|
||||
.unwrap_or(std::env::current_dir()?)
|
||||
.as_os_str()
|
||||
.to_str()
|
||||
.unwrap()
|
||||
.to_string();
|
||||
|
||||
let dir_pathbuf = PathBuf::from(&dir_string);
|
||||
let dir = fs::canonicalize(&dir_pathbuf)?
|
||||
.as_os_str()
|
||||
.to_str()
|
||||
.unwrap()
|
||||
.to_string();
|
||||
pub fn get_min_size(&self) -> Option<u64> {
|
||||
match &self.min_size {
|
||||
Some(msize) => match msize.parse::<bytesize::ByteSize>() {
|
||||
Ok(units) => Some(units.0),
|
||||
Err(_) => None,
|
||||
},
|
||||
None => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn get_directory(&self) -> Result<PathBuf> {
|
||||
let current_dir = std::env::current_dir()?;
|
||||
let dir_path = self.dir.as_ref().unwrap_or(¤t_dir).as_path();
|
||||
let dir = fs::canonicalize(dir_path)?;
|
||||
Ok(dir)
|
||||
}
|
||||
|
||||
pub fn types_intersection(itypes: &str, xtypes: &str) -> String {
|
||||
let iset = itypes
|
||||
.split(",")
|
||||
.map(String::from)
|
||||
.collect::<HashSet<String>>();
|
||||
let xset = xtypes
|
||||
.split(",")
|
||||
.map(String::from)
|
||||
.collect::<HashSet<String>>();
|
||||
|
||||
iset.difference(&xset)
|
||||
.cloned()
|
||||
.collect::<Vec<String>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
pub fn get_types(&self) -> Option<String> {
|
||||
match &self.types {
|
||||
Some(itypes) => match &self.exclude_types {
|
||||
Some(xtypes) => Some(Self::types_intersection(itypes, xtypes)),
|
||||
None => Some(itypes.to_string()),
|
||||
},
|
||||
None => self.exclude_types.as_ref().map(|xtypes| xtypes.to_string()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn mixing_include_and_exclude_types_works_as_expected() {
|
||||
let params = Params {
|
||||
types: Some(String::from("js,xml,ts,pdf,tiff")),
|
||||
exclude_types: Some(String::from("js,ts,xml")),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
assert!(params
|
||||
.get_types()
|
||||
.is_some_and(|x| x == "pdf,tiff" || x == "tiff,pdf"))
|
||||
}
|
||||
}
|
||||
|
||||
381
src/processor.rs
Normal file
381
src/processor.rs
Normal file
@@ -0,0 +1,381 @@
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{MultiProgress, ProgressBar, ProgressStyle};
|
||||
use rayon::iter::IntoParallelRefMutIterator;
|
||||
use rayon::prelude::{IntoParallelIterator, ParallelIterator};
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::sync::{Arc, Mutex, TryLockError, TryLockResult};
|
||||
use std::time::Duration;
|
||||
|
||||
use crate::fileinfo::FileInfo;
|
||||
use crate::params::Params;
|
||||
|
||||
pub struct Processor {}
|
||||
|
||||
impl Processor {
|
||||
pub fn hashwise(
|
||||
app_args: Arc<Params>,
|
||||
sw_store: Arc<DashMap<u64, Vec<FileInfo>>>,
|
||||
hw_store: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
progress_bar_box: Arc<MultiProgress>,
|
||||
max_file_size: Arc<AtomicU64>,
|
||||
seed: i64,
|
||||
sw_sorting_finished: Arc<AtomicBool>,
|
||||
) -> Result<()> {
|
||||
let progress_bar = match app_args.progress {
|
||||
true => progress_bar_box.add(ProgressBar::new_spinner()),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("files grouped by hash.");
|
||||
|
||||
loop {
|
||||
let keys: Vec<u64> = sw_store
|
||||
.clone()
|
||||
.iter()
|
||||
.filter(|i| !i.value().iter().all(|x| x.is_sw_processed()))
|
||||
.filter(|i| i.value().len() > 1)
|
||||
.map(|i| *i.key())
|
||||
.collect();
|
||||
|
||||
if keys.is_empty() {
|
||||
match sw_sorting_finished.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
true => {
|
||||
progress_bar.finish_with_message("files grouped by hash.");
|
||||
break Ok(());
|
||||
}
|
||||
false => continue,
|
||||
}
|
||||
} else {
|
||||
keys.into_par_iter().for_each(|key| {
|
||||
let mut group: Vec<FileInfo> = sw_store.get(&key).unwrap().to_vec();
|
||||
if group.len() > 1 {
|
||||
group.par_iter_mut().for_each(|file| {
|
||||
progress_bar.inc(1);
|
||||
file.sw_processed();
|
||||
|
||||
let fhash = match app_args.strict {
|
||||
true => file.hash(seed).expect("hashing file failed."),
|
||||
false => file.initpages_hash(seed).expect("hashing file failed."),
|
||||
};
|
||||
|
||||
Self::compare_and_update_max_path_len(
|
||||
max_file_size.clone(),
|
||||
file.path.to_string_lossy().len() as u64,
|
||||
);
|
||||
|
||||
hw_store
|
||||
.entry(fhash)
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file.clone()]);
|
||||
});
|
||||
};
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn compare_and_update_max_path_len(current: Arc<AtomicU64>, next: u64) {
|
||||
if current.load(Ordering::Relaxed) < next {
|
||||
current.store(next, Ordering::Release);
|
||||
}
|
||||
}
|
||||
|
||||
pub fn sizewise(
|
||||
app_args: Arc<Params>,
|
||||
scanner_finished: Arc<AtomicBool>,
|
||||
store: Arc<DashMap<u64, Vec<FileInfo>>>,
|
||||
files: Arc<Mutex<Vec<FileInfo>>>,
|
||||
progress_bar_box: Arc<MultiProgress>,
|
||||
) -> Result<()> {
|
||||
let progress_bar = match app_args.progress {
|
||||
true => progress_bar_box.add(ProgressBar::new_spinner()),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("files grouped by size");
|
||||
|
||||
loop {
|
||||
let fileopt: Option<FileInfo> = {
|
||||
match files.try_lock() {
|
||||
Ok(mut flist) => flist.pop(),
|
||||
TryLockResult::Err(TryLockError::WouldBlock) => None,
|
||||
_ => None,
|
||||
}
|
||||
};
|
||||
|
||||
match fileopt {
|
||||
Some(file) => {
|
||||
progress_bar.inc(1);
|
||||
store
|
||||
.entry(file.size)
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file]);
|
||||
continue;
|
||||
}
|
||||
None => match scanner_finished.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
true => {
|
||||
progress_bar.finish_with_message("files grouped by size");
|
||||
break Ok(());
|
||||
}
|
||||
false => continue,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::MultiProgress;
|
||||
use rand::Rng;
|
||||
use std::fs::File;
|
||||
use std::io::Write;
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use tempfile::TempDir;
|
||||
|
||||
use crate::{fileinfo::FileInfo, params::Params};
|
||||
|
||||
use super::Processor;
|
||||
|
||||
fn generate_bytes(size: usize) -> Vec<u8> {
|
||||
let mut rng = rand::rng();
|
||||
(0..size).map(|_| rng.random::<u8>()).collect::<Vec<u8>>()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_init_pages_only_strict_mode() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(16384);
|
||||
|
||||
let mut content_x = content.clone();
|
||||
let mut content_y = content.clone();
|
||||
|
||||
content_x.extend(generate_bytes(1720320));
|
||||
content_y.extend(generate_bytes(1720320));
|
||||
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), content_x),
|
||||
(root.path().join("filetwo.bin"), content_y),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let hw_dupstore = Arc::new(DashMap::new());
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
let args = Params {
|
||||
strict: true,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
Processor::hashwise(
|
||||
Arc::new(args),
|
||||
dupstore.clone(),
|
||||
hw_dupstore.clone(),
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 2);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_init_pages_only_fast_mode() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(16384);
|
||||
|
||||
let mut content_x = content.clone();
|
||||
let mut content_y = content.clone();
|
||||
|
||||
content_x.extend(generate_bytes(1720320));
|
||||
content_y.extend(generate_bytes(1720320));
|
||||
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), content_x),
|
||||
(root.path().join("filetwo.bin"), content_y),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let hw_dupstore = Arc::new(DashMap::new());
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
Processor::hashwise(
|
||||
Arc::new(Params::default()),
|
||||
dupstore.clone(),
|
||||
hw_dupstore.clone(),
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 1);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_data() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(282624);
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), content.clone()),
|
||||
(root.path().join("filetwo.bin"), content.clone()),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let hw_dupstore = Arc::new(DashMap::new());
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
Processor::hashwise(
|
||||
Arc::new(Params::default()),
|
||||
dupstore.clone(),
|
||||
hw_dupstore.clone(),
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 1);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sizewise_sorting_two_files_of_different_sizes() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), generate_bytes(282624)),
|
||||
(root.path().join("filetwo.bin"), generate_bytes(1720320)),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
assert_eq!(dupstore.len(), 2);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sizewise_sorting_two_files_of_same_size() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let files = [
|
||||
(root.path().join("fileone.bin"), generate_bytes(282624)),
|
||||
(root.path().join("filetwo.bin"), generate_bytes(282624)),
|
||||
];
|
||||
|
||||
for (fpath, content) in files.iter() {
|
||||
let mut f = File::create_new(fpath)?;
|
||||
f.write_all(content)?;
|
||||
}
|
||||
|
||||
let file_queue = Arc::new(Mutex::new(
|
||||
files
|
||||
.iter()
|
||||
.map(|f| FileInfo::new(f.0.clone()).unwrap())
|
||||
.collect::<Vec<FileInfo>>(),
|
||||
));
|
||||
|
||||
let dupstore = Arc::new(DashMap::new());
|
||||
|
||||
Processor::sizewise(
|
||||
Arc::new(Params::default()),
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
dupstore.clone(),
|
||||
file_queue,
|
||||
Arc::new(MultiProgress::new()),
|
||||
)?;
|
||||
|
||||
assert_eq!(dupstore.len(), 1);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
165
src/scanner.rs
165
src/scanner.rs
@@ -1,85 +1,100 @@
|
||||
use std::{fs, path::PathBuf};
|
||||
|
||||
use crate::{fileinfo::FileInfo, params::Params};
|
||||
use anyhow::Result;
|
||||
use fxhash::hash32 as hasher;
|
||||
use glob::glob;
|
||||
use itertools::Itertools;
|
||||
use rayon::prelude::*;
|
||||
use indicatif::{MultiProgress, ProgressBar, ProgressStyle};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::{path::Path, time::Duration};
|
||||
|
||||
use crate::{
|
||||
database::{self, File},
|
||||
params::Params,
|
||||
};
|
||||
use globwalk::{GlobWalker, GlobWalkerBuilder};
|
||||
|
||||
pub fn duplicates(app_opts: &Params, connection: &sqlite::Connection) -> Result<Vec<File>> {
|
||||
let scan_results = scan(app_opts, connection)?;
|
||||
let base_path = app_opts.get_directory()?;
|
||||
|
||||
index_files(scan_results, connection)?;
|
||||
database::duplicate_hashes(connection, &base_path)
|
||||
pub struct Scanner {
|
||||
pub directory: Box<Path>,
|
||||
pub filetypes: Option<String>,
|
||||
pub min_depth: Option<usize>,
|
||||
pub max_depth: Option<usize>,
|
||||
pub min_size: Option<u64>,
|
||||
pub follow_links: bool,
|
||||
pub progress: bool,
|
||||
}
|
||||
|
||||
fn get_glob_patterns(opts: &Params, directory: &str) -> Vec<PathBuf> {
|
||||
opts.types
|
||||
.clone()
|
||||
.unwrap_or_else(|| String::from("*"))
|
||||
.split(',')
|
||||
.map(|filetype| format!("*.{}", filetype))
|
||||
.map(|filetype| {
|
||||
vec![directory.to_owned(), String::from("**"), filetype]
|
||||
.iter()
|
||||
.collect()
|
||||
impl Scanner {
|
||||
pub fn new(app_args: Arc<Params>) -> Result<Self> {
|
||||
Ok(Self {
|
||||
directory: app_args.get_directory()?.into_boxed_path(),
|
||||
filetypes: app_args.get_types(),
|
||||
min_depth: app_args.min_depth,
|
||||
max_depth: app_args.max_depth,
|
||||
min_size: app_args.get_min_size(),
|
||||
follow_links: app_args.follow_links,
|
||||
progress: app_args.progress,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn is_indexed_file(path: impl Into<String>, indexed: &[File]) -> bool {
|
||||
indexed
|
||||
.iter()
|
||||
.map(|file| file.path.clone())
|
||||
.contains(&path.into())
|
||||
}
|
||||
|
||||
fn scan(app_opts: &Params, connection: &sqlite::Connection) -> Result<Vec<String>> {
|
||||
let directory = app_opts.get_directory()?;
|
||||
let glob_patterns: Vec<PathBuf> = get_glob_patterns(app_opts, &directory);
|
||||
let indexed_paths = database::indexed_paths(connection)?;
|
||||
let files: Vec<String> = glob_patterns
|
||||
.par_iter()
|
||||
.filter_map(|glob_pattern| glob(glob_pattern.as_os_str().to_str()?).ok())
|
||||
.flat_map(|file_vec| {
|
||||
file_vec
|
||||
.filter_map(|x| Some(x.ok()?.as_os_str().to_str()?.to_string()))
|
||||
.filter(|fpath| !is_indexed_file(fpath, &indexed_paths))
|
||||
.filter(|glob_result| {
|
||||
fs::metadata(glob_result)
|
||||
.map(|f| f.is_file())
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect::<Vec<String>>()
|
||||
fn scan_patterns(&self) -> Result<String> {
|
||||
Ok(match &self.filetypes {
|
||||
Some(ftypes) => format!("**/*{{{ftypes}}}"),
|
||||
None => "**/*".to_string(),
|
||||
})
|
||||
.collect();
|
||||
}
|
||||
|
||||
Ok(files)
|
||||
}
|
||||
|
||||
fn index_files(files: Vec<String>, connection: &sqlite::Connection) -> Result<()> {
|
||||
let hashed: Vec<File> = files
|
||||
.into_par_iter()
|
||||
.filter_map(|file| {
|
||||
let hash = hash_file(&file).ok()?;
|
||||
Some(database::File { path: file, hash })
|
||||
})
|
||||
.collect();
|
||||
|
||||
hashed
|
||||
.iter()
|
||||
.try_for_each(|file| database::put(file, connection))
|
||||
}
|
||||
|
||||
pub fn hash_file(filepath: &str) -> Result<String> {
|
||||
let file = fs::read(filepath)?;
|
||||
let hash = hasher(&*file).to_string();
|
||||
|
||||
Ok(hash)
|
||||
fn attach_link_opts(&self, walker: GlobWalkerBuilder) -> Result<GlobWalkerBuilder> {
|
||||
Ok(walker.follow_links(self.follow_links))
|
||||
}
|
||||
|
||||
fn attach_walker_min_depth(&self, walker: GlobWalkerBuilder) -> Result<GlobWalkerBuilder> {
|
||||
match self.min_depth {
|
||||
Some(min_depth) => Ok(walker.min_depth(min_depth)),
|
||||
None => Ok(walker),
|
||||
}
|
||||
}
|
||||
|
||||
fn attach_walker_max_depth(&self, walker: GlobWalkerBuilder) -> Result<GlobWalkerBuilder> {
|
||||
match self.max_depth {
|
||||
Some(max_depth) => Ok(walker.max_depth(max_depth)),
|
||||
None => Ok(walker),
|
||||
}
|
||||
}
|
||||
fn build_walker(&self) -> Result<GlobWalker> {
|
||||
let walker = Ok(GlobWalkerBuilder::from_patterns(
|
||||
self.directory.clone(),
|
||||
&[self.scan_patterns()?],
|
||||
))
|
||||
.and_then(|walker| self.attach_walker_min_depth(walker))
|
||||
.and_then(|walker| self.attach_walker_max_depth(walker))
|
||||
.and_then(|walker| self.attach_link_opts(walker))?;
|
||||
|
||||
Ok(walker.build()?)
|
||||
}
|
||||
|
||||
pub fn scan(
|
||||
&self,
|
||||
files: Arc<Mutex<Vec<FileInfo>>>,
|
||||
progress_bar_box: Arc<MultiProgress>,
|
||||
) -> Result<()> {
|
||||
let progress_bar = match self.progress {
|
||||
true => progress_bar_box.add(ProgressBar::new_spinner()),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("paths mapped");
|
||||
let min_size = self.min_size.unwrap_or_default();
|
||||
|
||||
self.build_walker()?
|
||||
.filter_map(Result::ok)
|
||||
.map(|entity| entity.into_path())
|
||||
.inspect(|_path| progress_bar.inc(1))
|
||||
.filter(|path| path.is_file())
|
||||
.map(FileInfo::new)
|
||||
.filter_map(Result::ok)
|
||||
.filter(|file| file.size > min_size)
|
||||
.for_each(|file| {
|
||||
let mut flock = files.lock().unwrap();
|
||||
flock.push(file);
|
||||
});
|
||||
|
||||
progress_bar.finish_with_message("paths mapped");
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
109
src/server.rs
Normal file
109
src/server.rs
Normal file
@@ -0,0 +1,109 @@
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use crate::processor::Processor;
|
||||
use crate::scanner::Scanner;
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{MultiProgress, ProgressDrawTarget};
|
||||
use rand::Rng;
|
||||
use threadpool::ThreadPool;
|
||||
|
||||
use crate::fileinfo::FileInfo;
|
||||
use crate::params::Params;
|
||||
|
||||
pub struct Server {
|
||||
filequeue: Arc<Mutex<Vec<FileInfo>>>,
|
||||
sw_duplicate_set: Arc<DashMap<u64, Vec<FileInfo>>>,
|
||||
pub hw_duplicate_set: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
threadpool: ThreadPool,
|
||||
app_args: Arc<Params>,
|
||||
pub max_file_path_len: Arc<AtomicU64>,
|
||||
}
|
||||
|
||||
impl Server {
|
||||
pub fn new(opts: Params) -> Self {
|
||||
Self {
|
||||
filequeue: Arc::new(Mutex::new(Vec::new())),
|
||||
sw_duplicate_set: Arc::new(DashMap::new()),
|
||||
hw_duplicate_set: Arc::new(DashMap::new()),
|
||||
threadpool: ThreadPool::new(4),
|
||||
app_args: Arc::new(opts),
|
||||
max_file_path_len: Arc::new(AtomicU64::new(0)),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn start(&self) -> Result<()> {
|
||||
let progbarbox = Arc::new(MultiProgress::new());
|
||||
let mut rng = rand::rng();
|
||||
let seed: i64 = rng.random();
|
||||
|
||||
if !self.app_args.progress {
|
||||
progbarbox.set_draw_target(ProgressDrawTarget::hidden());
|
||||
}
|
||||
|
||||
let app_args_clone_for_sc = self.app_args.clone();
|
||||
let app_args_clone_for_sw = self.app_args.clone();
|
||||
let app_args_clone_for_hw = self.app_args.clone();
|
||||
let file_queue_clone_sc = self.filequeue.clone();
|
||||
let file_queue_clone_pr = self.filequeue.clone();
|
||||
let scanner_finished = Arc::new(AtomicBool::new(false));
|
||||
let sw_sort_finished = Arc::new(AtomicBool::new(false));
|
||||
|
||||
let sfin_sc_tr_cl = scanner_finished.clone();
|
||||
let sfin_pr_tr_cl = scanner_finished.clone();
|
||||
|
||||
let swfin_pr_tr_sw = sw_sort_finished.clone();
|
||||
let swfin_pr_tr_hw = sw_sort_finished.clone();
|
||||
|
||||
let store_dupl_sw_for_sw = self.sw_duplicate_set.clone();
|
||||
let store_dupl_sw_for_hw = self.sw_duplicate_set.clone();
|
||||
let store_dupl_hw = self.hw_duplicate_set.clone();
|
||||
let max_file_path_len_clone = self.max_file_path_len.clone();
|
||||
|
||||
let progbarbox_sc_clone = progbarbox.clone();
|
||||
let progbarbox_pr_clone_for_sw = progbarbox.clone();
|
||||
let progbarbox_pr_clone_for_hw = progbarbox.clone();
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Scanner::new(app_args_clone_for_sc)
|
||||
.expect("unable to initialize scanner.")
|
||||
.scan(file_queue_clone_sc, progbarbox_sc_clone)
|
||||
.expect("scanner failed.");
|
||||
|
||||
sfin_sc_tr_cl.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
});
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Processor::sizewise(
|
||||
app_args_clone_for_sw,
|
||||
sfin_pr_tr_cl,
|
||||
store_dupl_sw_for_sw,
|
||||
file_queue_clone_pr,
|
||||
progbarbox_pr_clone_for_sw,
|
||||
)
|
||||
.expect("sizewise scanner failed.");
|
||||
|
||||
swfin_pr_tr_sw.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
});
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Processor::hashwise(
|
||||
app_args_clone_for_hw,
|
||||
store_dupl_sw_for_hw,
|
||||
store_dupl_hw,
|
||||
progbarbox_pr_clone_for_hw,
|
||||
max_file_path_len_clone,
|
||||
seed,
|
||||
swfin_pr_tr_hw,
|
||||
)
|
||||
.expect("sizewise scanner failed.");
|
||||
});
|
||||
|
||||
progbarbox.clear()?;
|
||||
|
||||
self.threadpool.join();
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user