mirror of
https://github.com/sreedevk/deduplicator.git
synced 2026-08-26 18:15:33 +00:00
Compare commits
14 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d06fe93171 | ||
|
|
6a24ca5ecf | ||
|
|
6f23a51a53 | ||
|
|
4894d1b4db | ||
|
|
c0286b17cc | ||
|
|
e78dfc4211 | ||
|
|
66f49a9aee | ||
|
|
711eb36fb9 | ||
|
|
130a8f99ba | ||
|
|
d97af6b8f6 | ||
|
|
90198502a3 | ||
|
|
e1b81b9981 | ||
|
|
48e7c62036 | ||
|
|
58e33cc8da |
@@ -1,2 +0,0 @@
|
||||
[build]
|
||||
rustflags = ["-C", "target-feature=+aes,+sse2"]
|
||||
208
.github/workflows/release.yml
vendored
208
.github/workflows/release.yml
vendored
@@ -1,137 +1,121 @@
|
||||
# CI that:
|
||||
#
|
||||
# * checks for a Git Tag that looks like a release
|
||||
# * creates a Github Release™ and fills in its text
|
||||
# * builds artifacts with cargo-dist (executable-zips, installers)
|
||||
# * uploads those artifacts to the Github Release™
|
||||
#
|
||||
# Note that the Github Release™ will be created before the artifacts,
|
||||
# so there will be a few minutes where the release has no artifacts
|
||||
# and then they will slowly trickle in, possibly failing. To make
|
||||
# this more pleasant we mark the release as a "draft" until all
|
||||
# artifacts have been successfully uploaded. This allows you to
|
||||
# choose what to do with partial successes and avoids spamming
|
||||
# anyone with notifications before the release is actually ready.
|
||||
name: Release
|
||||
name: Deduplicator Release Build CI Pipeline
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
# This task will run whenever you push a git tag that looks like a version
|
||||
# like "v1", "v1.2.0", "v0.1.0-prerelease01", "my-app-v1.0.0", etc.
|
||||
# The version will be roughly parsed as ({PACKAGE_NAME}-)?v{VERSION}, where
|
||||
# PACKAGE_NAME must be the name of a Cargo package in your workspace, and VERSION
|
||||
# must be a Cargo-style SemVer Version.
|
||||
#
|
||||
# If PACKAGE_NAME is specified, then we will create a Github Release™ for that
|
||||
# package (erroring out if it doesn't have the given version or isn't cargo-dist-able).
|
||||
#
|
||||
# If PACKAGE_NAME isn't specified, then we will create a Github Release™ for all
|
||||
# (cargo-dist-able) packages in the workspace with that version (this is mode is
|
||||
# intended for workspaces with only one dist-able package, or with all dist-able
|
||||
# packages versioned/released in lockstep).
|
||||
#
|
||||
# If you push multiple tags at once, separate instances of this workflow will
|
||||
# spin up, creating an independent Github Release™ for each one.
|
||||
#
|
||||
# If there's a prerelease-style suffix to the version then the Github Release™
|
||||
# will be marked as a prerelease.
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- '*-?v[0-9]+*'
|
||||
- 'v*.*.*'
|
||||
|
||||
jobs:
|
||||
# Create the Github Release™ so the packages have something to be uploaded to
|
||||
create-release:
|
||||
lint_test:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
has-releases: ${{ steps.create-release.outputs.has-releases }}
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
RUSTFLAGS: "-C target-feature=+aes,+sse2"
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
run: rustup update 1.71.0 --no-self-update && rustup default 1.71.0
|
||||
- name: Install cargo-dist
|
||||
run: curl --proto '=https' --tlsv1.2 -LsSf https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.sh | sh
|
||||
- id: create-release
|
||||
run: |
|
||||
cargo dist plan --tag=${{ github.ref_name }} --output-format=json > dist-manifest.json
|
||||
echo "dist plan ran successfully"
|
||||
cat dist-manifest.json
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
components: clippy
|
||||
|
||||
# Create the Github Release™ based on what cargo-dist thinks it should be
|
||||
ANNOUNCEMENT_TITLE=$(jq --raw-output ".announcement_title" dist-manifest.json)
|
||||
IS_PRERELEASE=$(jq --raw-output ".announcement_is_prerelease" dist-manifest.json)
|
||||
jq --raw-output ".announcement_github_body" dist-manifest.json > new_dist_announcement.md
|
||||
gh release create ${{ github.ref_name }} --draft --prerelease="$IS_PRERELEASE" --title="$ANNOUNCEMENT_TITLE" --notes-file=new_dist_announcement.md
|
||||
echo "created announcement!"
|
||||
- name: Run cargo check
|
||||
run: cargo check
|
||||
|
||||
# Upload the manifest to the Github Release™
|
||||
gh release upload ${{ github.ref_name }} dist-manifest.json
|
||||
echo "uploaded manifest!"
|
||||
- name: Run cargo clippy
|
||||
run: cargo clippy -- -D warnings
|
||||
|
||||
# Disable all the upload-artifacts tasks if we have no actual releases
|
||||
HAS_RELEASES=$(jq --raw-output ".releases != null" dist-manifest.json)
|
||||
echo "has-releases=$HAS_RELEASES" >> "$GITHUB_OUTPUT"
|
||||
- name: Run tests
|
||||
run: cargo test -- --test-threads=1
|
||||
|
||||
# Build and packages all the things
|
||||
upload-artifacts:
|
||||
# Let the initial task tell us to not run (currently very blunt)
|
||||
needs: create-release
|
||||
if: ${{ needs.create-release.outputs.has-releases == 'true' }}
|
||||
build:
|
||||
needs: lint_test
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
# For these target platforms
|
||||
include:
|
||||
- os: macos-11
|
||||
dist-args: --artifacts=local --target=aarch64-apple-darwin --target=x86_64-apple-darwin
|
||||
install-dist: curl --proto '=https' --tlsv1.2 -LsSf https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.sh | sh
|
||||
- os: ubuntu-20.04
|
||||
dist-args: --artifacts=local --target=x86_64-unknown-linux-gnu
|
||||
install-dist: curl --proto '=https' --tlsv1.2 -LsSf https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.sh | sh
|
||||
- os: windows-2019
|
||||
dist-args: --artifacts=local --target=x86_64-pc-windows-msvc
|
||||
install-dist: irm https://github.com/axodotdev/cargo-dist/releases/download/v0.0.7/cargo-dist-installer.ps1 | iex
|
||||
- target: aarch64-unknown-linux-gnu
|
||||
os: ubuntu-latest
|
||||
artifact_name: linux-aarch64
|
||||
rustflags: "-C target-feature=+aes,+neon"
|
||||
- target: x86_64-unknown-linux-gnu
|
||||
os: ubuntu-latest
|
||||
artifact_name: linux-amd64
|
||||
rustflags: "-C target-feature=+aes,+sse2"
|
||||
- target: x86_64-apple-darwin
|
||||
os: macos-14
|
||||
artifact_name: macos-amd64
|
||||
rustflags: "-C target-feature=+aes,+sse2"
|
||||
- target: aarch64-apple-darwin
|
||||
os: macos-14
|
||||
artifact_name: macos-arm64
|
||||
rustflags: "-C target-feature=+aes,+neon"
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
run: rustup update 1.71.0 --no-self-update && rustup default 1.71.0
|
||||
- name: Install cargo-dist
|
||||
run: ${{ matrix.install-dist }}
|
||||
- name: Run cargo-dist
|
||||
# This logic is a bit janky because it's trying to be a polyglot between
|
||||
# powershell and bash since this will run on windows, macos, and linux!
|
||||
# The two platforms don't agree on how to talk about env vars but they
|
||||
# do agree on 'cat' and '$()' so we use that to marshal values between commands.
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Add target
|
||||
run: rustup target add ${{ matrix.target }}
|
||||
|
||||
- name: Install cross-compilation toolchain (Linux ARM only)
|
||||
if: matrix.target == 'aarch64-unknown-linux-gnu'
|
||||
run: |
|
||||
# Actually do builds and make zips and whatnot
|
||||
cargo dist build --tag=${{ github.ref_name }} --output-format=json ${{ matrix.dist-args }} > dist-manifest.json
|
||||
echo "dist ran successfully"
|
||||
cat dist-manifest.json
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y gcc-aarch64-linux-gnu
|
||||
|
||||
# Parse out what we just built and upload it to the Github Release™
|
||||
jq --raw-output ".artifacts[]?.path | select( . != null )" dist-manifest.json > uploads.txt
|
||||
echo "uploading..."
|
||||
cat uploads.txt
|
||||
gh release upload ${{ github.ref_name }} $(cat uploads.txt)
|
||||
echo "uploaded!"
|
||||
- name: Build release binary
|
||||
env:
|
||||
RUSTFLAGS: ${{ matrix.rustflags }}
|
||||
CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER: ${{ matrix.target == 'aarch64-unknown-linux-gnu' && 'aarch64-linux-gnu-gcc' || '' }}
|
||||
run: |
|
||||
if [ "${{ matrix.target }}" = "aarch64-unknown-linux-gnu" ]; then
|
||||
export CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER=aarch64-linux-gnu-gcc
|
||||
fi
|
||||
cargo build --release --target ${{ matrix.target }}
|
||||
|
||||
# Mark the Github Release™ as a non-draft now that everything has succeeded!
|
||||
publish-release:
|
||||
# Only run after all the other tasks, but it's ok if upload-artifacts was skipped
|
||||
needs: [create-release, upload-artifacts]
|
||||
if: ${{ always() && needs.create-release.result == 'success' && (needs.upload-artifacts.result == 'skipped' || needs.upload-artifacts.result == 'success') }}
|
||||
- name: Package binary
|
||||
run: |
|
||||
if [[ "${{ matrix.os }}" == macos* ]]; then
|
||||
BINARY_NAME=$(find target/${{ matrix.target }}/release -maxdepth 1 -type f -perm +111 -print0 | xargs -0 basename | head -1)
|
||||
else
|
||||
BINARY_NAME=$(find target/${{ matrix.target }}/release -maxdepth 1 -type f -executable -print0 | xargs -0 basename | head -1)
|
||||
fi
|
||||
|
||||
echo "Packaging binary: $BINARY_NAME"
|
||||
mkdir -p release
|
||||
cp target/${{ matrix.target }}/release/$BINARY_NAME release/$BINARY_NAME
|
||||
tar -C release -czf ${{ matrix.artifact_name }}.tar.gz $BINARY_NAME
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: ${{ matrix.artifact_name }}-binary
|
||||
path: ${{ matrix.artifact_name }}.tar.gz
|
||||
|
||||
create_release:
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- name: mark release as non-draft
|
||||
run: |
|
||||
gh release edit ${{ github.ref_name }} --draft=false
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
path: artifacts
|
||||
|
||||
- name: Create GitHub Release
|
||||
id: create_release
|
||||
uses: softprops/action-gh-release@v1
|
||||
with:
|
||||
tag_name: ${{ github.ref_name }}
|
||||
name: Release ${{ github.ref_name }}
|
||||
body: "Automated release for version ${{ github.ref_name }}"
|
||||
files: |
|
||||
artifacts/*/*.tar.gz
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
569
Cargo.lock
generated
569
Cargo.lock
generated
File diff suppressed because it is too large
Load Diff
12
Cargo.toml
12
Cargo.toml
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "deduplicator"
|
||||
version = "0.3.0"
|
||||
version = "0.3.1"
|
||||
edition = "2021"
|
||||
description = "find,filter and delete duplicate files"
|
||||
repository = "https://github.com/sreedevk/deduplicator"
|
||||
@@ -18,14 +18,14 @@ path = "src/main.rs"
|
||||
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
|
||||
[dependencies]
|
||||
anyhow = "1.0.68"
|
||||
bytesize = "1.1.0"
|
||||
bytesize = "2.0.1"
|
||||
chrono = "0.4.23"
|
||||
clap = { version = "4.0.32", features = ["derive"] }
|
||||
dashmap = { version = "5.4.0", features = ["rayon"] }
|
||||
globwalk = "0.8.1"
|
||||
dashmap = { version = "6.1.0", features = ["rayon"] }
|
||||
globwalk = "0.9.1"
|
||||
gxhash = { version = "3.4.1", default-features = false }
|
||||
indicatif = { version = "0.17.2", features = ["rayon"] }
|
||||
memmap2 = "0.5.8"
|
||||
indicatif = { version = "0.18.0", features = ["rayon"] }
|
||||
memmap2 = "0.9.7"
|
||||
pathdiff = "0.2.1"
|
||||
prettytable-rs = "0.10.0"
|
||||
rand = "0.9.1"
|
||||
|
||||
154
README.md
154
README.md
@@ -15,16 +15,17 @@ Arguments:
|
||||
[scan_dir_path] Run Deduplicator on dir different from pwd (e.g., ~/Pictures )
|
||||
|
||||
Options:
|
||||
-t, --types <TYPES> Filetypes to deduplicate [default = all]
|
||||
-i, --interactive Delete files interactively
|
||||
-m, --min-size <MIN_SIZE> Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T) [default: 1b]
|
||||
-D, --max-depth <MAX_DEPTH> Max Depth to scan while looking for duplicates
|
||||
-d, --min-depth <MIN_DEPTH> Min Depth to scan while looking for duplicates
|
||||
-f, --follow-links Follow links while scanning directories
|
||||
-s, --strict Guarantees that two files are duplicate (performs a full hash)
|
||||
-p, --progress Show Progress spinners & metrics
|
||||
-h, --help Print help
|
||||
-V, --version Print version
|
||||
-T, --exclude-types <EXCLUDE_TYPES> Exclude Filetypes [default = none]
|
||||
-t, --types <TYPES> Filetypes to deduplicate [default = all]
|
||||
-i, --interactive Delete files interactively
|
||||
-m, --min-size <MIN_SIZE> Minimum filesize of duplicates to scan (e.g., 100B/1K/2M/3G/4T) [default: 1b]
|
||||
-D, --max-depth <MAX_DEPTH> Max Depth to scan while looking for duplicates
|
||||
-d, --min-depth <MIN_DEPTH> Min Depth to scan while looking for duplicates
|
||||
-f, --follow-links Follow links while scanning directories
|
||||
-s, --strict Guarantees that two files are duplicate (performs a full hash)
|
||||
-p, --progress Show Progress spinners & metrics
|
||||
-h, --help Print help
|
||||
-V, --version Print version
|
||||
```
|
||||
### Examples
|
||||
|
||||
@@ -32,6 +33,9 @@ Options:
|
||||
# Scan for duplicates recursively from the current dir, only look for png, jpg & pdf file types & interactively delete files
|
||||
deduplicator -t pdf,jpg,png -i
|
||||
|
||||
# Scan for duplicates recursively from current dir, exclude png and jpg file types
|
||||
deduplicator -T jpg,png
|
||||
|
||||
# Scan for duplicates recursively from the ~/Pictures dir, only look for png, jpeg, jpg & pdf file types & interactively delete files
|
||||
deduplicator ~/Pictures/ -t png,jpeg,jpg,pdf -i
|
||||
|
||||
@@ -57,14 +61,23 @@ Currently, you can only install deduplicator using cargo package manager.
|
||||
> GxHash relies on aes hardware acceleration, so please set `RUSTFLAGS` to `"-C target-feature=+aes"` or `"-C target-cpu=native"` before
|
||||
> installing.
|
||||
|
||||
#### install from crates.io
|
||||
#### install from crates.io (stable)
|
||||
|
||||
```bash
|
||||
$ RUSTFLAGS="-C target-cpu=native" cargo install deduplicator
|
||||
|
||||
# or
|
||||
|
||||
$ RUSTFLAGS="-C target-feature=+aes,+sse2" cargo install deduplicator
|
||||
```
|
||||
|
||||
#### install from git
|
||||
#### install from git (nightly)
|
||||
```bash
|
||||
$ RUSTFLAGS="-C target-cpu=native" cargo install deduplicator --git https://github.com/sreedevk/deduplicator
|
||||
|
||||
# or
|
||||
|
||||
$ RUSTFLAGS="-C target-feature=+aes,+sse2" cargo install --git https://github.com/sreedevk/deduplicator
|
||||
```
|
||||
|
||||
## Performance
|
||||
@@ -77,77 +90,98 @@ I've used hyperfine to run deduplicator on files generated by the rake file at `
|
||||
```
|
||||
# hyperfine -N --warmup 80 './target/release/deduplicator bench_artifacts'
|
||||
Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
Time (mean ± σ): 2.5 ms ± 0.4 ms [User: 2.2 ms, System: 4.8 ms]
|
||||
Range (min … max): 1.9 ms … 6.8 ms 1322 runs
|
||||
Time (mean ± σ): 2.2 ms ± 0.4 ms [User: 2.2 ms, System: 4.4 ms]
|
||||
Range (min … max): 1.3 ms … 7.1 ms 1522 runs
|
||||
|
||||
# dust 'bench_artifacts'
|
||||
37M ┌── file_1_fwds.bin │██ │ 1%
|
||||
201M ├── file_0_fwds.bin │██████████ │ 8%
|
||||
390M ├── file_0_fwdcbss.bin│████████████████████ │ 15%
|
||||
390M ├── file_0_fwscas.bin │████████████████████ │ 15%
|
||||
390M ├── file_0_fwss.bin │████████████████████ │ 15%
|
||||
390M ├── file_1_fwdcbss.bin│████████████████████ │ 15%
|
||||
390M ├── file_1_fwscas.bin │████████████████████ │ 15%
|
||||
390M ├── file_1_fwss.bin │████████████████████ │ 15%
|
||||
2.5G ┌─┴ bench_artifacts │███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████ │ 100%
|
||||
dust 'bench_artifacts'
|
||||
|
||||
54M ┌── file_0_fwds.bin │████ │ 2%
|
||||
122M ├── file_1_fwds.bin │████████ │ 5%
|
||||
390M ├── file_0_fwdcbss.bin│██████████████████████████ │ 15%
|
||||
390M ├── file_0_fwscas.bin │██████████████████████████ │ 15%
|
||||
390M ├── file_0_fwss.bin │██████████████████████████ │ 15%
|
||||
390M ├── file_1_fwdcbss.bin│██████████████████████████ │ 15%
|
||||
390M ├── file_1_fwscas.bin │██████████████████████████ │ 15%
|
||||
390M ├── file_1_fwss.bin │██████████████████████████ │ 15%
|
||||
2.5G ┌─┴ bench_artifacts │██████████████████████████████████████████████████████████████████ │ 100%
|
||||
```
|
||||
|
||||
#### Many Small Files
|
||||
```
|
||||
# hyperfine --warmup 20 './target/release/deduplicator bench_artifacts'
|
||||
Benchmark 1: ./target/release/deduplicator bench_artifacts
|
||||
Time (mean ± σ): 22.3 ms ± 1.7 ms [User: 35.8 ms, System: 46.3 ms]
|
||||
Range (min … max): 18.9 ms … 27.2 ms 112 runs
|
||||
Time (mean ± σ): 40.1 ms ± 2.3 ms [User: 251.0 ms, System: 277.3 ms]
|
||||
Range (min … max): 35.0 ms … 45.9 ms 72 runs
|
||||
|
||||
# dust 'bench_artifacts'
|
||||
3.9M ┌── file_992_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_993_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_994_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_995_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_996_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_997_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_998_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_999_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwss.bin │█ │ 0%
|
||||
11G ┌─┴ bench_artifacts │█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████ │ 100%
|
||||
dust 'bench_artifacts'
|
||||
3.9M ┌── file_992_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_992_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_993_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_993_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_994_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_994_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_995_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_995_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_996_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_996_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_997_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_997_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_998_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_998_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwdcbss.bin│█ │ 0%
|
||||
3.9M ├── file_999_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_999_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_99_fwss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwdcbss.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwscas.bin │█ │ 0%
|
||||
3.9M ├── file_9_fwss.bin │█ │ 0%
|
||||
11G ┌─┴ bench_artifacts │████████████████████████████████████████████████████████████████ │ 100%
|
||||
```
|
||||
|
||||
## proposed
|
||||
- [ ] parallelization
|
||||
- [ ] (scanning + processing sw + processing hw) & formatting & printing
|
||||
- [ ] scanning + processing sw + processing hw + formatting + printing
|
||||
- [ ] max file path size should use the last set of duplicates
|
||||
- [ ] add more unit tests
|
||||
- [ ] test against different filesystems
|
||||
- [ ] test against different file name encodings
|
||||
- [ ] restore json output (was removed in 0.3)
|
||||
- [ ] restore json output (was removed in 0.3 due to quality issues)
|
||||
- [ ] fix memory leak on very large filesystems
|
||||
- [ ] maybe use a bloom filter
|
||||
- [ ] reduce FileInfo size
|
||||
- [ ] output in a tree format
|
||||
- [ ] tui
|
||||
- [ ] change the default hashing method to include the first & last page of a file (8K)
|
||||
- [ ] provide option to localize duplicate detection to arbitrary levels relative to current directory
|
||||
- [ ] bulk operations
|
||||
- [ ] --keep-latest
|
||||
- [ ] --keep-oldest
|
||||
- [ ] --keep-last-modified
|
||||
- [ ] --keep-first-modified
|
||||
|
||||
## v0.3
|
||||
- [ ] fix: partial hash collision - a file full of null bytes ("\0") and an empty file. This is a known trade off in gxhash.
|
||||
- [ ] include initial pages and final pages of the file
|
||||
- [ ] append the offset between the last initial page hashed and the first final page hashed in the content passed to the hasher.
|
||||
|
||||
## v0.3.1
|
||||
- [x] parallelization
|
||||
- [x] (scanning + processing sw + processing hw) & formatting & printing
|
||||
- [x] remove formatting step and write directly to stdout
|
||||
- [x] simplify output to improve performance
|
||||
- [x] increase the number of pages hashed in partial hashing
|
||||
- [x] updated dependencies
|
||||
- [x] fix: full file hash collision between a file full of null bytes ("\0") and an empty file. This is a known trade off in gxhash.
|
||||
- [x] appending the file size at the end of content before hashing.
|
||||
- [x] created an automated release pipeline
|
||||
|
||||
## v0.3.0
|
||||
- [x] parallelization
|
||||
- [x] (scanning) + (processing sw & processing hw & formatting & printing)
|
||||
- [x] reduce cloning values on the heap
|
||||
|
||||
@@ -13,9 +13,11 @@ namespace :benchmark do
|
||||
# Range (min … max): 1.8 ms … 9.7 ms 1474 runs
|
||||
|
||||
root = "bench_artifacts"
|
||||
FileUtils.rm_rf(root)
|
||||
Dir.mkdir(root)
|
||||
|
||||
# files with same size
|
||||
puts "generating files of same size ..."
|
||||
2.times.map do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 100_000))
|
||||
@@ -23,6 +25,7 @@ namespace :benchmark do
|
||||
end
|
||||
|
||||
# files with different sizes
|
||||
puts "generating files of different sizes ..."
|
||||
2.times.map do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwds.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * (rand * 100_000).ceil))
|
||||
@@ -30,6 +33,7 @@ namespace :benchmark do
|
||||
end
|
||||
|
||||
# files with same content & size
|
||||
puts "generating files of same content and sizes ..."
|
||||
2.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwscas.bin"), 'wb') do |f|
|
||||
f.write("\0" * (4096 * 100_000))
|
||||
@@ -37,6 +41,7 @@ namespace :benchmark do
|
||||
end
|
||||
|
||||
# files with different content but same size
|
||||
puts "generating files of different content but same sizes ..."
|
||||
2.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwdcbss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 100_000))
|
||||
@@ -58,6 +63,7 @@ namespace :benchmark do
|
||||
Dir.mkdir(root)
|
||||
|
||||
# files with same size
|
||||
puts "generating 1000 files of the same size ... "
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 1000))
|
||||
@@ -65,6 +71,7 @@ namespace :benchmark do
|
||||
end
|
||||
|
||||
# files with different sizes
|
||||
puts "generating 1000 files of different sizes ... "
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwds.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * (rand * 100).ceil))
|
||||
@@ -72,6 +79,7 @@ namespace :benchmark do
|
||||
end
|
||||
|
||||
# files with same content & size
|
||||
puts "generating files of same content and sizes ..."
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwscas.bin"), 'wb') do |f|
|
||||
f.write("\0" * (4096 * 1000))
|
||||
@@ -79,6 +87,7 @@ namespace :benchmark do
|
||||
end
|
||||
|
||||
# files with different content but same size
|
||||
puts "generating files of different content but same sizes ..."
|
||||
1000.times.each do |i|
|
||||
File.open(File.join(root, "file_#{i}_fwdcbss.bin"), 'wb') do |f|
|
||||
f.write(SecureRandom.bytes(4096 * 1000))
|
||||
|
||||
@@ -5,30 +5,43 @@ use std::{
|
||||
fs,
|
||||
io::Read,
|
||||
path::{Path, PathBuf},
|
||||
sync::{Arc, Mutex},
|
||||
time::SystemTime,
|
||||
};
|
||||
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum FileState {
|
||||
Unprocessed,
|
||||
SwProcessed,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct FileInfo {
|
||||
pub path: Box<Path>,
|
||||
pub size: u64,
|
||||
pub modified: SystemTime,
|
||||
pub state: Arc<Mutex<FileState>>,
|
||||
}
|
||||
|
||||
impl FileInfo {
|
||||
pub fn hash(&self, seed: i64) -> Result<u128> {
|
||||
if self.size == 0 {
|
||||
return Ok(0u128);
|
||||
};
|
||||
|
||||
let file = fs::File::open(&self.path)?;
|
||||
let mapper = unsafe { Mmap::map(&file)? };
|
||||
let final_hash = mapper.chunks(4096).fold(0u128, |acc, chunk: &[u8]| {
|
||||
acc.wrapping_add(gxhash128(chunk, seed))
|
||||
});
|
||||
let content_hash = mapper
|
||||
.chunks(4096)
|
||||
.fold(0u128, |acc, chunk: &[u8]| acc ^ gxhash128(chunk, seed));
|
||||
|
||||
Ok(final_hash)
|
||||
// NOTE: avoids collision bw an empty file & a file full of null bytes.
|
||||
Ok(content_hash ^ gxhash128(&self.size.to_ne_bytes(), seed))
|
||||
}
|
||||
|
||||
pub fn initial_page_hash(&self, seed: i64) -> Result<u128> {
|
||||
pub fn initpages_hash(&self, seed: i64) -> Result<u128> {
|
||||
let mut file = fs::File::open(&self.path)?;
|
||||
let mut buffer = [0; 4096];
|
||||
let mut buffer = [0; 16384];
|
||||
let bytes_read = file.read(&mut buffer)?;
|
||||
|
||||
Ok(gxhash128(&buffer[..bytes_read], seed))
|
||||
@@ -40,6 +53,52 @@ impl FileInfo {
|
||||
path: path.into_boxed_path(),
|
||||
size: filemeta.len(),
|
||||
modified: filemeta.modified()?,
|
||||
state: Arc::new(Mutex::new(FileState::Unprocessed)),
|
||||
})
|
||||
}
|
||||
|
||||
pub fn sw_processed(&self) {
|
||||
let mut self_state = self.state.lock().unwrap();
|
||||
*self_state = FileState::SwProcessed;
|
||||
}
|
||||
|
||||
pub fn is_sw_processed(&self) -> bool {
|
||||
let self_state = self.state.lock().unwrap();
|
||||
*self_state == FileState::SwProcessed
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod test {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
use std::fs::File;
|
||||
use std::io::Write;
|
||||
use anyhow::Result;
|
||||
|
||||
fn generate_null_bytes(size: usize) -> Vec<u8> {
|
||||
(0..size).map(|_| 0).collect::<Vec<u8>>()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hash_differentiates_between_a_file_of_null_bytes_vs_an_empty_file() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let empty_file_name = root.path().join("empty_file.bin");
|
||||
|
||||
File::create_new(&empty_file_name)?;
|
||||
|
||||
let file_with_null_bytes_name = root.path().join("file_with_null_bytes.bin");
|
||||
let mut file_with_null_bytes = File::create_new(&file_with_null_bytes_name)?;
|
||||
|
||||
file_with_null_bytes.write_all(&generate_null_bytes(1000 * 4096))?;
|
||||
|
||||
let empty_file_info = FileInfo::new(empty_file_name)?;
|
||||
let file_with_empty_bytes_info = FileInfo::new(file_with_null_bytes_name)?;
|
||||
|
||||
let seed: i64 = 246910456374;
|
||||
|
||||
assert_ne!(empty_file_info.hash(seed)?, file_with_empty_bytes_info.hash(seed)?);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,20 +2,17 @@ use crate::{fileinfo::FileInfo, params::Params};
|
||||
use anyhow::Result;
|
||||
use chrono::{DateTime, Utc};
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{ParallelProgressIterator, ProgressBar, ProgressFinish, ProgressStyle};
|
||||
use pathdiff::diff_paths;
|
||||
use prettytable::{format, row, Row, Table};
|
||||
use rayon::prelude::*;
|
||||
use std::{borrow::Cow, path::PathBuf, sync::Arc, time::Duration};
|
||||
use std::{path::PathBuf, sync::Arc};
|
||||
|
||||
const YELLOW: &str = "\x1b[33m";
|
||||
const RESET: &str = "\x1b[0m";
|
||||
|
||||
pub struct Formatter;
|
||||
impl Formatter {
|
||||
pub fn human_path(
|
||||
file: &FileInfo,
|
||||
app_args: &Params,
|
||||
min_path_length: usize,
|
||||
) -> Result<String> {
|
||||
let base_directory: PathBuf = app_args.get_directory()?;
|
||||
pub fn human_path(file: &FileInfo, aargs: &Params, min_path_length: usize) -> Result<String> {
|
||||
let base_directory: PathBuf = aargs.get_directory()?;
|
||||
let relative_path = diff_paths(&file.path, base_directory).unwrap_or_default();
|
||||
|
||||
let formatted_path = format!(
|
||||
@@ -36,66 +33,32 @@ impl Formatter {
|
||||
Ok(modified_time.format("%Y-%m-%d %H:%M:%S").to_string())
|
||||
}
|
||||
|
||||
pub fn gen_sub_tbl(items: Vec<FileInfo>, app_args: &Params, max_path_len: u64) -> Table {
|
||||
let mut inner_table = Table::new();
|
||||
inner_table.set_format(*format::consts::FORMAT_NO_BORDER_LINE_SEPARATOR);
|
||||
items.iter().for_each(|file| {
|
||||
inner_table.add_row(row![
|
||||
Self::human_path(file, app_args, max_path_len as usize).unwrap_or_default(),
|
||||
Self::human_filesize(file).unwrap_or_default(),
|
||||
Self::human_mtime(file).unwrap_or_default()
|
||||
]);
|
||||
});
|
||||
inner_table
|
||||
}
|
||||
pub fn print(raw: Arc<DashMap<u128, Vec<FileInfo>>>, max_path_len: u64, aargs: &Params) {
|
||||
print!("{}", "\n".repeat(if aargs.progress { 2 } else { 1 })); // spacing
|
||||
|
||||
pub fn generate_table(
|
||||
raw: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
mpath_len: u64,
|
||||
args: &Params,
|
||||
) -> Result<Table> {
|
||||
let progress_bar = match args.progress {
|
||||
true => ProgressBar::new_spinner(),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("generating output");
|
||||
|
||||
let rows = raw
|
||||
.par_iter_mut()
|
||||
.progress_with(progress_bar)
|
||||
.with_finish(ProgressFinish::WithMessage(Cow::from("output generated")))
|
||||
.filter(|i| i.value().len() > 1)
|
||||
.map(|i| {
|
||||
row![
|
||||
i.key(),
|
||||
Self::gen_sub_tbl(i.value().to_vec(), args, mpath_len)
|
||||
]
|
||||
})
|
||||
.collect::<Vec<Row>>();
|
||||
|
||||
let mut output_table = Table::new();
|
||||
output_table.set_titles(row!["hash", "duplicates"]);
|
||||
output_table.extend(rows);
|
||||
Ok(output_table)
|
||||
}
|
||||
|
||||
pub fn print(
|
||||
raw: Arc<DashMap<u128, Vec<FileInfo>>>,
|
||||
max_path_len: u64,
|
||||
app_args: &Params,
|
||||
) -> Result<()> {
|
||||
if raw.is_empty() {
|
||||
println!("\n\nNo duplicates found matching your search criteria.\n");
|
||||
return Ok(());
|
||||
println!("No duplicates found matching your search criteria.");
|
||||
} else {
|
||||
raw.par_iter().for_each(|sref| {
|
||||
let mut ostring = format!("{}{:32x}{}\n", YELLOW, sref.key(), RESET);
|
||||
let subfields = sref
|
||||
.value()
|
||||
.par_iter()
|
||||
.map(|finfo| {
|
||||
format!(
|
||||
"├─ {}\t{}\t{}\n",
|
||||
Self::human_path(finfo, aargs, max_path_len as usize)
|
||||
.expect("path formatting failed."),
|
||||
Self::human_filesize(finfo).expect("filesize formatting failed."),
|
||||
Self::human_mtime(finfo).expect("modified time formatting failed.")
|
||||
)
|
||||
})
|
||||
.collect::<String>();
|
||||
|
||||
ostring.push_str(&subfields);
|
||||
|
||||
println!("{ostring}");
|
||||
});
|
||||
}
|
||||
|
||||
let output_table = Self::generate_table(raw, max_path_len, app_args)?;
|
||||
output_table.printstd();
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -24,7 +24,7 @@ fn main() -> Result<()> {
|
||||
server.hw_duplicate_set,
|
||||
server.max_file_path_len.load(Ordering::Acquire),
|
||||
&app_args,
|
||||
)?;
|
||||
);
|
||||
}
|
||||
true => {
|
||||
Interactive::init(server.hw_duplicate_set, &app_args)?;
|
||||
|
||||
@@ -2,10 +2,14 @@ use std::{fs, path::PathBuf};
|
||||
|
||||
use anyhow::Result;
|
||||
use clap::{Parser, ValueHint};
|
||||
use std::collections::HashSet;
|
||||
|
||||
#[derive(Parser, Debug, Default, Clone)]
|
||||
#[command(author, version, about, long_about = None)]
|
||||
pub struct Params {
|
||||
/// Exclude Filetypes [default = none]
|
||||
#[arg(short = 'T', long)]
|
||||
pub exclude_types: Option<String>,
|
||||
/// Filetypes to deduplicate [default = all]
|
||||
#[arg(short, long)]
|
||||
pub types: Option<String>,
|
||||
@@ -53,7 +57,47 @@ impl Params {
|
||||
Ok(dir)
|
||||
}
|
||||
|
||||
pub fn types_intersection(itypes: &str, xtypes: &str) -> String {
|
||||
let iset = itypes
|
||||
.split(",")
|
||||
.map(String::from)
|
||||
.collect::<HashSet<String>>();
|
||||
let xset = xtypes
|
||||
.split(",")
|
||||
.map(String::from)
|
||||
.collect::<HashSet<String>>();
|
||||
|
||||
iset.difference(&xset)
|
||||
.cloned()
|
||||
.collect::<Vec<String>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
pub fn get_types(&self) -> Option<String> {
|
||||
self.types.clone()
|
||||
match &self.types {
|
||||
Some(itypes) => match &self.exclude_types {
|
||||
Some(xtypes) => Some(Self::types_intersection(itypes, xtypes)),
|
||||
None => Some(itypes.to_string()),
|
||||
},
|
||||
None => self.exclude_types.as_ref().map(|xtypes| xtypes.to_string()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn mixing_include_and_exclude_types_works_as_expected() {
|
||||
let params = Params {
|
||||
types: Some(String::from("js,xml,ts,pdf,tiff")),
|
||||
exclude_types: Some(String::from("js,ts,xml")),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
assert!(params
|
||||
.get_types()
|
||||
.is_some_and(|x| x == "pdf,tiff" || x == "tiff,pdf"))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,10 +1,8 @@
|
||||
use anyhow::Result;
|
||||
use dashmap::DashMap;
|
||||
use indicatif::{
|
||||
MultiProgress, ParallelProgressIterator, ProgressBar, ProgressFinish, ProgressStyle,
|
||||
};
|
||||
use indicatif::{MultiProgress, ProgressBar, ProgressStyle};
|
||||
use rayon::iter::IntoParallelRefMutIterator;
|
||||
use rayon::prelude::{IntoParallelIterator, ParallelIterator};
|
||||
use std::borrow::Cow;
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::sync::{Arc, Mutex, TryLockError, TryLockResult};
|
||||
use std::time::Duration;
|
||||
@@ -22,57 +20,68 @@ impl Processor {
|
||||
progress_bar_box: Arc<MultiProgress>,
|
||||
max_file_size: Arc<AtomicU64>,
|
||||
seed: i64,
|
||||
sw_sorting_finished: Arc<AtomicBool>,
|
||||
) -> Result<()> {
|
||||
let progress_bar = match app_args.progress {
|
||||
true => progress_bar_box.add(ProgressBar::new_spinner()),
|
||||
false => ProgressBar::hidden(),
|
||||
};
|
||||
|
||||
let keys: Vec<u64> = sw_store.clone().iter().map(|i| *i.key()).collect();
|
||||
|
||||
let progress_style = ProgressStyle::with_template("[{elapsed_precise}] {pos:>7} {msg}")?;
|
||||
progress_bar.set_style(progress_style);
|
||||
progress_bar.enable_steady_tick(Duration::from_millis(50));
|
||||
progress_bar.set_message("files grouped by hash.");
|
||||
|
||||
keys.into_par_iter()
|
||||
.progress_with(progress_bar)
|
||||
.with_finish(ProgressFinish::WithMessage(Cow::from(
|
||||
"files grouped by hash.",
|
||||
)))
|
||||
.for_each(|key| {
|
||||
let group: Vec<FileInfo> = sw_store.get(&key).unwrap().to_vec();
|
||||
if group.len() > 1 {
|
||||
group.into_par_iter().for_each(|file| {
|
||||
let fhash = if app_args.strict {
|
||||
file.hash(seed).expect("hashing file failed.")
|
||||
} else {
|
||||
file.initial_page_hash(seed).expect("hashing file failed.")
|
||||
};
|
||||
loop {
|
||||
let keys: Vec<u64> = sw_store
|
||||
.clone()
|
||||
.iter()
|
||||
.filter(|i| !i.value().iter().all(|x| x.is_sw_processed()))
|
||||
.filter(|i| i.value().len() > 1)
|
||||
.map(|i| *i.key())
|
||||
.collect();
|
||||
|
||||
Self::compare_and_update_max_path_len(
|
||||
max_file_size.clone(),
|
||||
file.path.to_string_lossy().len() as u64,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
hw_store
|
||||
.entry(fhash)
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file]);
|
||||
});
|
||||
if keys.is_empty() {
|
||||
match sw_sorting_finished.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
true => {
|
||||
progress_bar.finish_with_message("files grouped by hash.");
|
||||
break Ok(());
|
||||
}
|
||||
false => continue,
|
||||
}
|
||||
});
|
||||
} else {
|
||||
keys.into_par_iter().for_each(|key| {
|
||||
let mut group: Vec<FileInfo> = sw_store.get(&key).unwrap().to_vec();
|
||||
if group.len() > 1 {
|
||||
group.par_iter_mut().for_each(|file| {
|
||||
progress_bar.inc(1);
|
||||
file.sw_processed();
|
||||
|
||||
Ok(())
|
||||
let fhash = match app_args.strict {
|
||||
true => file.hash(seed).expect("hashing file failed."),
|
||||
false => file.initpages_hash(seed).expect("hashing file failed."),
|
||||
};
|
||||
|
||||
Self::compare_and_update_max_path_len(
|
||||
max_file_size.clone(),
|
||||
file.path.to_string_lossy().len() as u64,
|
||||
);
|
||||
|
||||
hw_store
|
||||
.entry(fhash)
|
||||
.and_modify(|fileset| fileset.push(file.clone()))
|
||||
.or_insert_with(|| vec![file.clone()]);
|
||||
});
|
||||
};
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn compare_and_update_max_path_len(current: Arc<AtomicU64>, next: u64) -> Result<()> {
|
||||
pub fn compare_and_update_max_path_len(current: Arc<AtomicU64>, next: u64) {
|
||||
if current.load(Ordering::Relaxed) < next {
|
||||
current.store(next, Ordering::Release);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn sizewise(
|
||||
@@ -144,9 +153,9 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_init_page_only_strict_mode() -> Result<()> {
|
||||
fn hashwise_sorting_two_files_with_identical_init_pages_only_strict_mode() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(4096);
|
||||
let content = generate_bytes(16384);
|
||||
|
||||
let mut content_x = content.clone();
|
||||
let mut content_y = content.clone();
|
||||
@@ -193,6 +202,7 @@ mod tests {
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 2);
|
||||
@@ -201,9 +211,9 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hashwise_sorting_two_files_with_identical_init_page_only_fast_mode() -> Result<()> {
|
||||
fn hashwise_sorting_two_files_with_identical_init_pages_only_fast_mode() -> Result<()> {
|
||||
let root = TempDir::new()?;
|
||||
let content = generate_bytes(4096);
|
||||
let content = generate_bytes(16384);
|
||||
|
||||
let mut content_x = content.clone();
|
||||
let mut content_y = content.clone();
|
||||
@@ -245,6 +255,7 @@ mod tests {
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 1);
|
||||
@@ -290,6 +301,7 @@ mod tests {
|
||||
Arc::new(MultiProgress::new()),
|
||||
Arc::new(AtomicU64::new(32)),
|
||||
300,
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
)?;
|
||||
|
||||
assert_eq!(hw_dupstore.len(), 1);
|
||||
|
||||
@@ -43,51 +43,61 @@ impl Server {
|
||||
}
|
||||
|
||||
let app_args_clone_for_sc = self.app_args.clone();
|
||||
let app_args_clone_for_pr = self.app_args.clone();
|
||||
let app_args_clone_for_sw = self.app_args.clone();
|
||||
let app_args_clone_for_hw = self.app_args.clone();
|
||||
let file_queue_clone_sc = self.filequeue.clone();
|
||||
let file_queue_clone_pr = self.filequeue.clone();
|
||||
let scanner_finished = Arc::new(AtomicBool::new(false));
|
||||
let sw_sort_finished = Arc::new(AtomicBool::new(false));
|
||||
|
||||
let sfin_sc_tr_cl = scanner_finished.clone();
|
||||
let sfin_pr_tr_cl = scanner_finished.clone();
|
||||
|
||||
let swfin_pr_tr_sw = sw_sort_finished.clone();
|
||||
let swfin_pr_tr_hw = sw_sort_finished.clone();
|
||||
|
||||
let store_dupl_sw_for_sw = self.sw_duplicate_set.clone();
|
||||
let store_dupl_sw_for_hw = self.sw_duplicate_set.clone();
|
||||
let store_dupl_hw = self.hw_duplicate_set.clone();
|
||||
let max_file_path_len_clone = self.max_file_path_len.clone();
|
||||
|
||||
let progbarbox_sc_clone = progbarbox.clone();
|
||||
let progbarbox_pr_clone_for_sw = progbarbox.clone();
|
||||
let progbarbox_pr_clone_for_hw = progbarbox.clone();
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Scanner::new(app_args_clone_for_sc)
|
||||
.unwrap()
|
||||
.expect("unable to initialize scanner.")
|
||||
.scan(file_queue_clone_sc, progbarbox_sc_clone)
|
||||
.unwrap();
|
||||
.expect("scanner failed.");
|
||||
|
||||
sfin_sc_tr_cl.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
});
|
||||
|
||||
let progbarbox_pr_clone = progbarbox.clone();
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Processor::sizewise(
|
||||
app_args_clone_for_pr.clone(),
|
||||
app_args_clone_for_sw,
|
||||
sfin_pr_tr_cl,
|
||||
store_dupl_sw_for_sw,
|
||||
file_queue_clone_pr,
|
||||
progbarbox_pr_clone.clone(),
|
||||
progbarbox_pr_clone_for_sw,
|
||||
)
|
||||
.unwrap();
|
||||
.expect("sizewise scanner failed.");
|
||||
|
||||
swfin_pr_tr_sw.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
});
|
||||
|
||||
self.threadpool.execute(move || {
|
||||
Processor::hashwise(
|
||||
app_args_clone_for_pr,
|
||||
app_args_clone_for_hw,
|
||||
store_dupl_sw_for_hw,
|
||||
store_dupl_hw,
|
||||
progbarbox_pr_clone,
|
||||
progbarbox_pr_clone_for_hw,
|
||||
max_file_path_len_clone,
|
||||
seed,
|
||||
swfin_pr_tr_hw,
|
||||
)
|
||||
.unwrap();
|
||||
.expect("sizewise scanner failed.");
|
||||
});
|
||||
|
||||
progbarbox.clear()?;
|
||||
|
||||
Reference in New Issue
Block a user