27 Commits
0.0.3 ... 0.0.5

Author SHA1 Message Date
sreedev
96fe667d3f version 0.0.5 2023-01-09 11:08:20 -05:00
sreedev
c32e64450f bug fix: connection result issue 2023-01-09 11:04:29 -05:00
Sreedev Kodichath
85acbded46 Merge pull request #13 from beeb/panics
Fix some more unhandled errors
2023-01-09 11:03:12 -05:00
Sreedev Kodichath
f12cecc9ee Merge branch 'main' into panics 2023-01-09 11:02:56 -05:00
sreedev
05f513735e tempdir fixes 2023-01-09 11:01:52 -05:00
Valentin Bersier
552f6c73f2 Merge branch 'main' into panics 2023-01-09 16:59:42 +01:00
Valentin Bersier
138b66038f refactor: various clippy fixes 2023-01-09 16:57:43 +01:00
Sreedev Kodichath
d8e1de169d Merge pull request #12 from dhruvasagar/fix/temp_file_path
Fix Temporary File Path #6
2023-01-09 10:48:28 -05:00
Valentin Bersier
bc170f139e style: remove trailing spaces 2023-01-09 16:46:12 +01:00
Valentin Bersier
c76ad81a55 refactor: do no unwrap in database.rs 2023-01-09 16:41:55 +01:00
Valentin Bersier
254e61cabe fix: clippy warnings 2023-01-09 16:27:52 +01:00
Valentin Bersier
a76163f24f style: format 2023-01-09 16:21:13 +01:00
Valentin Bersier
27fef21be0 fix: no unwrap in params.rs 2023-01-09 16:17:50 +01:00
Dhruva Sagar
8b00faf075 Fix Temporary File Path #6
Better cross platform support
2023-01-09 10:53:20 +05:30
sreedev
e158a8267a added authors & updated version 2023-01-08 17:16:46 -05:00
Sreedev Kodichath
e6f93ce3d6 Merge pull request #9 from beeb/scanner
Refactor scanner
2023-01-08 17:11:36 -05:00
Sreedev Kodichath
011da05c59 Merge pull request #10 from beeb/output
Refactor output.rs
2023-01-08 17:11:04 -05:00
beeb
6087abe960 refactor: no need for into_iter 2023-01-08 13:24:08 +01:00
beeb
5eba7cdc30 fix: still print items even if file size or modified time cannot be retrieved 2023-01-08 13:21:47 +01:00
beeb
5a65550e58 refactor: output.rs 2023-01-08 13:20:15 +01:00
beeb
0b5effd06e refactor: no need to consume pathbuf iterator 2023-01-08 13:09:43 +01:00
beeb
cc243f413a refactor: no need to consume iterator items 2023-01-08 13:08:27 +01:00
beeb
ad467d70a8 style: combine use statements for std 2023-01-08 12:20:27 +01:00
beeb
4490ea4c69 style: combine use statements for database 2023-01-08 12:19:56 +01:00
beeb
8ed0b13ff0 style: order imports 2023-01-08 12:19:25 +01:00
beeb
be7e8d38ae refactor: avoid unwraps 2023-01-08 11:48:49 +01:00
beeb
c9694dea09 fix: clippy warnings 2023-01-08 11:22:43 +01:00
11 changed files with 137 additions and 118 deletions

2
Cargo.lock generated
View File

@@ -269,7 +269,7 @@ dependencies = [
[[package]]
name = "deduplicator"
version = "0.0.3"
version = "0.0.5"
dependencies = [
"anyhow",
"chrono",

View File

@@ -1,9 +1,10 @@
[package]
name = "deduplicator"
version = "0.0.3"
version = "0.0.5"
edition = "2021"
description = "find,filter,delete Duplicates"
license = "MIT"
authors = ["Sreedev Kodichath <sreedevpadmakumar@gmail.com>", "Valentin Bersier <vbersier@gmail.com>", "Dhruva Sagar <dhruva.sagar@gmail.com>"]
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html

View File

@@ -1,7 +1,8 @@
use std::time::Duration;
use crossterm::event::{self, KeyCode, KeyEvent};
use anyhow::Result;
use crossterm::event::{self, KeyCode, KeyEvent};
use super::events;
pub struct EventHandler;
@@ -21,8 +22,7 @@ impl EventHandler {
fn handle_keypress(keyevent: KeyEvent) -> Result<events::Event> {
match keyevent.code {
KeyCode::Char('q') => Ok(events::Event::Exit),
_ => Ok(events::Event::Noop)
_ => Ok(events::Event::Noop),
}
}
}

View File

@@ -1,4 +1,4 @@
pub enum Event {
Exit,
Noop
Noop,
}

View File

@@ -1,18 +1,13 @@
mod event_handler;
mod events;
mod ui;
mod formatter;
mod ui;
use std::{io, thread, time::Duration};
use crate::database;
use crate::output;
use crate::params::Params;
use crate::scanner;
use anyhow::{anyhow, Result};
use crossterm::{event, execute, terminal};
use event_handler::EventHandler;
use std::io;
use std::thread;
use std::time::Duration;
use tui::{
backend::CrosstermBackend,
widgets::{Block, Borders, Widget},
@@ -20,19 +15,24 @@ use tui::{
};
use ui::Ui;
use crate::database;
use crate::output;
use crate::params::Params;
use crate::scanner;
pub struct App;
impl App {
pub fn init(app_args: &Params) -> Result<()> {
// let mut term = Self::init_terminal()?;
let connection = database::get_connection(&app_args)?;
let duplicates = scanner::duplicates(&app_args, &connection)?;
let connection = database::get_connection(app_args)?;
let duplicates = scanner::duplicates(app_args, &connection)?;
// Self::init_render_loop(&mut term)?;
// Self::cleanup(&mut term)?;
output::print(duplicates, &app_args); /* TODO: APP TUI INIT FUNCTION */
output::print(duplicates, app_args); /* TODO: APP TUI INIT FUNCTION */
Ok(())
}
@@ -56,6 +56,8 @@ impl App {
}
fn init_render_loop(term: &mut Terminal<CrosstermBackend<io::Stdout>>) -> Result<()> {
// this could be simplified with a `while Self::render_cycle(term).is_ok() {}` in the current state, but maybe
// it's good to keep it to handle errors in the future
loop {
match Self::render_cycle(term) {
Ok(_) => continue,

View File

@@ -1,5 +1,6 @@
use anyhow::Result;
use std::io;
use anyhow::Result;
use tui::{
backend::{Backend, CrosstermBackend},
layout::{Constraint, Direction, Layout, Rect},

View File

@@ -1,4 +1,7 @@
use std::env::temp_dir;
use anyhow::Result;
use crate::params::Params;
#[derive(Debug, Clone)]
@@ -7,13 +10,18 @@ pub struct File {
pub hash: String,
}
pub fn get_connection(args: &Params) -> Result<sqlite::Connection, sqlite::Error> {
let connection_url = match args.nocache {
false => "/tmp/deduplicator.db",
true => ":memory:"
};
fn db_connection_url(args: &Params) -> String {
match args.nocache {
true => String::from(":memory:"),
false => {
let temp_dir_path = temp_dir();
format!("{}/deduplicator.db", temp_dir_path.display())
}
}
}
sqlite::open(connection_url).and_then(|conn| {
pub fn get_connection(args: &Params) -> Result<sqlite::Connection, sqlite::Error> {
sqlite::open(db_connection_url(args)).and_then(|conn| {
setup(&conn).ok();
Ok(conn)
})
@@ -30,48 +38,46 @@ pub fn put(file: &File, connection: &sqlite::Connection) -> Result<()> {
"INSERT INTO files (file_identifier, hash) VALUES (\"{}\", \"{}\")",
file.path, file.hash
);
let result = connection.execute(query)?;
Ok(result)
connection.execute(query)?;
Ok(())
}
pub fn indexed_paths(connection: &sqlite::Connection) -> Result<Vec<File>> {
let query = format!(
"SELECT * FROM files"
);
let query = "SELECT * FROM files";
let result: Vec<File> = connection
.prepare(query)?
.into_iter()
.map(|row_result| row_result.unwrap())
.map(|row| {
let path = row.read::<&str, _>("file_identifier").to_string();
let hash = row.read::<i64, _>("hash").to_string();
File { path, hash }
})
.collect();
Ok(result)
}
pub fn duplicate_hashes(connection: &sqlite::Connection, path: &String) -> Result<Vec<File>> {
let query = format!(
"
SELECT a.* FROM files a
JOIN (SELECT file_identifier, hash, COUNT(*)
FROM files
GROUP BY hash
HAVING count(*) > 1 ) b
ON a.hash = b.hash
WHERE a.file_identifier LIKE \"{}%\"
ORDER BY a.file_identifier
", path
);
let result: Vec<File> = connection
.prepare(query)?
.into_iter()
.map(|row_result| row_result.unwrap())
.filter_map(|row_result| row_result.ok())
.map(|row| {
let path = row.read::<&str, _>("file_identifier").to_string();
let hash = row.read::<i64, _>("hash").to_string();
File { path, hash }
})
.collect();
Ok(result)
}
pub fn duplicate_hashes(connection: &sqlite::Connection, path: &str) -> Result<Vec<File>> {
let query = format!(
"
SELECT a.* FROM files a
JOIN (SELECT file_identifier, hash, COUNT(*)
FROM files
GROUP BY hash
HAVING count(*) > 1 ) b
ON a.hash = b.hash
WHERE a.file_identifier LIKE \"{}%\"
ORDER BY a.file_identifier
",
path
);
let result: Vec<File> = connection
.prepare(query)?
.into_iter()
.filter_map(|row_result| row_result.ok())
.map(|row| {
let path = row.read::<&str, _>("file_identifier").to_string();
let hash = row.read::<i64, _>("hash").to_string();

View File

@@ -1,12 +1,13 @@
mod params;
#![allow(unused)] // TODO: remove this once TUI is implemented
mod app;
mod database;
mod output;
mod params;
mod scanner;
mod app;
use anyhow::Result;
use clap::Parser;
use app::App;
use clap::Parser;
#[tokio::main]
async fn main() -> Result<()> {

View File

@@ -1,35 +1,40 @@
use crate::database::File;
use std::{collections::HashMap, fs};
use anyhow::Result;
use chrono::offset::Utc;
use chrono::DateTime;
use colored::Colorize;
use humansize::{format_size, DECIMAL};
use std::{collections::HashMap, fs};
use crate::database::File;
use crate::params::Params;
fn format_path(path: &String, opts: &Params) -> String {
let display_path = path.replace(&opts.get_directory().unwrap(), "");
fn format_path(path: &str, opts: &Params) -> Result<String> {
let display_path = path.replace(&opts.get_directory()?, "");
let text_vec = display_path.chars().collect::<Vec<_>>();
let display_range = if text_vec.len() > 32 {
text_vec[(display_path.len() - 32)..].into_iter().collect::<String>()
text_vec[(display_path.len() - 32)..]
.iter()
.collect::<String>()
} else {
display_path
};
format!("...{}", display_range)
Ok(format!("...{}", display_range))
}
fn file_size(path: &String) -> String {
let mdata = fs::metadata(path).unwrap();
fn file_size(path: &String) -> Result<String> {
let mdata = fs::metadata(path)?;
let formatted_size = format_size(mdata.len(), DECIMAL);
format!("{}", formatted_size)
Ok(formatted_size)
}
fn modified_time(path: &String) -> String {
let mdata = fs::metadata(path).unwrap();
let modified_time: DateTime<Utc> = mdata.modified().unwrap().into();
fn modified_time(path: &String) -> Result<String> {
let mdata = fs::metadata(path)?;
let modified_time: DateTime<Utc> = mdata.modified()?.into();
modified_time.format("%Y-%m-%d %H:%M:%S").to_string()
Ok(modified_time.format("%Y-%m-%d %H:%M:%S").to_string())
}
fn print_divider() {
@@ -50,17 +55,17 @@ pub fn print(duplicates: Vec<File>, opts: &Params) {
dup_index
.entry(file.hash.clone())
.and_modify(|value| value.push(file.clone()))
.or_insert(vec![file]);
.or_insert_with(|| vec![file]);
});
dup_index.into_iter().for_each(|(_, group)| {
group.into_iter().for_each(|file| {
dup_index.iter().for_each(|(_, group)| {
group.iter().for_each(|file| {
println!(
"| {0: <16} | {1: <35} | {2: <16} | {3: <32} |",
file.hash.red(),
format_path(&file.path, opts).yellow(),
file_size(&file.path).blue(),
modified_time(&file.path).blue()
format_path(&file.path, opts).unwrap_or_default().yellow(),
file_size(&file.path).unwrap_or_default().blue(),
modified_time(&file.path).unwrap_or_default().blue()
);
});
print_divider();

View File

@@ -1,7 +1,7 @@
use std::path::PathBuf;
use std::{fs, path::PathBuf};
use anyhow::{anyhow, Result};
use clap::Parser;
use anyhow::Result;
use std::fs;
#[derive(Parser, Debug)]
#[command(author, version, about, long_about = None)]
@@ -19,20 +19,17 @@ pub struct Params {
impl Params {
pub fn get_directory(&self) -> Result<String> {
let dir_string: String = self
let dir_pathbuf: PathBuf = self
.dir
.clone()
.unwrap_or(std::env::current_dir()?)
.as_os_str()
.to_str()
.unwrap()
.to_string();
.into();
let dir_pathbuf = PathBuf::from(&dir_string);
let dir = fs::canonicalize(&dir_pathbuf)?
let dir = fs::canonicalize(dir_pathbuf)?
.as_os_str()
.to_str()
.unwrap()
.ok_or_else(|| anyhow!("Invalid directory"))?
.to_string();
Ok(dir)

View File

@@ -1,55 +1,61 @@
use crate::database;
use crate::{params::Params, database::File};
use std::{fs, path::PathBuf};
use anyhow::Result;
use fxhash::hash32 as hasher;
use glob::glob;
use itertools::Itertools;
use rayon::prelude::*;
use std::fs;
use std::path::PathBuf;
use fxhash::hash32 as hasher;
use crate::{
database::{self, File},
params::Params,
};
pub fn duplicates(app_opts: &Params, connection: &sqlite::Connection) -> Result<Vec<File>> {
let scan_results = scan(app_opts, connection)?;
let base_path = app_opts.get_directory()?;
index_files(scan_results, connection);
index_files(scan_results, connection)?;
database::duplicate_hashes(connection, &base_path)
}
fn get_glob_patterns(opts: &Params, directory: &String) -> Vec<PathBuf> {
fn get_glob_patterns(opts: &Params, directory: &str) -> Vec<PathBuf> {
opts.types
.clone()
.unwrap_or(String::from("*"))
.split(",")
.unwrap_or_else(|| String::from("*"))
.split(',')
.map(|filetype| format!("*.{}", filetype))
.map(|filetype| {
vec![directory.clone(), String::from("**"), filetype]
vec![directory.to_owned(), String::from("**"), filetype]
.iter()
.collect()
})
.collect()
}
fn is_indexed_file(path: &String, indexed: &Vec<File>) -> bool {
fn is_indexed_file(path: impl Into<String>, indexed: &[File]) -> bool {
indexed
.into_iter()
.iter()
.map(|file| file.path.clone())
.contains(path)
.contains(&path.into())
}
fn scan(app_opts: &Params, connection: &sqlite::Connection) -> Result<Vec<String>> {
let directory = app_opts.get_directory()?;
let glob_patterns: Vec<PathBuf> = get_glob_patterns(&app_opts, &directory);
let glob_patterns: Vec<PathBuf> = get_glob_patterns(app_opts, &directory);
let indexed_paths = database::indexed_paths(connection)?;
let files: Vec<String> = glob_patterns
.into_par_iter()
.map(|glob_pattern| glob(&glob_pattern.as_os_str().to_str().unwrap()))
.map(|glob_result| glob_result.unwrap())
.par_iter()
.filter_map(|glob_pattern| glob(glob_pattern.as_os_str().to_str()?).ok())
.flat_map(|file_vec| {
file_vec
.map(|x| x.unwrap().as_os_str().to_str().unwrap().to_string())
.filter_map(|x| Some(x.ok()?.as_os_str().to_str()?.to_string()))
.filter(|fpath| !is_indexed_file(fpath, &indexed_paths))
.filter(|glob_result| fs::metadata(glob_result).unwrap().is_file())
.filter(|glob_result| {
fs::metadata(glob_result)
.map(|f| f.is_file())
.unwrap_or(false)
})
.collect::<Vec<String>>()
})
.collect();
@@ -57,18 +63,18 @@ fn scan(app_opts: &Params, connection: &sqlite::Connection) -> Result<Vec<String
Ok(files)
}
fn index_files(files: Vec<String>, connection: &sqlite::Connection) {
fn index_files(files: Vec<String>, connection: &sqlite::Connection) -> Result<()> {
let hashed: Vec<File> = files
.into_par_iter()
.map(|file| {
let hash = hash_file(&file).unwrap();
database::File { path: file, hash }
.filter_map(|file| {
let hash = hash_file(&file).ok()?;
Some(database::File { path: file, hash })
})
.collect();
hashed.into_iter().for_each(|file| {
database::put(&file, connection).unwrap();
});
hashed
.iter()
.try_for_each(|file| database::put(file, connection))
}
pub fn hash_file(filepath: &str) -> Result<String> {