#![doc = include_str!("../docs/download.md")]
use crate::utils::logger::Logger;
use anyhow::{anyhow, bail, Context, Result};
use clap::{Arg, ArgAction, Command};
use indicatif::ProgressBar;
use polars::frame::DataFrame;
use polars::prelude::{AnyValue, DataType, Field, Schema};
use rand::rngs::StdRng;
use rand::seq::SliceRandom as _;
use rand::SeedableRng;
use reqwest::blocking::{Client, Response};
use reqwest::header::{HeaderMap, HeaderValue, AUTHORIZATION, USER_AGENT};
use reqwest::Url;
use std::collections::HashSet;
use std::fs::File;
use std::io::{copy, BufRead};
use std::iter::FromIterator as _;
use std::path::{Path, PathBuf};
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::Mutex;
use std::thread::sleep;
use std::time::Duration;
use tracing::{debug, info, warn};
use walkdir::WalkDir;
use zip::ZipArchive;
use crate::utils::csv::*;
use crate::utils::fs::*;
use crate::utils::regex::*;
pub fn cli() -> Command {
Command::new("download")
.about("Downloads all github repositories from a list and keeps only the files that satisfy user defined criteria.")
.long_about(include_str!("../docs/download.md"))
.author("Andrea Gilot <andrea.gilot@it.uu.se>")
.disable_version_flag(true)
.arg(
Arg::new("input")
.short('i')
.long("input")
.value_name("INPUT_FILE.csv")
.help("Path to the input csv file to use. It must be a valid CSV file where the first column is the id of the project, \
the second column is the full name of the project and the third column is the hash of the latest commit. Other columns are ignored.")
.required(true)
)
.arg(
Arg::new("projects")
.short('p')
.long("projects")
.value_name("OUTPUT_FILE_PROJECTS.csv")
.help("Path to the output csv file storing the project statistics.")
.required(false),
)
.arg(
Arg::new("files")
.short('f')
.long("files")
.value_name("OUTPUT_FILE_FILES.csv")
.help("Path to the output csv file storing the file statistics.")
.required(false),
)
.arg(
Arg::new("tokens")
.short('t')
.long("tokens")
.value_name("TOKENS_FILE.csv")
.help("Path to the file containing the GitHub tokens to use. It must be a valid CSV file with one column named 'token' and where every line is a \
valid GitHub token (e.g ghp_Ab0C1D2eFg3hIjk4LM56oPqRsTuvWX7yZa8B).")
.conflicts_with("skip")
.required_unless_present("skip"),
)
.arg(
Arg::new("dest")
.short('d')
.long("dest")
.aliases(["target", "destination"])
.value_name("DESTINATION")
.help("Path to the directory where projects will be downloaded. The directory will be created if it does not exist.")
.required(true)
)
.arg(
Arg::new("keywords")
.short('k')
.long("keywords")
.num_args(1..)
.action(ArgAction::Append)
.value_name("KEYWORDS_FILES.json")
.help("List of files containing the list of extensions and keywords to use. The files must be in JSON format.\n\
The files must have the following structure:\n \
{\n\
\"languages\": [\n\
{\n\
\"name\": \"LanguageName\",\n\
\"extensions\": [\".ext1\", \".ext2\", ...],\n\
\"keywords\": [\"localKeyword1\", \"localKeyword2\", ...] // optional\n\
},\n\
...\n\
],\n\
\"keywords\": [\"globalKeyword1\", \"globalKeyword2\", ...] // optional\n\
}")
.required(true)
)
.arg(
Arg::new("regex")
.long("regex")
.help("Whether to interpret the keywords as regular expressions. If not specified, the keywords are interpreted as whole words to match.")
.default_value("false")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new("case-sensitive")
.long("case-sensitive")
.help("Match the keywords case-sensitively. By default, letter case is ignored when matching keywords. File extensions are always case-sensitive.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new("skip")
.long("skip")
.help("Skip the downloading of the repositories.")
.action(ArgAction::SetTrue)
)
.arg(
Arg::new("count")
.long("count")
.help("Compute statistics on the downloaded projects without deleting any file.")
.action(ArgAction::SetTrue)
)
.arg(
Arg::new("order")
.long("order")
.help("Order in which the projects are processed.")
.value_parser(["random", "sequential"])
.default_value("random")
)
.arg(
Arg::new("force")
.long("force")
.help("Overwrite the log files if they exist.")
.action(ArgAction::SetTrue)
)
.arg(
Arg::new("sub")
.long("sub")
.value_name("NUMBER_OF_PROJECTS")
.help("Number of projects to sample from the input file. \
If not specified, all remaining projects in the input file are used.")
.value_parser(clap::value_parser!(usize))
)
.arg(
Arg::new("threads")
.short('n')
.long("threads")
.value_name("THREADS")
.help("Number of threads to use when not downloading and computing statistic locally instead (with --skip).")
.conflicts_with("tokens")
.default_value("1")
.value_parser(clap::builder::RangedU64ValueParser::<usize>::new().range(1..)),
)
.arg(
Arg::new("seed")
.short('s')
.long("seed")
.value_name("SEED")
.help("Seed used to randomly shuffle the input data.")
.default_value("12393566520031723923")
.value_parser(clap::value_parser!(u64)),
)
}
type FileRow = Vec<String>;
type ProjectRow = Vec<String>;
pub fn run(
input_file_path: &str,
projects_output_path: Option<&str>,
files_output_path: Option<&str>,
target: &str,
tokens_file: Option<&str>,
keywords_file_paths: &[&str],
regex_syntax: bool,
case_sensitive: bool,
skip: bool,
count: bool,
overwrite: bool,
sub: Option<usize>,
seed: u64,
logger: &Logger,
thread: usize,
order: &str,
) -> Result<()> {
let tokens: Vec<String> = if skip {
(0..thread).map(|n| n.to_string()).collect()
} else {
logger.log_tokens(tokens_file.unwrap())? };
let input_file: DataFrame = logger.run_task("Loading input file", || {
open_csv(
input_file_path,
Some(Schema::from_iter(vec![
Field::new("id".into(), DataType::UInt32),
Field::new("name".into(), DataType::String),
Field::new("path".into(), DataType::String),
Field::new("latest_commit".into(), DataType::String),
])),
Some(if skip {
vec!["path"]
} else {
vec!["id", "name", "latest_commit"]
}),
)
})?;
let mut shuffled_idx: Vec<usize> = (0..input_file.height()).collect::<Vec<usize>>();
if order == "random" {
logger.run_task("Loading project IDs in random order", || {
let mut rng: StdRng = SeedableRng::seed_from_u64(seed);
shuffled_idx.shuffle(&mut rng);
Ok(())
})?;
}
let shuffled_rows = shuffled_idx
.into_iter()
.map(|idx| {
let row = input_file.get_row(idx).unwrap().0;
if skip {
match row[0].clone() {
AnyValue::String(path) => Ok((idx, None, path, None)),
_ => Err(idx),
}
} else {
match (row[0].clone(), row[1].clone(), row[2].clone()) {
(
AnyValue::UInt32(id),
AnyValue::String(name),
AnyValue::String(latest_commit),
) => Ok((idx, Some(id), name, Some(latest_commit))),
_ => Err(idx),
}
}
})
.take(match sub {
Some(n) => n,
None => usize::MAX,
});
let n_proj = input_file.height();
info!(" {} projects found.", n_proj);
const MAX_SUBDIRS: usize = 30000;
if !skip {
create_dir(target)?;
for i in 0..(n_proj / MAX_SUBDIRS + 1) {
create_dir(Path::new(target).join(i.to_string()))?;
}
}
let default_project_log_path = format!("{input_file_path}.project_log.csv");
let project_log_path: &str = projects_output_path.unwrap_or(&default_project_log_path);
let previous_results: HashSet<(Option<u32>, Option<String>)> =
logger.run_task("Resuming progress", || {
Ok(if overwrite || !Path::new(&project_log_path).exists() {
HashSet::<(Option<u32>, Option<String>)>::new()
} else {
let project_log_file: CSVFile = CSVFile::new(project_log_path, FileMode::Read)?;
let prev_res: HashSet<(Option<u32>, Option<String>)> = if skip {
project_log_file
.column::<String>(0)?
.into_iter()
.map(|s| (None, Some(s)))
.collect()
} else {
project_log_file
.column::<u32>(0)?
.into_iter()
.map(|id| (Some(id), None))
.collect()
};
prev_res
})
})?;
if previous_results.is_empty() {
info!(" No previously downloaded projects found, starting from scratch.",);
} else {
info!(
" {} projects have already been downloaded",
previous_results.len()
)
}
let keyword_files: KeywordFiles = logger.run_task("Loading keywords", || {
KeywordFiles::new(regex_syntax)
.case_sensitive(case_sensitive)
.add_files(keywords_file_paths, true)
})?;
info!(
" {} languages found in {} keyword files.",
keyword_files.languages().len(),
keyword_files.len()
);
debug!(" Languages: {}", keyword_files.languages().join(", "));
debug!(
" File extensions: {}",
keyword_files.extensions().join(", ")
);
debug!(
" Regexes: {}",
keyword_files
.debug_regexes()
.into_iter()
.map(|(lang, regexes)| format!("{}:\n{}", lang, regexes.join("\n")))
.collect::<Vec<_>>()
.join("\n")
);
let word_counter: Matcher = Matcher::words_matcher();
let mut project_log_file = CSVFile::new(
project_log_path,
if overwrite {
FileMode::Overwrite
} else {
FileMode::Append
},
)?;
let prefix: &[&str] = if skip {
&["path", "files", "loc", "words"]
} else {
&[
"id",
"path",
"name",
"latest_commit",
"files",
"loc",
"words",
]
};
let project_log_headers = prefix
.iter()
.map(|s| s.to_string())
.chain(std::iter::once("files_with_kw".to_string()))
.chain(
keyword_files
.paths
.iter()
.map(|p| format!("files_with_{p}")),
)
.chain(std::iter::once("loc_with_kw".to_string()))
.chain(
keyword_files
.paths
.iter()
.map(|p| format!("loc_of_files_with_{p}")),
)
.chain(std::iter::once("words_with_kw".to_string()))
.chain(
keyword_files
.paths
.iter()
.map(|p| format!("words_of_files_with_{p}")),
)
.chain(keyword_files.paths.iter().cloned());
project_log_file.write_header(project_log_headers)?;
let default_file_log_path = format!("{input_file_path}.file_log.csv");
let file_log_path: &str = files_output_path.unwrap_or(&default_file_log_path);
let mut file_log = CSVFile::new(
file_log_path,
if overwrite {
FileMode::Overwrite
} else {
FileMode::Append
},
)?;
let prefix: &[&str] = if skip {
&["path", "language", "loc", "words"]
} else {
&["id", "name", "language", "loc", "words"]
};
let file_log_headers = prefix
.iter()
.map(|s| s.to_string())
.chain(keyword_files.paths.iter().cloned());
file_log.write_header(file_log_headers)?;
let iter = Mutex::new(shuffled_rows);
let failed = AtomicBool::new(false);
info!("Starting download...");
let n = tokens.len();
debug!("Spawning {n} threads for downloading and processing the repositories.");
let (tx, rx) = crossbeam_channel::unbounded::<Option<Result<(ProjectRow, Vec<FileRow>)>>>();
crossbeam::thread::scope(|s: &crossbeam::thread::Scope<'_>| {
for t in tokens {
let my_tx = tx.clone();
let keyword_files = &keyword_files;
let word_counter = &word_counter;
let iter = &iter;
let previous_results = &previous_results;
let failed = &failed;
s.spawn(move |_| {
loop {
let next_item = if failed.load(Ordering::Relaxed) {
None
} else {
iter.lock().expect("Mutex poisoned").next()
};
match next_item {
Some(row) => {
match row {
Ok((row_nr, id_opt, full_name, last_commit)) => {
let project_path: String = match (last_commit, id_opt) {
(Some(commit), Some(id)) => Path::new(target)
.join((row_nr / MAX_SUBDIRS).to_string())
.join(format!("{id}-{commit}"))
.to_string_lossy()
.into_owned(),
(None, None) => full_name.to_string(),
_ => unreachable!(),
};
let path_opt = if skip {
Some(project_path.clone())
} else {
None
};
if (!skip || Path::new(&project_path).exists())
&& !previous_results.contains(&(id_opt, path_opt))
{
match download_repo(
t.as_str(),
id_opt,
&project_path,
full_name,
last_commit,
keyword_files,
word_counter,
skip,
!count,
) {
Ok(r) => {
let _ = my_tx.send(Some(Ok(r)));
}
Err(e) => {
failed.store(true, Ordering::Relaxed);
let _ = my_tx.send(Some(Err(e)));
break;
}
}
}
}
Err(row_nr) => {
failed.store(true, Ordering::Relaxed);
let _ = my_tx
.send(Some(Err(anyhow!("Could not parse row {row_nr}"))));
break;
}
}
}
None => {
let _ = my_tx.send(None);
break;
}
}
}
anyhow::Ok(())
});
}
drop(tx);
let mut ended_threads: usize = 0;
let progress = ProgressBar::new(n_proj as u64);
progress.set_style(
indicatif::ProgressStyle::default_bar().template("{elapsed} {wide_bar} {percent}%")?,
);
progress.inc(previous_results.len() as u64);
while let Ok(msg) = rx.recv() {
match msg {
Some(msg_content) => {
let (project_row, file_rows) = msg_content?;
project_log_file.write_record(project_row)?;
for row in file_rows {
file_log.write_record(row)?;
}
progress.inc(1);
}
None => {
ended_threads += 1;
if ended_threads == n {
break;
}
}
}
}
progress.finish();
Ok(())
})
.map_err(|e| anyhow!("Thread panicked: {e:?}"))?
}
fn download_repo(
token: &str,
id_opt: Option<u32>,
project_path: &str,
full_name: &str,
last_commit: Option<&str>,
keywords_files: &KeywordFiles,
word_counter: &Matcher,
skip: bool,
delete: bool,
) -> Result<(ProjectRow, Vec<FileRow>)> {
if !skip {
let id = id_opt.with_context(|| {
format!(
"Project {} does not have an id, cannot be downloaded",
full_name
)
})?;
let http_client = reqwest::blocking::Client::builder()
.connect_timeout(Duration::from_secs(10))
.timeout(None)
.pool_idle_timeout(Duration::from_secs(90))
.build()?;
let mut headers = HeaderMap::new();
headers.insert(
AUTHORIZATION,
HeaderValue::from_str(&format!("Bearer {token}"))?,
);
headers.insert(USER_AGENT, HeaderValue::from_static("Scyros"));
let url_str: String = format!(
"https://api.github.com/repositories/{}/zipball/{}",
id,
last_commit.with_context(|| format!(
"Last commit not found for project {full_name} (id: {id})"
))?
);
let url: Url =
reqwest::Url::parse(&url_str).with_context(|| format!("Bad URL {url_str}"))?;
let zip_path: String = format!("{project_path}.zip");
let downloaded: bool = fetch_zipball(&http_client, &url, &headers, &zip_path)
.with_context(|| format!("Could not download repository {full_name} (id: {id})"))?;
if !downloaded {
return Ok((
error_row(id, full_name, last_commit, keywords_files.len()),
Vec::new(),
));
}
delete_dir(project_path, true)?;
let extracted = ZipArchive::new(
File::open(&zip_path).with_context(|| format!("Failed to open archive {zip_path}"))?,
)
.and_then(|mut archive| archive.extract(Path::new(project_path)));
delete_file(&zip_path, true)?;
if let Err(e) = extracted {
warn!("Could not extract repository {full_name} (id: {id}): {e}");
delete_dir(project_path, true)?;
return Ok((
error_row(id, full_name, last_commit, keywords_files.len()),
Vec::new(),
));
}
}
if delete {
for entry in WalkDir::new(project_path)
.contents_first(true)
.into_iter()
.filter_map(Result::ok)
.filter(|e| e.file_type().is_file())
.filter(|e| !keywords_files.path_has_extension(e.path()))
{
delete_file(entry.path(), false)?;
}
for entry in WalkDir::new(project_path)
.contents_first(true)
.into_iter()
.filter_map(Result::ok)
.filter(|e| e.file_type().is_symlink())
{
let path = entry.path();
if path.is_dir() {
delete_dir(path, false)?;
} else {
delete_file(path, false)?;
}
}
}
let mut dir_loc_before_filter: usize = 0;
let mut dir_files_before_filter: usize = 0;
let mut dir_words_before_filter: usize = 0;
let mut file_rows: Vec<FileRow> = Vec::new();
let mut dir_loc_after_filter_any: usize = 0;
let mut dir_loc_after_filter: Vec<usize> = vec![0; keywords_files.len()];
let mut dir_files_after_filter_any: usize = 0;
let mut dir_files_after_filter: Vec<usize> = vec![0; keywords_files.len()];
let mut dir_words_after_filter_any: usize = 0;
let mut dir_words_after_filter: Vec<usize> = vec![0; keywords_files.len()];
let mut dir_matches: Vec<usize> = vec![0; keywords_files.len()];
for (ext, lang) in keywords_files.extensions_to_language.iter() {
let file_list: Vec<PathBuf> = WalkDir::new(project_path)
.into_iter()
.filter_map(Result::ok)
.filter(|e| e.file_type().is_file())
.filter(|e| has_extension(e.path(), ext))
.map(|e| e.into_path())
.collect();
for path in file_list {
if let Ok(file) = &load_file(&path, 1024 * 1024 * 1024) {
let words = match file {
Ok(content) => word_counter.count_matches_in_text(content),
Err(_) => word_counter.count_matches_in_file(&path)?,
};
let loc = match file {
Ok(content) => content.lines().count(),
Err(_) => file_lines_count(&path)?,
};
let matches: Vec<usize> = match file {
Ok(content) => keywords_files.count_matches_in_text(lang, content),
Err(_) => keywords_files.count_matches_in_file(lang, &path)?,
};
dir_files_before_filter += 1;
dir_loc_before_filter += loc;
dir_words_before_filter += words;
if matches.iter().any(|m| m > &0) {
dir_files_after_filter_any += 1;
dir_loc_after_filter_any += loc;
dir_words_after_filter_any += words;
for i in 0..keywords_files.len() {
if matches[i] > 0 {
dir_files_after_filter[i] += 1;
dir_loc_after_filter[i] += loc;
dir_words_after_filter[i] += words;
}
}
for (i, match_count) in matches.iter().enumerate() {
dir_matches[i] += match_count;
}
let path_str = path.to_string_lossy();
let mut row: Vec<String> = Vec::new();
if let Some(id) = id_opt {
row.push(id.to_string());
}
row.push(path_str.to_string());
row.push(lang.to_string());
row.push(loc.to_string());
row.push(words.to_string());
row.extend(matches.iter().map(|m| m.to_string()));
file_rows.push(row);
} else if delete {
delete_file(&path, false)?
}
}
}
}
if delete {
delete_empty_dirs(project_path)?
}
let mut project_row: Vec<String> = Vec::new();
if let Some(id) = id_opt {
project_row.push(id.to_string());
}
project_row.push(project_path.to_string());
if !skip {
project_row.push(full_name.to_string());
let last_commit = last_commit
.with_context(|| format!("Last commit not found for project {full_name}"))?;
project_row.push(last_commit.to_string());
}
project_row.push(dir_files_before_filter.to_string());
project_row.push(dir_loc_before_filter.to_string());
project_row.push(dir_words_before_filter.to_string());
project_row.push(dir_files_after_filter_any.to_string());
project_row.extend(dir_files_after_filter.iter().map(|m| m.to_string()));
project_row.push(dir_loc_after_filter_any.to_string());
project_row.extend(dir_loc_after_filter.iter().map(|m| m.to_string()));
project_row.push(dir_words_after_filter_any.to_string());
project_row.extend(dir_words_after_filter.iter().map(|m| m.to_string()));
project_row.extend(dir_matches.iter().map(|m| m.to_string()));
Ok((project_row, file_rows))
}
fn fetch_zipball(client: &Client, url: &Url, headers: &HeaderMap, zip_path: &str) -> Result<bool> {
const MAX_ATTEMPTS: u32 = 5;
let mut failures: u32 = 0;
loop {
let attempt = || -> Result<Option<bool>> {
let mut response: Response = client.get(url.clone()).headers(headers.clone()).send()?;
let status = response.status();
if let Some(wait) = rate_limit_wait(&response) {
warn!("Rate limit reached: waiting {} s", wait.as_secs());
sleep(wait);
Ok(None)
} else if status.is_server_error() {
bail!("Server error: {status}")
} else if !status.is_success() {
Ok(Some(false))
} else {
copy(
&mut response,
&mut open_file(zip_path, FileMode::Overwrite)?,
)?;
Ok(Some(true))
}
};
match attempt() {
Ok(Some(downloaded)) => return Ok(downloaded),
Ok(None) => (),
Err(e) => {
delete_file(zip_path, true)?;
failures += 1;
if failures >= MAX_ATTEMPTS {
return Err(e.context(format!("{MAX_ATTEMPTS} attempts failed")));
}
warn!("Download of {url} failed ({e}), retrying");
sleep(Duration::from_secs(2u64.pow(failures)));
}
}
}
}
fn rate_limit_wait(response: &Response) -> Option<Duration> {
const MAX_WAIT: Duration = Duration::from_secs(60 * 60);
let header = |name: &str| -> Option<u64> {
response
.headers()
.get(name)?
.to_str()
.ok()?
.trim()
.parse()
.ok()
};
let status = response.status().as_u16();
let token_exhausted = header("x-ratelimit-remaining") == Some(0);
if status != 429 && !(status == 403 && token_exhausted) {
return None;
}
let until_reset = header("x-ratelimit-reset").and_then(|reset| {
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.ok()?
.as_secs();
Some(reset.saturating_sub(now) + 1)
});
let wait = header("retry-after").or(until_reset).unwrap_or(60);
Some(Duration::from_secs(wait).min(MAX_WAIT))
}
fn error_row(
id: u32,
full_name: &str,
last_commit: Option<&str>,
n_kw_files: usize,
) -> Vec<String> {
let mut row: Vec<String> = vec![
id.to_string(),
"error".to_string(),
full_name.to_string(),
last_commit.unwrap_or_default().to_string(),
"0".to_string(), "0".to_string(), "0".to_string(), ];
for _ in 0..3 {
row.push("0".to_string()); row.extend((0..n_kw_files).map(|_| "0".to_string()));
}
row.extend((0..n_kw_files).map(|_| "0".to_string()));
row
}
#[cfg(test)]
mod tests {
use crate::utils::logger::test_logger;
use anyhow::ensure;
use super::*;
const TEST_DATA: &str = "tests/data/phases/download";
fn download_test(
input: &str,
target: Option<&str>,
keywords_files: &[&str],
count: bool,
skip: bool,
) -> Result<()> {
let input_file: String = format!("{TEST_DATA}/{input}");
let output_file_project: String = format!("{input_file}.project_log.csv");
let output_file_file: String = format!("{input_file}.file_log.csv");
ensure!(
std::path::Path::new(&input_file).exists(),
"Input file {input_file} does not exist"
);
delete_file(&output_file_file, true)?;
delete_file(&output_file_project, true)?;
let target_def: String = match target {
Some(t) => format!("target/tests/{t}"),
None => String::new(),
};
if target.is_some() {
delete_dir(&target_def, true)?;
}
let tokens_file: String = "ghtokens.csv".to_string();
for keywords_file in keywords_files {
ensure!(
std::path::Path::new(keywords_file).exists(),
"Keywords file {keywords_file} does not exist"
);
}
run(
&input_file,
None,
None,
&target_def,
Some(&tokens_file),
keywords_files,
false,
false,
skip,
count,
false,
None,
0,
test_logger(),
2,
"random",
)?;
assert_eq!(
CSVFile::new(&output_file_project, FileMode::Read)?.indexed_lines::<String>(0)?,
CSVFile::new(
&format!("{TEST_DATA}/{input}.project_log.csv.expected"),
FileMode::Read
)?
.indexed_lines(0)?
);
delete_file(&output_file_file, false)?;
delete_file(&output_file_project, false)
}
#[test]
#[ignore = "requires network access and valid GitHub tokens"]
fn download_java_scala_float_double() -> Result<()> {
download_test(
"to_download.csv",
Some("java_scala_float_double"),
&[
"tests/data/keywords/java_float.json",
"tests/data/keywords/scala_float.json",
],
false,
false,
)
}
#[test]
fn download_float_local() -> Result<()> {
download_test(
"to_download_local.csv",
None,
&[
"tests/data/keywords/fp_types.json",
"tests/data/keywords/fp_transcendental.json",
"tests/data/keywords/fp_others.json",
"tests/data/keywords/std_math.json",
],
true,
true,
)
}
#[test]
fn download_local() -> Result<()> {
download_test(
"to_download_local_c.csv",
None,
&["tests/data/keywords/c.json"],
true,
true,
)
}
#[test]
#[ignore = "requires network access, valid GitHub tokens, and is fragile (depends on live repository counts)"]
fn download_github_repos_count() -> Result<()> {
download_test(
"to_download_fragile.csv",
Some("github_repos"),
&["tests/data/keywords/rust.json"],
true,
false,
)
}
#[test]
#[ignore = "requires network access and valid GitHub tokens"]
fn download_live_resumes() -> Result<()> {
let input_file = format!("{TEST_DATA}/to_download.csv");
let output_file_project = format!("{input_file}.resume_test.project_log.csv");
let output_file_file = format!("{input_file}.resume_test.file_log.csv");
let keywords_files = &[
"tests/data/keywords/java_float.json",
"tests/data/keywords/scala_float.json",
];
let tokens_file = "ghtokens.csv";
let target = "target/tests/live_resume_test";
delete_file(&output_file_project, true)?;
delete_file(&output_file_file, true)?;
delete_dir(target, true)?;
run(
&input_file,
Some(&output_file_project),
Some(&output_file_file),
target,
Some(tokens_file),
keywords_files,
false,
false,
false,
false,
true,
None,
0,
test_logger(),
1,
"sequential",
)?;
let rows_after_first = CSVFile::new(&output_file_project, FileMode::Read)?
.column::<u32>(0)?
.len();
assert!(rows_after_first > 0, "First run produced no rows");
run(
&input_file,
Some(&output_file_project),
Some(&output_file_file),
target,
Some(tokens_file),
keywords_files,
false,
false,
false,
false,
false,
None,
0,
test_logger(),
1,
"sequential",
)?;
let rows_after_second = CSVFile::new(&output_file_project, FileMode::Read)?
.column::<u32>(0)?
.len();
assert_eq!(
rows_after_first, rows_after_second,
"Second run re-processed already-downloaded repos: expected {rows_after_first} rows, got {rows_after_second}"
);
delete_file(&output_file_project, false)?;
delete_file(&output_file_file, false)?;
delete_dir(target, false)
}
#[test]
fn download_local_resumes() -> Result<()> {
let input_file = format!("{TEST_DATA}/to_download_local_c.csv");
let output_file_project = format!("{input_file}.resume_test.project_log.csv");
let output_file_file = format!("{input_file}.resume_test.file_log.csv");
let keywords_files = &["tests/data/keywords/c.json"];
delete_file(&output_file_project, true)?;
delete_file(&output_file_file, true)?;
run(
&input_file,
Some(&output_file_project),
Some(&output_file_file),
"",
None,
keywords_files,
false,
false,
true,
true,
true,
None,
0,
test_logger(),
1,
"sequential",
)?;
let rows_after_first = CSVFile::new(&output_file_project, FileMode::Read)?
.column::<String>(0)?
.len();
assert!(rows_after_first > 0, "First run produced no rows");
run(
&input_file,
Some(&output_file_project),
Some(&output_file_file),
"",
None,
keywords_files,
false,
false,
true,
true,
false,
None,
0,
test_logger(),
1,
"sequential",
)?;
let rows_after_second = CSVFile::new(&output_file_project, FileMode::Read)?
.column::<String>(0)?
.len();
assert_eq!(
rows_after_first, rows_after_second,
"Second run re-processed already-logged repos: expected {rows_after_first} rows, got {rows_after_second}"
);
delete_file(&output_file_project, false)?;
delete_file(&output_file_file, false)
}
}