monolith/src/main.rs

129 lines
3.3 KiB
Rust
Raw Normal View History

2019-08-23 11:49:14 +02:00
#[macro_use]
2019-08-23 05:17:15 +02:00
extern crate clap;
mod args;
2019-12-26 06:41:03 +01:00
mod macros;
2020-01-02 16:31:55 +01:00
use crate::args::AppArgs;
use monolith::html::{html_to_dom, stringify_document, walk_and_embed_assets};
2019-09-29 23:15:49 +02:00
use monolith::http::retrieve_asset;
2020-02-13 06:56:30 +01:00
use monolith::utils::{data_url_to_text, is_data_url, is_http_url};
use reqwest::blocking::Client;
use reqwest::header::{HeaderMap, HeaderValue, USER_AGENT};
use std::collections::HashMap;
use std::fs::File;
use std::io::{self, Error, Write};
use std::process;
use std::time::Duration;
2019-08-23 05:17:15 +02:00
enum Output {
Stdout(io::Stdout),
File(File),
}
2019-12-26 06:41:03 +01:00
impl Output {
fn new(file_path: &str) -> Result<Output, Error> {
if file_path.is_empty() {
Ok(Output::Stdout(io::stdout()))
} else {
Ok(Output::File(File::create(file_path)?))
}
2019-12-26 06:41:03 +01:00
}
fn writeln_str(&mut self, s: &str) -> Result<(), Error> {
match self {
Output::Stdout(stdout) => {
writeln!(stdout, "{}", s)?;
stdout.flush()
}
Output::File(f) => {
writeln!(f, "{}", s)?;
f.flush()
}
}
}
2019-12-26 06:41:03 +01:00
}
2019-08-23 05:17:15 +02:00
fn main() {
let app_args = AppArgs::get();
2020-02-13 06:56:30 +01:00
let target_url: &str = app_args.url_target.as_str();
let base_url;
let dom;
2019-12-26 06:41:03 +01:00
2020-02-13 06:56:30 +01:00
if !is_http_url(target_url) && !is_data_url(target_url) {
eprintln!(
2020-02-14 05:46:08 +01:00
"Only HTTP(S) or data URLs are supported but got: {}",
2020-02-13 06:56:30 +01:00
&target_url
);
process::exit(1);
2019-12-26 06:41:03 +01:00
}
let mut output = Output::new(&app_args.output).expect("Could not prepare output");
// Initialize client
let mut cache = HashMap::new();
let mut header_map = HeaderMap::new();
header_map.insert(
USER_AGENT,
HeaderValue::from_str(&app_args.user_agent).expect("Invalid User-Agent header specified"),
);
let timeout: u64 = if app_args.timeout > 0 {
app_args.timeout
} else {
std::u64::MAX / 4
};
let client = Client::builder()
.timeout(Duration::from_secs(timeout))
.danger_accept_invalid_certs(app_args.insecure)
.default_headers(header_map)
.build()
.expect("Failed to initialize HTTP client");
// Retrieve root document
2020-02-13 06:56:30 +01:00
if is_http_url(target_url) {
let (data, final_url) =
retrieve_asset(&mut cache, &client, target_url, false, "", app_args.silent)
.expect("Could not retrieve assets in HTML");
base_url = final_url;
2020-02-14 05:46:08 +01:00
dom = html_to_dom(&data);
2020-02-13 06:56:30 +01:00
} else if is_data_url(target_url) {
2020-02-14 05:46:08 +01:00
let text: String = data_url_to_text(target_url);
if text.len() == 0 {
eprintln!("Unsupported data URL input");
process::exit(1);
}
base_url = str!();
dom = html_to_dom(&text);
2020-02-13 06:56:30 +01:00
} else {
process::exit(1);
}
2019-08-23 05:17:15 +02:00
walk_and_embed_assets(
&mut cache,
&client,
2020-02-13 06:56:30 +01:00
&base_url,
&dom.document,
app_args.no_css,
app_args.no_js,
app_args.no_images,
app_args.silent,
app_args.no_frames,
);
let html: String = stringify_document(
&dom.document,
app_args.no_css,
app_args.no_frames,
app_args.no_js,
app_args.no_images,
app_args.isolate,
);
output
.writeln_str(&html)
.expect("Could not write HTML output");
2019-08-23 05:17:15 +02:00
}