2021-05-21 01:56:37 +02:00
|
|
|
/* This file is part of epubgrep.
|
|
|
|
* Copyright © 2021 tastytea <tastytea@tastytea.de>
|
|
|
|
*
|
|
|
|
* This program is free software: you can redistribute it and/or modify
|
|
|
|
* it under the terms of the GNU Affero General Public License as published by
|
|
|
|
* the Free Software Foundation, version 3.
|
|
|
|
*
|
|
|
|
* This program is distributed in the hope that it will be useful,
|
|
|
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
|
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
|
|
* GNU Affero General Public License for more details.
|
|
|
|
*
|
|
|
|
* You should have received zipfile copy of the GNU Affero General Public
|
|
|
|
* License along with this program. If not, see <http://www.gnu.org/licenses/>.
|
|
|
|
*/
|
|
|
|
|
|
|
|
#include "zip.hpp"
|
|
|
|
|
|
|
|
#include "fs-compat.hpp"
|
2021-05-30 21:52:52 +02:00
|
|
|
#include "helpers.hpp"
|
2021-05-31 19:10:54 +02:00
|
|
|
#include "log.hpp"
|
2021-05-21 01:56:37 +02:00
|
|
|
|
|
|
|
#include <archive.h>
|
|
|
|
#include <archive_entry.h>
|
2021-05-21 03:25:42 +02:00
|
|
|
#include <boost/locale/message.hpp>
|
|
|
|
#include <fmt/format.h>
|
|
|
|
#include <fmt/ostream.h> // For compatibility with fmt 4.
|
2021-05-21 01:56:37 +02:00
|
|
|
|
2021-05-23 08:56:58 +02:00
|
|
|
#include <cstdlib>
|
|
|
|
#include <cstring>
|
2021-05-27 21:39:01 +02:00
|
|
|
#include <fstream>
|
|
|
|
#include <stdexcept>
|
2021-05-21 01:56:37 +02:00
|
|
|
#include <string>
|
2021-05-29 15:50:03 +02:00
|
|
|
#include <string_view>
|
2021-05-21 01:56:37 +02:00
|
|
|
#include <vector>
|
|
|
|
|
|
|
|
namespace epubgrep::zip
|
|
|
|
{
|
|
|
|
|
2021-05-21 03:25:42 +02:00
|
|
|
using boost::locale::translate;
|
|
|
|
using fmt::format;
|
|
|
|
|
2021-05-21 01:56:37 +02:00
|
|
|
std::vector<std::string> list(const fs::path &filepath)
|
2021-05-23 08:56:58 +02:00
|
|
|
{
|
|
|
|
auto *zipfile{open_file(filepath)};
|
|
|
|
|
|
|
|
struct archive_entry *entry{};
|
|
|
|
std::vector<std::string> toc;
|
|
|
|
while (archive_read_next_header(zipfile, &entry) == ARCHIVE_OK)
|
|
|
|
{
|
2021-05-28 13:55:11 +02:00
|
|
|
const auto *in_epub_filepath{archive_entry_pathname_utf8(entry)};
|
|
|
|
if (in_epub_filepath == nullptr)
|
|
|
|
{ // If the encoding is broken, we skip the file.
|
2021-05-31 22:43:30 +02:00
|
|
|
LOG(log::sev::warning)
|
2021-05-31 19:10:54 +02:00
|
|
|
<< format(translate("File in {0:s} is damaged. "
|
|
|
|
"Skipping in-EPUB file.\n")
|
|
|
|
.str()
|
2022-08-16 16:26:17 +02:00
|
|
|
.c_str(),
|
2022-08-16 18:10:15 +02:00
|
|
|
filepath.c_str());
|
2021-05-29 15:50:03 +02:00
|
|
|
continue;
|
2021-05-28 13:55:11 +02:00
|
|
|
}
|
|
|
|
toc.emplace_back(in_epub_filepath);
|
2021-06-01 15:32:10 +02:00
|
|
|
DEBUGLOG << "Found in file: " << in_epub_filepath;
|
2021-05-23 08:56:58 +02:00
|
|
|
archive_read_data_skip(zipfile);
|
|
|
|
}
|
2021-05-27 20:11:59 +02:00
|
|
|
|
2021-05-23 08:56:58 +02:00
|
|
|
close_file(zipfile, filepath);
|
|
|
|
|
|
|
|
return toc;
|
|
|
|
}
|
|
|
|
|
|
|
|
std::string read_file(const fs::path &filepath, std::string_view entry_path)
|
|
|
|
{
|
|
|
|
auto *zipfile{open_file(filepath)};
|
|
|
|
|
|
|
|
struct archive_entry *entry{};
|
|
|
|
while (archive_read_next_header(zipfile, &entry) == ARCHIVE_OK)
|
|
|
|
{
|
|
|
|
const auto *path{archive_entry_pathname_utf8(entry)};
|
2021-05-29 15:50:03 +02:00
|
|
|
if (path == nullptr)
|
|
|
|
{ // If the encoding is broken, we skip the file.
|
2021-05-31 22:43:30 +02:00
|
|
|
LOG(log::sev::warning)
|
2021-05-31 19:10:54 +02:00
|
|
|
<< format(translate("File in {0:s} is damaged. "
|
|
|
|
"Skipping in-EPUB file.\n")
|
|
|
|
.str()
|
|
|
|
.data(),
|
2022-08-16 17:59:03 +02:00
|
|
|
filepath.c_str());
|
2021-05-29 15:50:03 +02:00
|
|
|
continue;
|
|
|
|
}
|
2021-05-23 08:56:58 +02:00
|
|
|
if (std::strcmp(path, entry_path.data()) == 0)
|
|
|
|
{
|
|
|
|
const auto length{static_cast<size_t>(archive_entry_size(entry))};
|
|
|
|
std::string filecontents;
|
|
|
|
filecontents.resize(length);
|
|
|
|
auto result_length{static_cast<size_t>(
|
|
|
|
archive_read_data(zipfile, &filecontents[0], length))};
|
|
|
|
|
|
|
|
if (result_length != length)
|
|
|
|
{
|
2021-05-27 20:11:59 +02:00
|
|
|
close_file(zipfile, filepath);
|
|
|
|
|
2022-08-16 17:59:03 +02:00
|
|
|
throw exception{format(
|
|
|
|
translate("Could not read {0:s} in {1:s}.").str().c_str(),
|
|
|
|
entry_path, filepath.string())};
|
2021-05-23 08:56:58 +02:00
|
|
|
}
|
|
|
|
|
2021-05-27 19:49:32 +02:00
|
|
|
close_file(zipfile, filepath);
|
2021-05-27 20:11:59 +02:00
|
|
|
|
2021-05-23 08:56:58 +02:00
|
|
|
return filecontents;
|
|
|
|
}
|
|
|
|
archive_read_data_skip(zipfile);
|
|
|
|
}
|
|
|
|
|
|
|
|
close_file(zipfile, filepath);
|
|
|
|
|
2021-05-29 18:12:56 +02:00
|
|
|
if (entry_path == "META-INF/container.xml")
|
|
|
|
{ // File is probably not an EPUB.
|
2022-08-16 17:59:03 +02:00
|
|
|
exception e{format(translate("{0:s} not found in {1:s}.").str().c_str(),
|
2021-05-23 08:56:58 +02:00
|
|
|
entry_path, filepath.string())};
|
2021-05-29 18:12:56 +02:00
|
|
|
e.code = 1;
|
|
|
|
throw exception{e};
|
|
|
|
}
|
|
|
|
|
2021-05-31 22:43:30 +02:00
|
|
|
LOG(log::sev::warning)
|
2021-05-31 19:10:54 +02:00
|
|
|
<< format(translate("{0:s} not found in {1:s}.").str(), entry_path,
|
|
|
|
filepath.string())
|
|
|
|
<< '\n';
|
2021-05-29 18:12:56 +02:00
|
|
|
return {};
|
2021-05-23 08:56:58 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
struct archive *open_file(const fs::path &filepath)
|
2021-05-21 01:56:37 +02:00
|
|
|
{
|
2021-05-29 12:42:29 +02:00
|
|
|
// Throw exception if we can't open the file.
|
|
|
|
std::ifstream file;
|
|
|
|
file.exceptions(std::ios::failbit);
|
|
|
|
file.open(filepath);
|
2021-05-27 21:39:01 +02:00
|
|
|
file.close();
|
|
|
|
|
2021-05-21 07:05:44 +02:00
|
|
|
auto *zipfile{archive_read_new()};
|
2021-05-21 01:56:37 +02:00
|
|
|
archive_read_support_filter_all(zipfile);
|
2021-05-23 08:56:58 +02:00
|
|
|
archive_read_support_format_zip(zipfile);
|
2021-05-21 01:56:37 +02:00
|
|
|
|
2021-05-21 07:05:44 +02:00
|
|
|
auto result{archive_read_open_filename(zipfile, filepath.c_str(), 10240)};
|
2021-05-21 01:56:37 +02:00
|
|
|
if (result != ARCHIVE_OK)
|
|
|
|
{
|
2021-05-27 20:11:59 +02:00
|
|
|
close_file(zipfile, filepath);
|
|
|
|
|
2022-08-16 17:59:03 +02:00
|
|
|
exception e{format(translate("Could not open {0:s}.").str().c_str(),
|
2021-05-27 21:39:01 +02:00
|
|
|
filepath.string())};
|
|
|
|
e.code = 1;
|
|
|
|
throw exception{e};
|
2021-05-21 01:56:37 +02:00
|
|
|
}
|
|
|
|
|
2021-05-23 08:56:58 +02:00
|
|
|
return zipfile;
|
|
|
|
}
|
2021-05-21 01:56:37 +02:00
|
|
|
|
2021-05-23 08:56:58 +02:00
|
|
|
void close_file(struct archive *zipfile, const fs::path &filepath)
|
|
|
|
{
|
|
|
|
auto result{archive_read_free(zipfile)};
|
2021-05-21 01:56:37 +02:00
|
|
|
if (result != ARCHIVE_OK)
|
|
|
|
{
|
2022-08-16 17:59:03 +02:00
|
|
|
throw exception{
|
|
|
|
format(translate("Could not close {0:s}.").str().c_str(),
|
|
|
|
filepath.string())};
|
2021-05-21 01:56:37 +02:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
} // namespace epubgrep::zip
|