summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--.gitignore1
-rw-r--r--src/cli.c1
-rw-r--r--src/cli.h2
-rw-r--r--src/defs.h20
-rw-r--r--src/main.c60
-rw-r--r--src/parse.c264
-rw-r--r--src/parse.h19
7 files changed, 329 insertions, 38 deletions
diff --git a/.gitignore b/.gitignore
index cd42ee3..5958ffc 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,2 +1,3 @@
bin/
obj/
+dump.bin
diff --git a/src/cli.c b/src/cli.c
index 9394424..dd3c263 100644
--- a/src/cli.c
+++ b/src/cli.c
@@ -54,7 +54,6 @@ struct configuration parse_args(int argc, char *argv[]) {
while (current >= 0) {
current = getopt_long(argc, argv, "hv", long_options, &option_index);
-
switch (current) {
case 'h':
printf("Usage: %s [OPTIONS] (ARCHIVE PATH)\n\n", argv[0]);
diff --git a/src/cli.h b/src/cli.h
index 7fac33b..80b4f01 100644
--- a/src/cli.h
+++ b/src/cli.h
@@ -19,7 +19,7 @@
/* Binary structure for storing command-line options. */
struct configuration {
- long long padding;
+ long long padding; // This is temporary.
};
/* General subroutine for parsing command-line arguments. Returns a
diff --git a/src/defs.h b/src/defs.h
new file mode 100644
index 0000000..cafc102
--- /dev/null
+++ b/src/defs.h
@@ -0,0 +1,20 @@
+/* This file is part of Nekopack.
+
+ Copyright (C) 2017 Jakob Tsar-Fox, All Rights Reserved.
+
+ Nekopack is free software: you can redistribute it and/or modify it
+ under the terms of the GNU General Public License as published by the
+ Free Software Foundation, either version 3 of the License, or (at
+ your option) any later version.
+
+ Nekopack is distributed in the hope that it will be useful, but
+ WITHOUT ANY WARRANTY; without even the implied warranty of
+ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ General Public License for more details.
+
+ You should have received a copy of the GNU General Public License
+ along with Nekopack. If not, see <http://www.gnu.org/licenses/>. */
+
+
+#define EXIT_SUCCESS 0
+#define EXIT_FAILURE 1
diff --git a/src/main.c b/src/main.c
index ac50b98..2665b2f 100644
--- a/src/main.c
+++ b/src/main.c
@@ -15,34 +15,33 @@
You should have received a copy of the GNU General Public License
along with Nekopack. If not, see <http://www.gnu.org/licenses/>. */
-#include <inttypes.h>
+#include <inttypes.h> // Needed for debugging at this point.
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
-#include <zlib.h>
-
#include "cli.h"
+#include "parse.h"
#define XP3_MAGIC "XP3\x0d\x0a\x20\x0a\x1a\x8b\x67\x01"
#define XP3_TABLE_OFFSET 11
#define XP3_VERSION_OFFSET 19
-#define EXIT_SUCCESS 0
-#define EXIT_FAILURE 1
-
int is_xp3_archive(FILE *archive);
int get_archive_version(FILE *archive);
uint64_t get_table_offset(FILE *archive, uint8_t archive_version);
-void *inflate_table(FILE *archive, uint64_t table_offset);
int main(int argc, char *argv[]) {
struct configuration arguments = parse_args(argc, argv);
FILE *archive = fopen(argv[1], "rb");
- if (!is_xp3_archive(archive)) {
+ if (archive == NULL) {
+ // perror?
+ fprintf(stderr, "Cannot open file.\n");
+ exit(EXIT_FAILURE);
+ } else if (!is_xp3_archive(archive)) {
fprintf(stderr, "File is not an XP3 archive.\n");
fclose(archive);
exit(EXIT_FAILURE);
@@ -50,7 +49,7 @@ int main(int argc, char *argv[]) {
int archive_version = get_archive_version(archive);
uint64_t table_offset = get_table_offset(archive, archive_version);
- inflate_table(archive, table_offset);
+ extract(archive, table_offset);
/* After the table_is_compressed byte is 8 bytes containing the
compressed size followed by 8 bytes containing the original size
@@ -75,10 +74,15 @@ int main(int argc, char *argv[]) {
whether or not it represents a valid XP3 archive. */
int is_xp3_archive(FILE *archive) {
char* magic_buffer = malloc(11);
+ /* The magic number is at the very beginning of the file, so
+ rewind must be called on the archive's file pointer. */
rewind(archive);
fread(magic_buffer, 11, 1, archive);
- if (memcmp(magic_buffer, XP3_MAGIC, 11))
+ if (memcmp(magic_buffer, XP3_MAGIC, 11)) {
+ free(magic_buffer);
return 0;
+ }
+ free(magic_buffer);
return 1;
}
@@ -88,6 +92,7 @@ int get_archive_version(FILE *archive) {
uint32_t version_word;
fseek(archive, XP3_VERSION_OFFSET, SEEK_SET);
fread(&version_word, sizeof(uint32_t), 1, archive);
+ /* 0x00 indicates version 1, and 0x01 indicates version 2. */
return version_word == 1 ? 2 : 1;
}
@@ -96,38 +101,21 @@ int get_archive_version(FILE *archive) {
minor_version is invalid, the program will exit. */
uint64_t get_table_offset(FILE *archive, uint8_t archive_version) {
fseek(archive, XP3_TABLE_OFFSET, SEEK_SET);
- if (archive_version == 1) {
- uint64_t table_offset;
- fread(&table_offset, sizeof(uint64_t), 1, archive);
+ uint64_t table_offset;
+ fread(&table_offset, sizeof(uint64_t), 1, archive);
+ if (archive_version == 1)
return table_offset;
- }
- uint64_t additional_header_offset, table_offset;
+ /* Version 2 of XP3 contains a minor version field. */
uint32_t minor_version;
- fread(&additional_header_offset, sizeof(uint64_t), 1, archive);
fread(&minor_version, sizeof(uint32_t), 1, archive);
if (minor_version != 1) {
- fprintf(stderr, "Version not implemented.\n");
- exit(EXIT_FAILURE);
+ fprintf(stderr, "Minor version not implemented.\n");
+ fclose(archive);
}
- fseek(archive, additional_header_offset, SEEK_SET);
+ /* The read table_offset is actually an offset to the real
+ table offset. XP3 Version 2 is a little strange. */
+ fseek(archive, table_offset, SEEK_SET);
fseek(archive, sizeof(uint8_t) + sizeof(uint64_t), SEEK_CUR);
fread(&table_offset, sizeof(uint64_t), 1, archive);
return table_offset;
}
-
-
-/* Primary decompression subroutine. Decides whether or not the archive
- has been compressed, and if it has been - inflates it. */
-void *inflate_table(FILE *archive, uint64_t table_offset) {
- uint8_t compressed;
- uint64_t compressed_size, decompressed_size;
- fseek(archive, table_offset, SEEK_SET);
- fread(&compressed, sizeof(uint8_t), 1, archive);
- fread(&compressed_size, sizeof(uint64_t), 1, archive);
- if (compressed) {
- fread(&decompressed_size, sizeof(uint64_t), 1, archive);
- } else {
- decompressed_size = compressed_size;
- }
- return 0x0;
-}
diff --git a/src/parse.c b/src/parse.c
new file mode 100644
index 0000000..c53a627
--- /dev/null
+++ b/src/parse.c
@@ -0,0 +1,264 @@
+/* This file is part of Nekopack.
+
+ Copyright (C) 2017 Jakob Tsar-Fox, All Rights Reserved.
+
+ Nekopack is free software: you can redistribute it and/or modify it
+ under the terms of the GNU General Public License as published by the
+ Free Software Foundation, either version 3 of the License, or (at
+ your option) any later version.
+
+ Nekopack is distributed in the hope that it will be useful, but
+ WITHOUT ANY WARRANTY; without even the implied warranty of
+ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ General Public License for more details.
+
+ You should have received a copy of the GNU General Public License
+ along with Nekopack. If not, see <http://www.gnu.org/licenses/>. */
+
+#include <inttypes.h> // Needed for debugging at this point.
+#include <stdio.h>
+#include <stdint.h>
+#include <stdlib.h>
+#include <string.h>
+
+#include <iconv.h>
+#include <zlib.h>
+
+#include "defs.h"
+
+#define ELIF_MAGIC 0x46696c65
+#define FILE_MAGIC 0x656c6946
+#define ADLR_MAGIC 0x726c6461
+#define SEGM_MAGIC 0x6d676573
+#define INFO_MAGIC 0x6f666e69
+#define TIME_MAGIC 0x656d6974
+
+/* A pointer to the start of the memory region
+ has to be kept for freeing purposes. */
+typedef struct {
+ uint64_t stream_length;
+ Bytef *start;
+ Bytef *data;
+} memory_stream;
+
+/* typedef struct { */
+/* uint64_t timestamp; */
+/* uint32_t hash_key; */
+/* } file_entry; */
+
+memory_stream decompress_stream(FILE *archive, uint64_t sizes_offset);
+void read_stream(void *destination, Bytef **source, size_t size);
+void read_file_entry(memory_stream *data_stream, Bytef *section_end);
+void read_elif_entry(memory_stream *data_stream);
+void read_info_chunk(memory_stream *data_stream);
+
+
+/* Parses and extracts entries from the archive. */
+void extract(FILE *archive, uint64_t table_offset) {
+ /* An 8-bit unsigned integers at the table offset indicates
+ whether or not the archive has to be decompressed. */
+ uint8_t compressed;
+ fseek(archive, table_offset, SEEK_SET);
+ fread(&compressed, sizeof(uint8_t), 1, archive);
+
+ memory_stream data_stream;
+ if (compressed) {
+ data_stream = decompress_stream(archive, ftell(archive));
+ } else {
+ fprintf(stderr, "Uncompressed archives not yet supported.\n");
+ exit(EXIT_FAILURE);
+ }
+
+ /* The header for every entry in the XP3 archive format contains
+ a magic number, followed by the size of the entry. */
+ uint32_t entry_magic;
+ uint64_t entry_size;
+ for (;;) {
+ read_stream(&entry_magic, &data_stream.data, sizeof(uint32_t));
+ read_stream(&entry_size, &data_stream.data, sizeof(uint64_t));
+ printf("entry at 0x%lx\n", data_stream.data - data_stream.start - 12);
+ printf("Magic: %" PRIx32 " Size: %" PRIx64 "\n", entry_magic, entry_size); // Debug.
+
+ switch (entry_magic) {
+ case ELIF_MAGIC:
+ /* The size given by the entry header doesn't match the
+ actual entry size, so it isn't passed to the function. */
+ read_elif_entry(&data_stream);
+ break;
+ case FILE_MAGIC:
+ // printf("[File entry]\n");
+ // data_stream.data += entry_size;
+ read_file_entry(&data_stream, data_stream.data + entry_size);
+ break;
+ default: // Debug.
+ printf("Unknown magic: %x\n", entry_magic);
+ exit(EXIT_FAILURE);
+ }
+ }
+
+ free(data_stream.start);
+}
+
+
+/* Inflates the file pointer and returns a struct containing the
+ size and a pointer to the decompressed data in memory. */
+memory_stream decompress_stream(FILE *archive, uint64_t sizes_offset) {
+ uint64_t compressed_size, decompressed_size;
+ fseek(archive, sizes_offset, SEEK_SET);
+ fread(&compressed_size, sizeof(uint64_t), 1, archive);
+ fread(&decompressed_size, sizeof(uint64_t), 1, archive);
+
+ /* Decompression is done in memory because it's $CURRENT_YEAR. */
+ Bytef *compressed_data = malloc(compressed_size);
+ Bytef *decompressed_data = malloc(decompressed_size);
+
+ /* This is a pretty shitty way of handling it, though. */
+ if (compressed_data == NULL || decompressed_data == NULL) {
+ if (compressed_data != NULL)
+ free(compressed_data);
+ fprintf(stderr, "Insufficient memory to decompress archive.\n");
+ fclose(archive);
+ exit(EXIT_FAILURE);
+ }
+
+ z_stream data_stream;
+ data_stream.zalloc = Z_NULL;
+ data_stream.zfree = Z_NULL;
+ data_stream.opaque = Z_NULL;
+ data_stream.avail_in = 0;
+ data_stream.next_in = Z_NULL;
+
+ if (inflateInit(&data_stream) != Z_OK) {
+ fprintf(stderr, "Could not initialize zlib.\n");
+ free(compressed_data);
+ free(decompressed_data);
+ fclose(archive);
+ exit(EXIT_FAILURE);
+ }
+
+ int status_code;
+ do {
+ fread(compressed_data, compressed_size, 1, archive);
+ data_stream.avail_in = compressed_size;
+
+ /* This really shouldn't happen. */
+ if (ferror(archive)) {
+ fprintf(stderr, "File corrupt.\n");
+ inflateEnd(&data_stream);
+ free(compressed_data);
+ free(decompressed_data);
+ exit(EXIT_FAILURE);
+ }
+
+ if (data_stream.avail_in == 0)
+ break;
+ data_stream.next_in = compressed_data;
+ // Flushing probably isn't required here.
+ do {
+ data_stream.avail_out = decompressed_size;
+ data_stream.next_out = decompressed_data;
+ status_code = inflate(&data_stream, Z_NO_FLUSH);
+ } while (data_stream.avail_out == 0);
+ } while (status_code != Z_STREAM_END);
+
+ /* The compressed data is irrelevant at this point. */
+ free(compressed_data);
+
+ return (memory_stream) {decompressed_size, decompressed_data,
+ decompressed_data};
+}
+
+
+/* Wrapper for memcpy which increments the source operand by
+ the amount of bytes read to simulate a file stream. */
+void read_stream(void *destination, Bytef **source, size_t size) {
+ memcpy(destination, *source, size);
+ *source += size;
+}
+
+
+/* Document */
+void read_file_entry(memory_stream *data_stream, Bytef *section_end) {
+ uint32_t entry_magic;
+ uint64_t entry_size;
+ while (data_stream->data < section_end) {
+ read_stream(&entry_magic, &data_stream->data, sizeof(uint32_t));
+ read_stream(&entry_size, &data_stream->data, sizeof(uint64_t));
+ switch (entry_magic) {
+ case ADLR_MAGIC:
+ printf("[ADLR found at 0x%lx]\n", data_stream->data - data_stream->start);
+ data_stream->data += entry_size;
+ break;
+ case SEGM_MAGIC:
+ printf("[SEGM found at 0x%lx]\n", data_stream->data - data_stream->start);
+ data_stream->data += entry_size;
+ break;
+ case INFO_MAGIC:
+ printf("[INFO found at 0x%lx]\n", data_stream->data - data_stream->start);
+ read_info_chunk(data_stream);
+ /* data_stream->data += entry_size; */
+ break;
+ case TIME_MAGIC:
+ printf("[TIME found at 0x%lx]\n", data_stream->data - data_stream->start);
+ data_stream->data += entry_size;
+ break;
+ default:
+ printf("New magic discovered: %" PRIx32 " size: %" PRIx64 "\n", entry_magic, entry_size); // Debug.
+ }
+ }
+ printf("\n\n\n\n"); // Debug.
+}
+
+
+/* Document */
+void read_elif_entry(memory_stream *data_stream) {
+ uint16_t name_size;
+ /* The first part of an ELIF entry is a 32-bit file name hash. */
+ data_stream->data += sizeof(uint32_t);
+ read_stream(&name_size, &data_stream->data, sizeof(uint16_t));
+
+ char *input_buffer = malloc(name_size * 2);
+ char *file_name = malloc(name_size);
+ read_stream(input_buffer, &data_stream->data, name_size * 2);
+
+ /* There seems to be an extra UTF-16 byte at the end of the entry. */
+ data_stream->data += 2;
+
+ printf("Filename (ASCII): ");
+ for (int i = 0; i < name_size * 2; i++) {
+ if (input_buffer[i] >= 0x20 && input_buffer[i] < 0x7f)
+ printf("%c", input_buffer[i]);
+ }
+ printf("\n");
+
+ free(input_buffer);
+ free(file_name);
+}
+
+
+/* Document */
+void read_info_chunk(memory_stream *data_stream) {
+ uint32_t flags;
+ uint64_t decompressed_size, compressed_size;
+ uint16_t file_name_size;
+ read_stream(&flags, &data_stream->data, sizeof(uint32_t));
+ read_stream(&decompressed_size, &data_stream->data, sizeof(uint64_t));
+ read_stream(&compressed_size, &data_stream->data, sizeof(uint64_t));
+ read_stream(&file_name_size, &data_stream->data, sizeof(uint16_t));
+
+ char *file_name = malloc(file_name_size * 2);
+ read_stream(file_name, &data_stream->data, file_name_size * 2);
+ printf("\nINFO SEGMENT\n");
+ printf("------------\n");
+ printf("FLAGS: %" PRIx32 "\n", flags);
+ printf("DECOMPRESSED_SIZE: %" PRIx64 "\n", decompressed_size);
+ printf("COMPRESSED_SIZE: %" PRIx64 "\n", compressed_size);
+ printf("FILENAME (HASH?): ");
+ for (int i = 0; i < file_name_size * 2; i++) {
+ if (file_name[i] >= 0x20 && file_name[i] < 0x7f)
+ printf("%c", file_name[i]);
+ }
+ printf("\n");
+
+ data_stream->data += 2;
+}
diff --git a/src/parse.h b/src/parse.h
new file mode 100644
index 0000000..c258711
--- /dev/null
+++ b/src/parse.h
@@ -0,0 +1,19 @@
+/* This file is part of Nekopack.
+
+ Copyright (C) 2017 Jakob Tsar-Fox, All Rights Reserved.
+
+ Nekopack is free software: you can redistribute it and/or modify it
+ under the terms of the GNU General Public License as published by the
+ Free Software Foundation, either version 3 of the License, or (at
+ your option) any later version.
+
+ Nekopack is distributed in the hope that it will be useful, but
+ WITHOUT ANY WARRANTY; without even the implied warranty of
+ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ General Public License for more details.
+
+ You should have received a copy of the GNU General Public License
+ along with Nekopack. If not, see <http://www.gnu.org/licenses/>. */
+
+
+void extract(FILE *archive, uint64_t table_offset);