2018-12-31 16:30:08 +00:00
|
|
|
use failure::*;
|
|
|
|
|
|
|
|
use super::chunk_store::*;
|
|
|
|
use super::chunker::*;
|
|
|
|
|
2019-01-02 11:54:40 +00:00
|
|
|
use std::io::{Read, Write, BufWriter};
|
2018-12-31 16:30:08 +00:00
|
|
|
use std::fs::File;
|
|
|
|
use std::path::{Path, PathBuf};
|
|
|
|
use std::os::unix::io::AsRawFd;
|
|
|
|
use uuid::Uuid;
|
|
|
|
use chrono::{Local, TimeZone};
|
|
|
|
|
|
|
|
#[repr(C)]
|
|
|
|
pub struct ArchiveIndexHeader {
|
|
|
|
pub magic: [u8; 12],
|
|
|
|
pub version: u32,
|
|
|
|
pub uuid: [u8; 16],
|
|
|
|
pub ctime: u64,
|
2019-01-02 11:56:04 +00:00
|
|
|
reserved: [u8; 4056], // overall size is one page (4096 bytes)
|
2018-12-31 16:30:08 +00:00
|
|
|
}
|
|
|
|
|
2019-01-02 13:27:04 +00:00
|
|
|
|
|
|
|
pub struct ArchiveIndexReader<'a> {
|
|
|
|
store: &'a ChunkStore,
|
|
|
|
file: File,
|
|
|
|
size: usize,
|
|
|
|
filename: PathBuf,
|
|
|
|
index: *const u8,
|
|
|
|
index_entries: usize,
|
|
|
|
uuid: [u8; 16],
|
|
|
|
ctime: u64,
|
|
|
|
}
|
|
|
|
|
|
|
|
impl <'a> Drop for ArchiveIndexReader<'a> {
|
|
|
|
|
|
|
|
fn drop(&mut self) {
|
|
|
|
if let Err(err) = self.unmap() {
|
|
|
|
eprintln!("Unable to unmap file {:?} - {}", self.filename, err);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
impl <'a> ArchiveIndexReader<'a> {
|
|
|
|
|
|
|
|
pub fn open(store: &'a ChunkStore, path: &Path) -> Result<Self, Error> {
|
|
|
|
|
|
|
|
let full_path = store.relative_path(path);
|
|
|
|
|
|
|
|
let mut file = std::fs::File::open(&full_path)?;
|
|
|
|
|
|
|
|
let header_size = std::mem::size_of::<ArchiveIndexHeader>();
|
|
|
|
|
|
|
|
// todo: use static assertion when available in rust
|
|
|
|
if header_size != 4096 { bail!("got unexpected header size for {:?}", path); }
|
|
|
|
|
|
|
|
let mut buffer = vec![0u8; header_size];
|
|
|
|
file.read_exact(&mut buffer)?;
|
|
|
|
|
|
|
|
let header = unsafe { &mut * (buffer.as_ptr() as *mut ArchiveIndexHeader) };
|
|
|
|
|
|
|
|
if header.magic != *b"PROXMOX-AIDX" {
|
|
|
|
bail!("got unknown magic number for {:?}", path);
|
|
|
|
}
|
|
|
|
|
|
|
|
let version = u32::from_le(header.version);
|
|
|
|
if version != 1 {
|
|
|
|
bail!("got unsupported version number ({}) for {:?}", version, path);
|
|
|
|
}
|
|
|
|
|
|
|
|
let ctime = u64::from_le(header.ctime);
|
|
|
|
|
|
|
|
let rawfd = file.as_raw_fd();
|
|
|
|
|
|
|
|
let stat = match nix::sys::stat::fstat(rawfd) {
|
|
|
|
Ok(stat) => stat,
|
|
|
|
Err(err) => bail!("fstat {:?} failed - {}", path, err),
|
|
|
|
};
|
|
|
|
|
|
|
|
let size = stat.st_size as usize;
|
|
|
|
|
|
|
|
let index_size = (size - header_size);
|
|
|
|
if (index_size % 40) != 0 {
|
|
|
|
bail!("got unexpected file size for {:?}", path);
|
|
|
|
}
|
|
|
|
|
|
|
|
let data = unsafe { nix::sys::mman::mmap(
|
|
|
|
std::ptr::null_mut(),
|
|
|
|
index_size,
|
|
|
|
nix::sys::mman::ProtFlags::PROT_READ,
|
|
|
|
nix::sys::mman::MapFlags::MAP_PRIVATE,
|
|
|
|
rawfd,
|
|
|
|
header_size as i64) }? as *const u8;
|
|
|
|
|
|
|
|
Ok(Self {
|
|
|
|
store,
|
|
|
|
filename: full_path,
|
|
|
|
file,
|
|
|
|
size,
|
|
|
|
index: data,
|
|
|
|
index_entries: index_size/40,
|
|
|
|
ctime,
|
|
|
|
uuid: header.uuid,
|
|
|
|
})
|
|
|
|
}
|
|
|
|
|
|
|
|
fn unmap(&mut self) -> Result<(), Error> {
|
|
|
|
|
|
|
|
if self.index == std::ptr::null_mut() { return Ok(()); }
|
|
|
|
|
2019-01-04 08:28:41 +00:00
|
|
|
if let Err(err) = unsafe { nix::sys::mman::munmap(self.index as *mut std::ffi::c_void, self.index_entries*40) } {
|
2019-01-02 13:27:04 +00:00
|
|
|
bail!("unmap file {:?} failed - {}", self.filename, err);
|
|
|
|
}
|
|
|
|
|
|
|
|
self.index = std::ptr::null_mut();
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
2019-01-05 13:47:56 +00:00
|
|
|
#[inline]
|
|
|
|
fn chunk_end(&self, pos: usize) -> u64 {
|
|
|
|
if pos >= self.index_entries {
|
|
|
|
panic!("chunk index out of range");
|
|
|
|
}
|
|
|
|
unsafe { *(self.index.add(pos*40) as *const u64) }
|
|
|
|
}
|
|
|
|
|
|
|
|
#[inline]
|
|
|
|
fn chunk_digest(&self, pos: usize) -> &[u8] {
|
|
|
|
if pos >= self.index_entries {
|
|
|
|
panic!("chunk index out of range");
|
|
|
|
}
|
|
|
|
unsafe { std::slice::from_raw_parts(self.index.add(pos*40+8), 32) }
|
|
|
|
}
|
|
|
|
|
2019-01-02 13:27:04 +00:00
|
|
|
pub fn mark_used_chunks(&self, status: &mut GarbageCollectionStatus) -> Result<(), Error> {
|
|
|
|
|
|
|
|
for pos in 0..self.index_entries {
|
2019-01-05 13:47:56 +00:00
|
|
|
let digest = self.chunk_digest(pos);
|
2019-01-02 13:27:04 +00:00
|
|
|
if let Err(err) = self.store.touch_chunk(digest) {
|
|
|
|
bail!("unable to access chunk {}, required by {:?} - {}",
|
|
|
|
digest_to_hex(digest), self.filename, err);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
Ok(())
|
|
|
|
}
|
2019-01-04 11:50:54 +00:00
|
|
|
|
|
|
|
pub fn dump_catar(&self, mut writer: Box<Write>) -> Result<(), Error> {
|
|
|
|
|
2019-01-04 16:16:56 +00:00
|
|
|
let mut buffer = Vec::with_capacity(1024*1024);
|
|
|
|
|
2019-01-04 11:50:54 +00:00
|
|
|
for pos in 0..self.index_entries {
|
2019-01-05 13:47:56 +00:00
|
|
|
let end = self.chunk_end(pos);
|
|
|
|
let digest = self.chunk_digest(pos);
|
|
|
|
//println!("Dump {:08x}", end );
|
2019-01-04 16:16:56 +00:00
|
|
|
self.store.read_chunk(digest, &mut buffer)?;
|
|
|
|
writer.write_all(&buffer)?;
|
2019-01-04 11:50:54 +00:00
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
2019-01-05 13:47:56 +00:00
|
|
|
|
|
|
|
fn binary_search(
|
|
|
|
&self,
|
|
|
|
start_idx: usize,
|
|
|
|
start: u64,
|
|
|
|
end_idx: usize,
|
|
|
|
end: u64,
|
|
|
|
offset: u64
|
|
|
|
) -> Result<usize, Error> {
|
|
|
|
|
|
|
|
if (offset >= end) || (offset < start) {
|
|
|
|
bail!("offset out of range");
|
|
|
|
}
|
|
|
|
|
|
|
|
if end_idx == start_idx {
|
|
|
|
return Ok(start_idx); // found
|
|
|
|
}
|
|
|
|
let middle_idx = (start_idx + end_idx)/2;
|
|
|
|
let middle_end = self.chunk_end(middle_idx);
|
|
|
|
|
|
|
|
if offset < middle_end {
|
|
|
|
return self.binary_search(start_idx, start, middle_idx, middle_end, offset);
|
|
|
|
} else {
|
|
|
|
return self.binary_search(middle_idx + 1, middle_end, end_idx, end, offset);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
pub struct BufferedArchiveReader<'a> {
|
|
|
|
index: &'a ArchiveIndexReader<'a>,
|
|
|
|
archive_size: u64,
|
|
|
|
read_buffer: Vec<u8>,
|
|
|
|
buffered_chunk_idx: usize,
|
|
|
|
buffered_chunk_start: u64,
|
|
|
|
read_offset: u64,
|
2019-01-02 13:27:04 +00:00
|
|
|
}
|
|
|
|
|
2019-01-05 13:47:56 +00:00
|
|
|
impl <'a> BufferedArchiveReader<'a> {
|
|
|
|
|
|
|
|
pub fn new(index: &'a ArchiveIndexReader) -> Self {
|
|
|
|
|
|
|
|
let archive_size = index.chunk_end(index.index_entries - 1);
|
|
|
|
Self {
|
|
|
|
index: index,
|
|
|
|
archive_size: archive_size,
|
|
|
|
read_buffer: Vec::with_capacity(1024*1024),
|
|
|
|
buffered_chunk_idx: 0,
|
|
|
|
buffered_chunk_start: 0,
|
|
|
|
read_offset: 0,
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
pub fn archive_size(&self) -> u64 { self.archive_size }
|
|
|
|
|
2019-01-05 16:28:20 +00:00
|
|
|
fn buffer_chunk(&mut self, idx: usize) -> Result<(), Error> {
|
|
|
|
|
|
|
|
let index = self.index;
|
|
|
|
let end = index.chunk_end(idx);
|
|
|
|
let digest = index.chunk_digest(idx);
|
|
|
|
index.store.read_chunk(digest, &mut self.read_buffer)?;
|
|
|
|
|
|
|
|
self.buffered_chunk_idx = idx;
|
|
|
|
self.buffered_chunk_start = end - (self.read_buffer.len() as u64);
|
|
|
|
//println!("BUFFER {} {}", self.buffered_chunk_start, end);
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
impl <'a> crate::tools::BufferedReader for BufferedArchiveReader<'a> {
|
|
|
|
|
|
|
|
fn buffered_read(&mut self, offset: u64) -> Result<&[u8], Error> {
|
2019-01-05 13:47:56 +00:00
|
|
|
|
2019-01-06 08:17:28 +00:00
|
|
|
if offset == self.archive_size { return Ok(&self.read_buffer[0..0]); }
|
|
|
|
|
2019-01-05 13:47:56 +00:00
|
|
|
let buffer_len = self.read_buffer.len();
|
|
|
|
let index = self.index;
|
|
|
|
|
|
|
|
// optimization for sequential read
|
|
|
|
if buffer_len > 0 &&
|
|
|
|
((self.buffered_chunk_idx + 1) < index.index_entries) &&
|
|
|
|
(offset >= (self.buffered_chunk_start + (self.read_buffer.len() as u64)))
|
|
|
|
{
|
|
|
|
let next_idx = self.buffered_chunk_idx + 1;
|
|
|
|
let next_end = index.chunk_end(next_idx);
|
|
|
|
if offset < next_end {
|
|
|
|
self.buffer_chunk(next_idx);
|
|
|
|
let buffer_offset = (offset - self.buffered_chunk_start) as usize;
|
|
|
|
return Ok(&self.read_buffer[buffer_offset..]);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
if (buffer_len == 0) ||
|
|
|
|
(offset < self.buffered_chunk_start) ||
|
|
|
|
(offset >= (self.buffered_chunk_start + (self.read_buffer.len() as u64)))
|
|
|
|
{
|
|
|
|
let end_idx = index.index_entries - 1;
|
|
|
|
let end = index.chunk_end(end_idx);
|
|
|
|
let idx = index.binary_search(0, 0, end_idx, end, offset)?;
|
|
|
|
self.buffer_chunk(idx);
|
|
|
|
}
|
|
|
|
|
|
|
|
let buffer_offset = (offset - self.buffered_chunk_start) as usize;
|
|
|
|
Ok(&self.read_buffer[buffer_offset..])
|
|
|
|
}
|
|
|
|
|
|
|
|
}
|
2019-01-02 13:27:04 +00:00
|
|
|
|
2019-01-06 08:35:39 +00:00
|
|
|
impl <'a> std::io::Seek for BufferedArchiveReader<'a> {
|
|
|
|
|
|
|
|
fn seek(&mut self, pos: std::io::SeekFrom) -> Result<u64, std::io::Error> {
|
|
|
|
|
|
|
|
use std::io::{SeekFrom, Error, ErrorKind};
|
|
|
|
|
|
|
|
let new_offset = match pos {
|
|
|
|
SeekFrom::Start(start_offset) => start_offset as i64,
|
|
|
|
SeekFrom::End(end_offset) => (self.archive_size as i64)+ end_offset,
|
|
|
|
SeekFrom::Current(offset) => (self.read_offset as i64) + offset,
|
|
|
|
};
|
|
|
|
|
|
|
|
if (new_offset < 0) || (new_offset > (self.archive_size as i64)) {
|
|
|
|
return Err(Error::new(
|
|
|
|
ErrorKind::Other,
|
|
|
|
format!("seek is out of range {} ([0..{}])", new_offset, self.archive_size)));
|
|
|
|
}
|
|
|
|
self.read_offset = new_offset as u64;
|
|
|
|
|
|
|
|
Ok(self.read_offset)
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2018-12-31 16:30:08 +00:00
|
|
|
pub struct ArchiveIndexWriter<'a> {
|
|
|
|
store: &'a ChunkStore,
|
|
|
|
chunker: Chunker,
|
2019-01-02 11:54:40 +00:00
|
|
|
writer: BufWriter<File>,
|
2019-01-02 10:02:56 +00:00
|
|
|
closed: bool,
|
2018-12-31 16:30:08 +00:00
|
|
|
filename: PathBuf,
|
|
|
|
tmp_filename: PathBuf,
|
|
|
|
uuid: [u8; 16],
|
|
|
|
ctime: u64,
|
|
|
|
|
|
|
|
chunk_offset: usize,
|
|
|
|
last_chunk: usize,
|
|
|
|
chunk_buffer: Vec<u8>,
|
|
|
|
}
|
|
|
|
|
|
|
|
impl <'a> ArchiveIndexWriter<'a> {
|
|
|
|
|
|
|
|
pub fn create(store: &'a ChunkStore, path: &Path, chunk_size: usize) -> Result<Self, Error> {
|
|
|
|
|
|
|
|
let full_path = store.relative_path(path);
|
|
|
|
let mut tmp_path = full_path.clone();
|
|
|
|
tmp_path.set_extension("tmp_aidx");
|
|
|
|
|
|
|
|
let mut file = std::fs::OpenOptions::new()
|
|
|
|
.create(true).truncate(true)
|
|
|
|
.read(true)
|
|
|
|
.write(true)
|
|
|
|
.open(&tmp_path)?;
|
|
|
|
|
2019-01-02 11:54:40 +00:00
|
|
|
let mut writer = BufWriter::with_capacity(1024*1024, file);
|
|
|
|
|
2018-12-31 16:30:08 +00:00
|
|
|
let header_size = std::mem::size_of::<ArchiveIndexHeader>();
|
|
|
|
|
|
|
|
// todo: use static assertion when available in rust
|
|
|
|
if header_size != 4096 { panic!("got unexpected header size"); }
|
|
|
|
|
|
|
|
let ctime = std::time::SystemTime::now().duration_since(
|
|
|
|
std::time::SystemTime::UNIX_EPOCH)?.as_secs();
|
|
|
|
|
|
|
|
let uuid = Uuid::new_v4();
|
|
|
|
|
|
|
|
let mut buffer = vec![0u8; header_size];
|
|
|
|
let header = crate::tools::map_struct_mut::<ArchiveIndexHeader>(&mut buffer)?;
|
|
|
|
|
|
|
|
header.magic = *b"PROXMOX-AIDX";
|
|
|
|
header.version = u32::to_le(1);
|
|
|
|
header.ctime = u64::to_le(ctime);
|
|
|
|
header.uuid = *uuid.as_bytes();
|
|
|
|
|
2019-01-02 11:54:40 +00:00
|
|
|
writer.write_all(&buffer)?;
|
2018-12-31 16:30:08 +00:00
|
|
|
|
|
|
|
Ok(Self {
|
|
|
|
store,
|
|
|
|
chunker: Chunker::new(chunk_size),
|
2019-01-02 11:54:40 +00:00
|
|
|
writer: writer,
|
2019-01-02 10:02:56 +00:00
|
|
|
closed: false,
|
2018-12-31 16:30:08 +00:00
|
|
|
filename: full_path,
|
|
|
|
tmp_filename: tmp_path,
|
|
|
|
ctime,
|
|
|
|
uuid: *uuid.as_bytes(),
|
|
|
|
|
|
|
|
chunk_offset: 0,
|
|
|
|
last_chunk: 0,
|
|
|
|
chunk_buffer: Vec::with_capacity(chunk_size*4),
|
|
|
|
})
|
|
|
|
}
|
2019-01-02 10:02:56 +00:00
|
|
|
|
|
|
|
pub fn close(&mut self) -> Result<(), Error> {
|
|
|
|
|
|
|
|
if self.closed {
|
|
|
|
bail!("cannot close already closed archive index file {:?}", self.filename);
|
|
|
|
}
|
|
|
|
|
|
|
|
self.closed = true;
|
|
|
|
|
|
|
|
self.write_chunk_buffer()?;
|
|
|
|
|
2019-01-02 11:54:40 +00:00
|
|
|
self.writer.flush()?;
|
2019-01-02 10:02:56 +00:00
|
|
|
|
|
|
|
// fixme:
|
|
|
|
|
|
|
|
if let Err(err) = std::fs::rename(&self.tmp_filename, &self.filename) {
|
|
|
|
bail!("Atomic rename file {:?} failed - {}", self.filename, err);
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
|
|
|
fn write_chunk_buffer(&mut self) -> Result<(), std::io::Error> {
|
|
|
|
|
|
|
|
use std::io::{Error, ErrorKind};
|
|
|
|
|
|
|
|
let chunk_size = self.chunk_buffer.len();
|
|
|
|
|
|
|
|
if chunk_size == 0 { return Ok(()); }
|
|
|
|
|
|
|
|
let expected_chunk_size = self.chunk_offset - self.last_chunk;
|
|
|
|
if expected_chunk_size != self.chunk_buffer.len() {
|
|
|
|
return Err(Error::new(
|
|
|
|
ErrorKind::Other,
|
|
|
|
format!("wrong chunk size {} != {}", expected_chunk_size, chunk_size)));
|
|
|
|
}
|
|
|
|
|
|
|
|
self.last_chunk = self.chunk_offset;
|
|
|
|
|
|
|
|
match self.store.insert_chunk(&self.chunk_buffer) {
|
|
|
|
Ok((is_duplicate, digest)) => {
|
2019-01-02 13:27:04 +00:00
|
|
|
println!("ADD CHUNK {:016x} {} {} {}", self.chunk_offset, chunk_size, is_duplicate, digest_to_hex(&digest));
|
2019-01-04 11:50:54 +00:00
|
|
|
let chunk_end =
|
2019-01-02 11:54:40 +00:00
|
|
|
self.writer.write(unsafe { &std::mem::transmute::<u64, [u8;8]>(self.chunk_offset as u64) })?;
|
|
|
|
self.writer.write(&digest)?;
|
2019-01-02 10:02:56 +00:00
|
|
|
self.chunk_buffer.truncate(0);
|
|
|
|
return Ok(());
|
|
|
|
}
|
|
|
|
Err(err) => {
|
|
|
|
self.chunk_buffer.truncate(0);
|
|
|
|
return Err(Error::new(ErrorKind::Other, err.to_string()));
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
2018-12-31 16:30:08 +00:00
|
|
|
}
|
|
|
|
|
|
|
|
impl <'a> Write for ArchiveIndexWriter<'a> {
|
|
|
|
|
|
|
|
fn write(&mut self, data: &[u8]) -> std::result::Result<usize, std::io::Error> {
|
|
|
|
|
|
|
|
use std::io::{Error, ErrorKind};
|
|
|
|
|
|
|
|
let chunker = &mut self.chunker;
|
|
|
|
|
|
|
|
let pos = chunker.scan(data);
|
|
|
|
|
|
|
|
if pos > 0 {
|
|
|
|
self.chunk_buffer.extend(&data[0..pos]);
|
|
|
|
self.chunk_offset += pos;
|
|
|
|
|
2019-01-02 10:02:56 +00:00
|
|
|
self.write_chunk_buffer()?;
|
|
|
|
Ok(pos)
|
2018-12-31 16:30:08 +00:00
|
|
|
|
|
|
|
} else {
|
|
|
|
self.chunk_offset += data.len();
|
|
|
|
self.chunk_buffer.extend(data);
|
2019-01-02 10:02:56 +00:00
|
|
|
Ok(data.len())
|
2018-12-31 16:30:08 +00:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
fn flush(&mut self) -> std::result::Result<(), std::io::Error> {
|
|
|
|
|
2018-12-31 17:01:07 +00:00
|
|
|
use std::io::{Error, ErrorKind};
|
|
|
|
|
2019-01-02 10:02:56 +00:00
|
|
|
Err(Error::new(ErrorKind::Other, "please use close() instead of flush()"))
|
2018-12-31 16:30:08 +00:00
|
|
|
}
|
|
|
|
}
|