2021-07-08 08:07:01 +00:00
|
|
|
use std::io::{self, Seek, SeekFrom};
|
2020-03-23 14:03:18 +00:00
|
|
|
use std::ops::Range;
|
2020-06-23 10:09:49 +00:00
|
|
|
use std::sync::{Arc, Mutex};
|
2020-06-24 09:57:12 +00:00
|
|
|
use std::task::Context;
|
2020-06-23 10:09:49 +00:00
|
|
|
use std::pin::Pin;
|
2019-08-13 10:59:03 +00:00
|
|
|
|
2020-04-17 12:11:25 +00:00
|
|
|
use anyhow::{bail, format_err, Error};
|
2018-12-31 16:30:08 +00:00
|
|
|
|
2020-06-24 09:57:12 +00:00
|
|
|
use pxar::accessor::{MaybeReady, ReadAt, ReadAtOperation};
|
2019-05-22 13:02:16 +00:00
|
|
|
|
2021-07-08 08:07:01 +00:00
|
|
|
use pbs_datastore::dynamic_index::DynamicIndexReader;
|
2021-07-08 07:17:28 +00:00
|
|
|
use pbs_datastore::read_chunk::ReadChunk;
|
2021-07-08 08:07:01 +00:00
|
|
|
use pbs_datastore::index::IndexFile;
|
2021-07-20 08:57:22 +00:00
|
|
|
use pbs_tools::lru_cache::LruCache;
|
2019-02-27 13:32:34 +00:00
|
|
|
|
2020-03-23 14:03:18 +00:00
|
|
|
struct CachedChunk {
|
|
|
|
range: Range<u64>,
|
|
|
|
data: Vec<u8>,
|
|
|
|
}
|
|
|
|
|
|
|
|
impl CachedChunk {
|
|
|
|
/// Perform sanity checks on the range and data size:
|
|
|
|
pub fn new(range: Range<u64>, data: Vec<u8>) -> Result<Self, Error> {
|
|
|
|
if data.len() as u64 != range.end - range.start {
|
|
|
|
bail!(
|
|
|
|
"read chunk with wrong size ({} != {})",
|
|
|
|
data.len(),
|
|
|
|
range.end - range.start,
|
|
|
|
);
|
|
|
|
}
|
|
|
|
Ok(Self { range, data })
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-06-28 14:35:00 +00:00
|
|
|
pub struct BufferedDynamicReader<S> {
|
|
|
|
store: S,
|
2019-02-12 11:05:33 +00:00
|
|
|
index: DynamicIndexReader,
|
2019-01-05 13:47:56 +00:00
|
|
|
archive_size: u64,
|
|
|
|
read_buffer: Vec<u8>,
|
|
|
|
buffered_chunk_idx: usize,
|
|
|
|
buffered_chunk_start: u64,
|
|
|
|
read_offset: u64,
|
2021-07-20 08:57:22 +00:00
|
|
|
lru_cache: LruCache<usize, CachedChunk>,
|
2020-02-27 14:56:28 +00:00
|
|
|
}
|
|
|
|
|
|
|
|
struct ChunkCacher<'a, S> {
|
|
|
|
store: &'a mut S,
|
|
|
|
index: &'a DynamicIndexReader,
|
|
|
|
}
|
|
|
|
|
2021-07-20 08:57:22 +00:00
|
|
|
impl<'a, S: ReadChunk> pbs_tools::lru_cache::Cacher<usize, CachedChunk> for ChunkCacher<'a, S> {
|
2020-03-23 14:03:18 +00:00
|
|
|
fn fetch(&mut self, index: usize) -> Result<Option<CachedChunk>, Error> {
|
2020-06-26 06:14:45 +00:00
|
|
|
let info = match self.index.chunk_info(index) {
|
|
|
|
Some(info) => info,
|
|
|
|
None => bail!("chunk index out of range"),
|
|
|
|
};
|
2020-03-23 14:03:18 +00:00
|
|
|
let range = info.range;
|
|
|
|
let data = self.store.read_chunk(&info.digest)?;
|
|
|
|
CachedChunk::new(range, data).map(Some)
|
2020-02-27 14:56:28 +00:00
|
|
|
}
|
2019-01-02 13:27:04 +00:00
|
|
|
}
|
|
|
|
|
2019-11-14 09:18:31 +00:00
|
|
|
impl<S: ReadChunk> BufferedDynamicReader<S> {
|
2019-06-28 14:35:00 +00:00
|
|
|
pub fn new(index: DynamicIndexReader, store: S) -> Self {
|
2020-06-12 11:57:56 +00:00
|
|
|
let archive_size = index.index_bytes();
|
2019-01-05 13:47:56 +00:00
|
|
|
Self {
|
2019-06-28 14:35:00 +00:00
|
|
|
store,
|
2019-09-11 10:06:59 +00:00
|
|
|
index,
|
|
|
|
archive_size,
|
2019-11-14 09:18:31 +00:00
|
|
|
read_buffer: Vec::with_capacity(1024 * 1024),
|
2019-01-05 13:47:56 +00:00
|
|
|
buffered_chunk_idx: 0,
|
|
|
|
buffered_chunk_start: 0,
|
|
|
|
read_offset: 0,
|
2021-07-20 08:57:22 +00:00
|
|
|
lru_cache: LruCache::new(32),
|
2019-01-05 13:47:56 +00:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-11-14 09:18:31 +00:00
|
|
|
pub fn archive_size(&self) -> u64 {
|
|
|
|
self.archive_size
|
|
|
|
}
|
2019-01-05 13:47:56 +00:00
|
|
|
|
2019-01-05 16:28:20 +00:00
|
|
|
fn buffer_chunk(&mut self, idx: usize) -> Result<(), Error> {
|
2020-03-23 14:03:18 +00:00
|
|
|
//let (start, end, data) = self.lru_cache.access(
|
|
|
|
let cached_chunk = self.lru_cache.access(
|
2020-02-27 14:56:28 +00:00
|
|
|
idx,
|
|
|
|
&mut ChunkCacher {
|
|
|
|
store: &mut self.store,
|
|
|
|
index: &self.index,
|
|
|
|
},
|
|
|
|
)?.ok_or_else(|| format_err!("chunk not found by cacher"))?;
|
|
|
|
|
|
|
|
// fixme: avoid copy
|
2019-06-13 09:47:23 +00:00
|
|
|
self.read_buffer.clear();
|
2020-03-23 14:03:18 +00:00
|
|
|
self.read_buffer.extend_from_slice(&cached_chunk.data);
|
2019-01-05 16:28:20 +00:00
|
|
|
|
|
|
|
self.buffered_chunk_idx = idx;
|
2019-06-28 14:35:00 +00:00
|
|
|
|
2020-03-23 14:03:18 +00:00
|
|
|
self.buffered_chunk_start = cached_chunk.range.start;
|
2019-01-05 16:28:20 +00:00
|
|
|
//println!("BUFFER {} {}", self.buffered_chunk_start, end);
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-11-14 09:18:31 +00:00
|
|
|
impl<S: ReadChunk> crate::tools::BufferedRead for BufferedDynamicReader<S> {
|
2019-01-05 16:28:20 +00:00
|
|
|
fn buffered_read(&mut self, offset: u64) -> Result<&[u8], Error> {
|
2019-11-14 09:18:31 +00:00
|
|
|
if offset == self.archive_size {
|
|
|
|
return Ok(&self.read_buffer[0..0]);
|
|
|
|
}
|
2019-01-06 08:17:28 +00:00
|
|
|
|
2019-01-05 13:47:56 +00:00
|
|
|
let buffer_len = self.read_buffer.len();
|
2019-01-20 08:39:32 +00:00
|
|
|
let index = &self.index;
|
2019-01-05 13:47:56 +00:00
|
|
|
|
|
|
|
// optimization for sequential read
|
2019-11-14 09:18:31 +00:00
|
|
|
if buffer_len > 0
|
2021-07-08 08:07:01 +00:00
|
|
|
&& ((self.buffered_chunk_idx + 1) < index.index().len())
|
2019-11-14 09:18:31 +00:00
|
|
|
&& (offset >= (self.buffered_chunk_start + (self.read_buffer.len() as u64)))
|
2019-01-05 13:47:56 +00:00
|
|
|
{
|
|
|
|
let next_idx = self.buffered_chunk_idx + 1;
|
|
|
|
let next_end = index.chunk_end(next_idx);
|
|
|
|
if offset < next_end {
|
2019-01-10 10:19:54 +00:00
|
|
|
self.buffer_chunk(next_idx)?;
|
2019-01-05 13:47:56 +00:00
|
|
|
let buffer_offset = (offset - self.buffered_chunk_start) as usize;
|
|
|
|
return Ok(&self.read_buffer[buffer_offset..]);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-11-14 09:18:31 +00:00
|
|
|
if (buffer_len == 0)
|
|
|
|
|| (offset < self.buffered_chunk_start)
|
|
|
|
|| (offset >= (self.buffered_chunk_start + (self.read_buffer.len() as u64)))
|
2019-01-05 13:47:56 +00:00
|
|
|
{
|
2021-07-08 08:07:01 +00:00
|
|
|
let end_idx = index.index().len() - 1;
|
2019-01-05 13:47:56 +00:00
|
|
|
let end = index.chunk_end(end_idx);
|
|
|
|
let idx = index.binary_search(0, 0, end_idx, end, offset)?;
|
2019-01-10 10:19:54 +00:00
|
|
|
self.buffer_chunk(idx)?;
|
2019-11-14 09:18:31 +00:00
|
|
|
}
|
2019-01-05 13:47:56 +00:00
|
|
|
|
|
|
|
let buffer_offset = (offset - self.buffered_chunk_start) as usize;
|
|
|
|
Ok(&self.read_buffer[buffer_offset..])
|
|
|
|
}
|
|
|
|
}
|
2019-01-02 13:27:04 +00:00
|
|
|
|
2019-11-14 09:18:31 +00:00
|
|
|
impl<S: ReadChunk> std::io::Read for BufferedDynamicReader<S> {
|
2019-01-06 09:04:45 +00:00
|
|
|
fn read(&mut self, buf: &mut [u8]) -> Result<usize, std::io::Error> {
|
2019-02-28 08:17:04 +00:00
|
|
|
use crate::tools::BufferedRead;
|
2019-11-14 09:18:31 +00:00
|
|
|
use std::io::{Error, ErrorKind};
|
2019-01-06 09:04:45 +00:00
|
|
|
|
|
|
|
let data = match self.buffered_read(self.read_offset) {
|
|
|
|
Ok(v) => v,
|
|
|
|
Err(err) => return Err(Error::new(ErrorKind::Other, err.to_string())),
|
|
|
|
};
|
|
|
|
|
2019-11-14 09:18:31 +00:00
|
|
|
let n = if data.len() > buf.len() {
|
|
|
|
buf.len()
|
|
|
|
} else {
|
|
|
|
data.len()
|
|
|
|
};
|
2019-01-06 09:04:45 +00:00
|
|
|
|
2020-06-18 11:55:19 +00:00
|
|
|
buf[0..n].copy_from_slice(&data[0..n]);
|
2019-01-06 09:04:45 +00:00
|
|
|
|
|
|
|
self.read_offset += n as u64;
|
|
|
|
|
2019-10-26 09:36:01 +00:00
|
|
|
Ok(n)
|
2019-01-06 09:04:45 +00:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-11-14 09:18:31 +00:00
|
|
|
impl<S: ReadChunk> std::io::Seek for BufferedDynamicReader<S> {
|
2019-06-28 14:35:00 +00:00
|
|
|
fn seek(&mut self, pos: SeekFrom) -> Result<u64, std::io::Error> {
|
2019-01-06 08:35:39 +00:00
|
|
|
let new_offset = match pos {
|
2019-11-14 09:18:31 +00:00
|
|
|
SeekFrom::Start(start_offset) => start_offset as i64,
|
|
|
|
SeekFrom::End(end_offset) => (self.archive_size as i64) + end_offset,
|
2019-01-06 08:35:39 +00:00
|
|
|
SeekFrom::Current(offset) => (self.read_offset as i64) + offset,
|
|
|
|
};
|
|
|
|
|
2019-01-11 07:41:33 +00:00
|
|
|
use std::io::{Error, ErrorKind};
|
2019-01-06 08:35:39 +00:00
|
|
|
if (new_offset < 0) || (new_offset > (self.archive_size as i64)) {
|
|
|
|
return Err(Error::new(
|
|
|
|
ErrorKind::Other,
|
2019-11-14 09:18:31 +00:00
|
|
|
format!(
|
|
|
|
"seek is out of range {} ([0..{}])",
|
|
|
|
new_offset, self.archive_size
|
|
|
|
),
|
|
|
|
));
|
2019-01-06 08:35:39 +00:00
|
|
|
}
|
|
|
|
self.read_offset = new_offset as u64;
|
|
|
|
|
|
|
|
Ok(self.read_offset)
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2020-06-23 10:09:49 +00:00
|
|
|
/// This is a workaround until we have cleaned up the chunk/reader/... infrastructure for better
|
|
|
|
/// async use!
|
|
|
|
///
|
|
|
|
/// Ideally BufferedDynamicReader gets replaced so the LruCache maps to `BroadcastFuture<Chunk>`,
|
|
|
|
/// so that we can properly access it from multiple threads simultaneously while not issuing
|
|
|
|
/// duplicate simultaneous reads over http.
|
|
|
|
#[derive(Clone)]
|
|
|
|
pub struct LocalDynamicReadAt<R: ReadChunk> {
|
|
|
|
inner: Arc<Mutex<BufferedDynamicReader<R>>>,
|
|
|
|
}
|
|
|
|
|
|
|
|
impl<R: ReadChunk> LocalDynamicReadAt<R> {
|
|
|
|
pub fn new(inner: BufferedDynamicReader<R>) -> Self {
|
|
|
|
Self {
|
|
|
|
inner: Arc::new(Mutex::new(inner)),
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2020-06-24 09:57:12 +00:00
|
|
|
impl<R: ReadChunk> ReadAt for LocalDynamicReadAt<R> {
|
|
|
|
fn start_read_at<'a>(
|
|
|
|
self: Pin<&'a Self>,
|
2020-06-23 10:09:49 +00:00
|
|
|
_cx: &mut Context,
|
2020-06-24 09:57:12 +00:00
|
|
|
buf: &'a mut [u8],
|
2020-06-23 10:09:49 +00:00
|
|
|
offset: u64,
|
2020-06-24 09:57:12 +00:00
|
|
|
) -> MaybeReady<io::Result<usize>, ReadAtOperation<'a>> {
|
2020-06-23 10:09:49 +00:00
|
|
|
use std::io::Read;
|
2020-06-24 09:57:12 +00:00
|
|
|
MaybeReady::Ready(tokio::task::block_in_place(move || {
|
2020-06-23 10:09:49 +00:00
|
|
|
let mut reader = self.inner.lock().unwrap();
|
|
|
|
reader.seek(SeekFrom::Start(offset))?;
|
2020-06-24 09:57:12 +00:00
|
|
|
Ok(reader.read(buf)?)
|
|
|
|
}))
|
|
|
|
}
|
|
|
|
|
|
|
|
fn poll_complete<'a>(
|
|
|
|
self: Pin<&'a Self>,
|
|
|
|
_op: ReadAtOperation<'a>,
|
|
|
|
) -> MaybeReady<io::Result<usize>, ReadAtOperation<'a>> {
|
|
|
|
panic!("LocalDynamicReadAt::start_read_at returned Pending");
|
2020-06-23 10:09:49 +00:00
|
|
|
}
|
|
|
|
}
|