X-Git-Url: http://git.vrable.net/?a=blobdiff_plain;f=bluesky%2Flog.c;h=97b4f304aea1552aa103cd3404c85fde3cb92dae;hb=2f883084c4393dc7b730df2e778b9f2fc14f70e0;hp=a3acf104239fe56e7a24b5d44c39d94d1aac5dc0;hpb=d27b934a06369794d21a3eeaf86c55f942518956;p=bluesky.git diff --git a/bluesky/log.c b/bluesky/log.c index a3acf10..97b4f30 100644 --- a/bluesky/log.c +++ b/bluesky/log.c @@ -41,20 +41,6 @@ #define HEADER_MAGIC 0x676f4c0a #define FOOTER_MAGIC 0x2e435243 -struct log_header { - uint32_t magic; // HEADER_MAGIC - uint8_t type; // Object type + '0' - uint32_t offset; // Starting byte offset of the log header - uint32_t size; // Size of the data item (bytes) - uint64_t inum; // Inode which owns this data, if any - BlueSkyCloudID id; // Object identifier -} __attribute__((packed)); - -struct log_footer { - uint32_t magic; // FOOTER_MAGIC - uint32_t crc; // Computed from log_header to log_footer.magic -} __attribute__((packed)); - static void writebuf(int fd, const char *buf, size_t len) { while (len > 0) { @@ -175,9 +161,15 @@ static gpointer log_thread(gpointer d) item->pending_write |= CLOUDLOG_JOURNAL; bluesky_cloudlog_stats_update(item, 1); + GString *data1 = g_string_new(""); + GString *data2 = g_string_new(""); + GString *data3 = g_string_new(""); + bluesky_serialize_cloudlog(item, data1, data2, data3); + struct log_header header; struct log_footer footer; - size_t size = sizeof(header) + sizeof(footer) + item->data->len; + size_t size = sizeof(header) + sizeof(footer); + size += data1->len + data2->len + data3->len; off_t offset = 0; if (log->fd >= 0) offset = lseek(log->fd, 0, SEEK_CUR); @@ -192,7 +184,9 @@ static gpointer log_thread(gpointer d) header.magic = GUINT32_TO_LE(HEADER_MAGIC); header.offset = GUINT32_TO_LE(offset); - header.size = GUINT32_TO_LE(item->data->len); + header.size1 = GUINT32_TO_LE(data1->len); + header.size2 = GUINT32_TO_LE(data2->len); + header.size3 = GUINT32_TO_LE(data3->len); header.type = item->type + '0'; header.id = item->id; header.inum = GUINT64_TO_LE(item->inum); @@ -203,8 +197,12 @@ static gpointer log_thread(gpointer d) writebuf(log->fd, (const char *)&header, sizeof(header)); crc = crc32c(crc, (const char *)&header, sizeof(header)); - writebuf(log->fd, item->data->data, item->data->len); - crc = crc32c(crc, item->data->data, item->data->len); + writebuf(log->fd, data1->str, data1->len); + crc = crc32c(crc, data1->str, data1->len); + writebuf(log->fd, data2->str, data2->len); + crc = crc32c(crc, data2->str, data2->len); + writebuf(log->fd, data3->str, data3->len); + crc = crc32c(crc, data3->str, data3->len); crc = crc32c(crc, (const char *)&footer, sizeof(footer) - sizeof(uint32_t)); @@ -212,10 +210,15 @@ static gpointer log_thread(gpointer d) writebuf(log->fd, (const char *)&footer, sizeof(footer)); item->log_seq = log->seq_num; - item->log_offset = offset + sizeof(header); - item->log_size = item->data->len; + item->log_offset = offset; + item->log_size = size; + item->data_size = item->data->len; + + offset += size; - offset += sizeof(header) + sizeof(footer) + item->data->len; + g_string_free(data1, TRUE); + g_string_free(data2, TRUE); + g_string_free(data3, TRUE); /* Replace the log item's string data with a memory-mapped copy of the * data, now that it has been written to the log file. (Even if it @@ -225,7 +228,7 @@ static gpointer log_thread(gpointer d) item->data = NULL; bluesky_cloudlog_fetch(item); - log->committed = g_slist_prepend(log->committed, item); + log->committed = g_slist_prepend(log->committed, item); g_atomic_int_add(&item->data_lock_count, -1); g_mutex_unlock(item->lock); @@ -282,6 +285,46 @@ void bluesky_log_finish_all(GList *log_items) } } +/* Return a committed cloud log record that can be used as a watermark for how + * much of the journal has been written. */ +BlueSkyCloudLog *bluesky_log_get_commit_point(BlueSkyFS *fs) +{ + BlueSkyCloudLog *marker = bluesky_cloudlog_new(fs, NULL); + marker->type = LOGTYPE_JOURNAL_MARKER; + marker->data = bluesky_string_new(g_strdup(""), 0); + bluesky_cloudlog_sync(marker); + + g_mutex_lock(marker->lock); + while ((marker->pending_write & CLOUDLOG_JOURNAL)) + g_cond_wait(marker->cond, marker->lock); + g_mutex_unlock(marker->lock); + + return marker; +} + +void bluesky_log_write_commit_point(BlueSkyFS *fs, BlueSkyCloudLog *marker) +{ + BlueSkyCloudLog *commit = bluesky_cloudlog_new(fs, NULL); + commit->type = LOGTYPE_JOURNAL_CHECKPOINT; + + uint32_t seq, offset; + seq = GUINT32_TO_LE(marker->log_seq); + offset = GUINT32_TO_LE(marker->log_offset); + GString *loc = g_string_new(""); + g_string_append_len(loc, (const gchar *)&seq, sizeof(seq)); + g_string_append_len(loc, (const gchar *)&offset, sizeof(offset)); + commit->data = bluesky_string_new_from_gstring(loc); + bluesky_cloudlog_sync(commit); + + g_mutex_lock(commit->lock); + while ((commit->pending_write & CLOUDLOG_JOURNAL)) + g_cond_wait(commit->cond, commit->lock); + g_mutex_unlock(commit->lock); + + bluesky_cloudlog_unref(marker); + bluesky_cloudlog_unref(commit); +} + /* Memory-map the given log object into memory (read-only) and return a pointer * to it. */ static int page_size = 0; @@ -603,7 +646,9 @@ static gboolean validate_journal_item(const char *buf, size_t len, off_t offset) return FALSE; if (GUINT32_FROM_LE(header->offset) != offset) return FALSE; - size_t size = GUINT32_FROM_LE(header->size); + size_t size = GUINT32_FROM_LE(header->size1) + + GUINT32_FROM_LE(header->size2) + + GUINT32_FROM_LE(header->size3); off_t footer_offset = offset + sizeof(struct log_header) + size; if (footer_offset + sizeof(struct log_footer) > len) @@ -627,50 +672,133 @@ static gboolean validate_journal_item(const char *buf, size_t len, off_t offset) /* Scan through a journal segment to extract correctly-written items (those * that pass sanity checks and have a valid checksum). */ -static void bluesky_replay_scan_journal(const char *buf, size_t len) +static void bluesky_replay_scan_journal(const char *buf, size_t len, + uint32_t *seq, uint32_t *start_offset) { const struct log_header *header; off_t offset = 0; while (validate_journal_item(buf, len, offset)) { header = (const struct log_header *)(buf + offset); - size_t size = GUINT32_FROM_LE(header->size); + size_t size = GUINT32_FROM_LE(header->size1) + + GUINT32_FROM_LE(header->size2) + + GUINT32_FROM_LE(header->size3); + + if (header->type - '0' == LOGTYPE_JOURNAL_CHECKPOINT) { + const uint32_t *data = (const uint32_t *)((const char *)header + sizeof(struct log_header)); + *seq = GUINT32_FROM_LE(data[0]); + *start_offset = GUINT32_FROM_LE(data[1]); + } + offset += sizeof(struct log_header) + size + sizeof(struct log_footer); } } +static void reload_item(BlueSkyCloudLog *log_item, + const char *data, + size_t len1, size_t len2, size_t len3) +{ + BlueSkyFS *fs = log_item->fs; + /*const char *data1 = data;*/ + const BlueSkyCloudID *data2 + = (const BlueSkyCloudID *)(data + len1); + /*const BlueSkyCloudPointer *data3 + = (const BlueSkyCloudPointer *)(data + len1 + len2);*/ + + bluesky_string_unref(log_item->data); + log_item->data = NULL; + log_item->location_flags = CLOUDLOG_JOURNAL; + + BlueSkyCloudID id0; + memset(&id0, 0, sizeof(id0)); + + int link_count = len2 / sizeof(BlueSkyCloudID); + GArray *new_links = g_array_new(FALSE, TRUE, sizeof(BlueSkyCloudLog *)); + for (int i = 0; i < link_count; i++) { + BlueSkyCloudID id = data2[i]; + BlueSkyCloudLog *ref = NULL; + if (memcmp(&id, &id0, sizeof(BlueSkyCloudID)) != 0) { + g_mutex_lock(fs->lock); + ref = g_hash_table_lookup(fs->locations, &id); + if (ref != NULL) { + bluesky_cloudlog_ref(ref); + } + g_mutex_unlock(fs->lock); + } + g_array_append_val(new_links, ref); + } + + for (int i = 0; i < log_item->links->len; i++) { + BlueSkyCloudLog *c = g_array_index(log_item->links, + BlueSkyCloudLog *, i); + bluesky_cloudlog_unref(c); + } + g_array_unref(log_item->links); + log_item->links = new_links; +} + static void bluesky_replay_scan_journal2(BlueSkyFS *fs, GList **objects, - int log_seq, + int log_seq, int start_offset, const char *buf, size_t len) { const struct log_header *header; - off_t offset = 0; + off_t offset = start_offset; while (validate_journal_item(buf, len, offset)) { header = (const struct log_header *)(buf + offset); g_print("In replay found valid item at offset %zd\n", offset); - size_t size = GUINT32_FROM_LE(header->size); - - g_mutex_lock(fs->lock); - BlueSkyCloudLog *log_item; - log_item = g_hash_table_lookup(fs->locations, &header->id); - if (log_item == NULL) { - log_item = bluesky_cloudlog_new(fs, &header->id); - g_hash_table_insert(fs->locations, &log_item->id, log_item); - g_mutex_lock(log_item->lock); - } else { - bluesky_cloudlog_ref(log_item); - g_mutex_lock(log_item->lock); - } - g_mutex_unlock(fs->lock); + size_t size = GUINT32_FROM_LE(header->size1) + + GUINT32_FROM_LE(header->size2) + + GUINT32_FROM_LE(header->size3); + + BlueSkyCloudLog *log_item = bluesky_cloudlog_get(fs, header->id); + g_mutex_lock(log_item->lock); *objects = g_list_prepend(*objects, log_item); - bluesky_string_unref(log_item->data); - log_item->location_flags = CLOUDLOG_JOURNAL; - log_item->data = NULL; + log_item->inum = GUINT64_FROM_LE(header->inum); + reload_item(log_item, buf + offset + sizeof(struct log_header), + GUINT32_FROM_LE(header->size1), + GUINT32_FROM_LE(header->size2), + GUINT32_FROM_LE(header->size3)); log_item->log_seq = log_seq; log_item->log_offset = offset + sizeof(struct log_header); - log_item->log_size = header->size; + log_item->log_size = header->size1; + + bluesky_string_unref(log_item->data); + log_item->data = bluesky_string_new(g_memdup(buf + offset + sizeof(struct log_header), GUINT32_FROM_LE(header->size1)), GUINT32_FROM_LE(header->size1)); + + /* For any inodes which were read from the journal, deserialize the + * inode information, overwriting any old inode data. */ + if (header->type - '0' == LOGTYPE_INODE) { + uint64_t inum = GUINT64_FROM_LE(header->inum); + BlueSkyInode *inode; + g_mutex_lock(fs->lock); + inode = (BlueSkyInode *)g_hash_table_lookup(fs->inodes, &inum); + if (inode == NULL) { + inode = bluesky_new_inode(inum, fs, BLUESKY_PENDING); + inode->change_count = 0; + bluesky_insert_inode(fs, inode); + } + g_mutex_lock(inode->lock); + bluesky_inode_free_resources(inode); + if (!bluesky_deserialize_inode(inode, log_item)) + g_print("Error deserializing inode %"PRIu64"\n", inum); + fs->next_inum = MAX(fs->next_inum, inum + 1); + bluesky_list_unlink(&fs->accessed_list, inode->accessed_list); + inode->accessed_list = bluesky_list_prepend(&fs->accessed_list, inode); + bluesky_list_unlink(&fs->dirty_list, inode->dirty_list); + inode->dirty_list = bluesky_list_prepend(&fs->dirty_list, inode); + bluesky_list_unlink(&fs->unlogged_list, inode->unlogged_list); + inode->unlogged_list = NULL; + inode->change_cloud = inode->change_commit; + bluesky_cloudlog_ref(log_item); + bluesky_cloudlog_unref(inode->committed_item); + inode->committed_item = log_item; + g_mutex_unlock(inode->lock); + g_mutex_unlock(fs->lock); + } + bluesky_string_unref(log_item->data); + log_item->data = NULL; g_mutex_unlock(log_item->lock); offset += sizeof(struct log_header) + size + sizeof(struct log_footer); @@ -685,6 +813,7 @@ void bluesky_replay(BlueSkyFS *fs) /* Scan through log files in reverse order to find the most recent commit * record. */ logfiles = g_list_reverse(logfiles); + uint32_t seq_num = 0, start_offset = 0; while (logfiles != NULL) { char *filename = g_strdup_printf("%s/%s", log->log_directory, (char *)logfiles->data); @@ -694,13 +823,16 @@ void bluesky_replay(BlueSkyFS *fs) g_warning("Mapping logfile %s failed!\n", filename); } else { bluesky_replay_scan_journal(g_mapped_file_get_contents(map), - g_mapped_file_get_length(map)); + g_mapped_file_get_length(map), + &seq_num, &start_offset); g_mapped_file_unref(map); } g_free(filename); g_free(logfiles->data); logfiles = g_list_delete_link(logfiles, logfiles); + if (seq_num != 0 || start_offset != 0) + break; } g_list_foreach(logfiles, (GFunc)g_free, NULL); g_list_free(logfiles); @@ -711,11 +843,10 @@ void bluesky_replay(BlueSkyFS *fs) * references, so that any objects which were not linked into persistent * filesystem data structures are freed. */ GList *objects = NULL; - int seq_num = 0; while (TRUE) { char *filename = g_strdup_printf("%s/journal-%08d", log->log_directory, seq_num); - g_print("Replaying file %s\n", filename); + g_print("Replaying file %s from offset %d\n", filename, start_offset); GMappedFile *map = g_mapped_file_new(filename, FALSE, NULL); g_free(filename); if (map == NULL) { @@ -723,11 +854,12 @@ void bluesky_replay(BlueSkyFS *fs) break; } - bluesky_replay_scan_journal2(fs, &objects, seq_num, + bluesky_replay_scan_journal2(fs, &objects, seq_num, start_offset, g_mapped_file_get_contents(map), g_mapped_file_get_length(map)); g_mapped_file_unref(map); seq_num++; + start_offset = 0; } while (objects != NULL) {