aboutsummaryrefslogtreecommitdiff
#include "csv.h"

#include <assert.h>
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

// === Compile time options ===

#ifndef CSV_DEFAULT_DELIMITER
#	define CSV_DEFAULT_DELIMITER '"'
#endif

#ifndef CSV_DEFAULT_SEPARATOR
#	define CSV_DEFAULT_SEPARATOR ','
#endif

// === some useful Makros ===

#define UNUSED(x) (void) (x) // mark a parameter as 'unused'
#define STR(s) #s            // Stringify a Makro
#define XSTR(s) STR(s)

// === Semantic Version Information ===

#define CSV_VER_MAJOR 1
#define CSV_VER_MINOR 0
#define CSV_VER_PATCH 0
#define CSV_VER_APPENDIX "-dev"

const char csv_version[] = XSTR(CSV_VER_MAJOR) "." XSTR(CSV_VER_MINOR) "." XSTR(CSV_VER_PATCH) CSV_VER_APPENDIX "\0" __DATE__ "\0" __TIME__;

// === CSV-MEMORY Interface ===

static void *
csv_mem_reallocate(void *ptr, size_t num, size_t size, void *cb_arg)
{
	UNUSED(cb_arg);

	return reallocarray(ptr, num, size);
}

static void
csv_mem_free(void *ptr, size_t num, size_t size, void *cb_arg)
{
	UNUSED(num);
	UNUSED(size);
	UNUSED(cb_arg);

	free(ptr);
}

// === CSV-OPTIONS Interface ===

const csv_options_t csv_default_options = {
	.field_delimiter = CSV_DEFAULT_DELIMITER,
	.field_separator = CSV_DEFAULT_SEPARATOR,
	.cb_error        = NULL,
	.cb_reallocate   = csv_mem_reallocate,
	.cb_free         = csv_mem_free
};

// === CSV-ERROR Interface ===

static void
csv_fatal_error(csv_err_t csv_err, const csv_options_t *const csv_options)
{
	if ( csv_options->cb_error != NULL ) {
		(*csv_options->cb_error)(csv_err, csv_options->cb_error_arg);
	}
	(void) fprintf(stderr, "fatal error: %s\n", csv_err_str(csv_err));
	abort();
}

// === CSV-STRING Interface ===

static inline void
csv_string_init(csv_string_t *csv_string)
{
	assert(csv_string != NULL);

	csv_string->str = NULL;
	csv_string->cap = 0;
	csv_string->pos = 0;
}

static inline void
csv_string_reset(csv_string_t *csv_string)
{
	assert(csv_string != NULL);

	csv_string->pos = 0;
}

static inline int
csv_string_isempty(csv_string_t *csv_string)
{
	assert(csv_string != NULL);

	return (csv_string->pos == 0) ? 1 : 0;
}

static inline size_t
growth_strategy(size_t current_cap)
{
	static const size_t INITIAL_CAP = 16;

	return (current_cap == 0) ? INITIAL_CAP : (current_cap * 3) / 2;
}

static inline void
csv_string_grow_if_needed(csv_string_t *csv_string, const csv_options_t *const csv_options)
{
	assert(csv_string != NULL);
	assert(csv_options != NULL);

	if ( csv_string->pos == csv_string->cap ) {
		size_t cap = growth_strategy(csv_string->cap);
		char  *str = csv_options->cb_reallocate(csv_string->str, cap, 1, csv_options->cb_memory_arg);
		if ( str == NULL ) {
			csv_fatal_error(CSV_ERR_OUT_OF_MEMORY, csv_options);
			return;
		}
		csv_string->str = str;
		csv_string->cap = cap;
	}
}

static inline void
csv_string_append(csv_string_t *csv_string, int ch, const csv_options_t *const csv_options)
{
	assert(csv_string != NULL);
	assert(csv_options != NULL);

	csv_string_grow_if_needed(csv_string, csv_options);

	csv_string->str[csv_string->pos++] = (char) ch; // append char
}

static inline void
csv_string_free(csv_string_t *csv_string, const csv_options_t *const csv_options)
{
	assert(csv_string != NULL);
	assert(csv_options != NULL);

	csv_options->cb_free(csv_string->str, csv_string->cap, 1, csv_options->cb_memory_arg);

	// call *_init() for sane default values; prevent possible double-free
	csv_string_init(csv_string);
}

// === CSV-FIELD Interface ===

static inline void
csv_field_init(csv_field_t *csv_field)
{
	assert(csv_field != NULL);

	csv_field->fields = NULL;
	csv_field->cap    = 0;
	csv_field->pos    = 0;
}

static inline void
csv_field_reset(csv_field_t *csv_field)
{
	assert(csv_field != NULL);

	csv_field->pos = 0;
}

static inline void
csv_field_grow_if_needed(csv_field_t *csv_field, const csv_options_t *const csv_options)
{
	assert(csv_field != NULL);
	assert(csv_options != NULL);

	if ( csv_field->pos == csv_field->cap ) {
		size_t  cap    = growth_strategy(csv_field->cap);
		size_t *fields = csv_options->cb_reallocate(csv_field->fields, cap, sizeof(csv_field->fields[0]), csv_options->cb_memory_arg);
		if ( fields == NULL ) {
			csv_fatal_error(CSV_ERR_OUT_OF_MEMORY, csv_options);
			return;
		}
		csv_field->fields = fields;
		csv_field->cap    = cap;
	}
}

static inline void
csv_field_append(csv_field_t *csv_field, size_t idx, const csv_options_t *const csv_options)
{
	assert(csv_field != NULL);
	assert(csv_options != NULL);

	csv_field_grow_if_needed(csv_field, csv_options);

	csv_field->fields[csv_field->pos++] = idx; // append index
}

static inline void
csv_field_free(csv_field_t *csv_field, const csv_options_t *const csv_options)
{
	assert(csv_field != NULL);
	assert(csv_options != NULL);

	csv_options->cb_free(csv_field->fields, csv_field->cap, sizeof(csv_field->fields[0]), csv_options->cb_memory_arg);

	csv_field_init(csv_field);
}

// === CSV Interface ===

void
csv_init(csv_t *csv)
{
	assert(csv != NULL);

	csv_init_opt(csv, NULL);
}

void
csv_init_opt(csv_t *csv, const csv_options_t *const csv_options)
{
	assert(csv != NULL);

	if ( csv_options != NULL ) {
		csv->csv_options = csv_options;
	}
	else {
		csv->csv_options = &csv_default_options;
	}

	csv_string_init(&csv->csv_string);
	csv_field_init(&csv->csv_field);
}

void
csv_cleanup(csv_t *csv)
{
	assert(csv != NULL);
	assert(csv->csv_options != NULL);

	csv_string_free(&csv->csv_string, csv->csv_options);
	csv_field_free(&csv->csv_field, csv->csv_options);
}

size_t
csv_nfields(const csv_t *const csv)
{
	assert(csv != NULL);

	return csv->csv_field.pos;
}

const char *
csv_field(const csv_t *const csv, size_t idx)
{
	assert(csv != NULL);
	assert(idx >= 0 && idx < csv->csv_field.pos);

	if ( idx >= csv->csv_field.pos ) {
		csv_fatal_error(CSV_ERR_OUT_OF_RANGE, csv->csv_options);
		return NULL;
	}
	return &csv->csv_string.str[csv->csv_field.fields[idx]];
}

size_t
csv_read(csv_t *csv, FILE *in)
{
	assert(csv != NULL);
	assert(in != NULL);

	// initialize if needed...
	if ( csv->csv_options == NULL ) {
		csv_init(csv);
	}

	if ( ferror(in) ) {
		csv_fatal_error(CSV_ERR_IO_READ, csv->csv_options);
		return 0;
	}

	// do not try to read if EOF has already been seen
	if ( feof(in) ) {
		csv_cleanup(csv);
		return 0;
	}

	enum {
		STATE_START_FIELD,
		STATE_QUOTED_FIELD,
		STATE_SIMPLE_FIELD,
		STATE_END_FIELD,
		STATE_END_LINE,
		STATE_END_FILE,
	};

	register const int DELIM = csv->csv_options->field_delimiter;
	register const int SEP   = csv->csv_options->field_separator;

	csv_string_reset(&csv->csv_string);
	csv_field_reset(&csv->csv_field);

	for ( int state = STATE_START_FIELD;; ) {
		int ch;

		switch ( state ) {
		case STATE_START_FIELD:
			csv_field_append(&csv->csv_field, csv->csv_string.pos, csv->csv_options);

			ch = getc(in);
			if ( ch == EOF ) {
				state = STATE_END_FILE;
			}
			else if ( ch == '\r' ) { // test for CR..
				ch = getc(in);
				if ( ch != '\n' ) { // ..LF
					(void) ungetc(ch, in);
				}
				state = STATE_END_LINE;
			}
			else if ( ch == '\n' ) {
				state = STATE_END_LINE;
			}
			else if ( ch == SEP ) {
				state = STATE_END_FIELD;
			}
			else if ( ch == DELIM ) {
				state = STATE_QUOTED_FIELD;
			}
			else {
				if ( ch != '\0' ) {
					csv_string_append(&csv->csv_string, ch, csv->csv_options);
				}
				state = STATE_SIMPLE_FIELD;
			}
			break;

		case STATE_QUOTED_FIELD:
			do {
				ch = getc(in);
				if ( ch == EOF ) {
					state = STATE_END_FILE;
				}
				else if ( ch == DELIM ) {
					ch = getc(in);
					if ( ch == EOF ) {
						state = STATE_END_FILE;
					}
					else if ( ch == DELIM ) {
						csv_string_append(&csv->csv_string, DELIM, csv->csv_options);
					}
					else if ( ch == SEP ) {
						state = STATE_END_FIELD;
					}
					else if ( ch == '\r' ) {
						ch = getc(in);
						if ( ch != '\n' ) {
							(void) ungetc(ch, in);
						}
						state = STATE_END_LINE;
					}
					else if ( ch == '\n' ) {
						state = STATE_END_LINE;
					}
					else {
						csv_string_append(&csv->csv_string, DELIM, csv->csv_options);
						(void) ungetc(ch, in); // we have read too far... Put the character back!
					}
				}
				else {
					if ( ch != '\0' ) {
						csv_string_append(&csv->csv_string, ch, csv->csv_options);
					}
				}
			} while ( state == STATE_QUOTED_FIELD );
			break;

		case STATE_SIMPLE_FIELD:
			do {
				ch = getc(in);
				if ( ch == EOF ) {
					state = STATE_END_FILE;
				}
				else if ( ch == SEP ) {
					state = STATE_END_FIELD;
				}
				else if ( ch == '\r' ) {
					ch = getc(in);
					if ( ch != '\n' ) {
						(void) ungetc(ch, in);
					}
					state = STATE_END_LINE;
				}
				else if ( ch == '\n' ) {
					state = STATE_END_LINE;
				}
				else {
					if ( ch != '\0' ) {
						csv_string_append(&csv->csv_string, ch, csv->csv_options);
					}
				}
			} while ( state == STATE_SIMPLE_FIELD );
			break;

		case STATE_END_FIELD:
			csv_string_append(&csv->csv_string, '\0', csv->csv_options);
			state = STATE_START_FIELD;
			break;

		case STATE_END_LINE:
			csv_string_append(&csv->csv_string, '\0', csv->csv_options);
			return csv->csv_field.pos;

		case STATE_END_FILE:
			if ( ferror(in) ) {
				csv_fatal_error(CSV_ERR_IO_READ, csv->csv_options);
				return 0;
			}

			if ( csv_string_isempty(&csv->csv_string) ) {
				csv_cleanup(csv);
				return 0; // EOF reached
			}

			/*
			 * The last data record was not terminated with a NEWLINE-Symbol.
			 * So we can't signal End-Of-File for now.  Terminate the current
			 * field and return the number of fields processed so far.
			 */
			csv_string_append(&csv->csv_string, '\0', csv->csv_options);

			return csv->csv_field.pos;

		default:
			assert(!"this should never be happen...");
			break;
		}
	}
	// NOT REACHED
}

const char *
csv_err_str(csv_err_t csv_err)
{
	switch ( csv_err ) {
	case CSV_ERR_OK:
		return "no error";
	case CSV_ERR_OUT_OF_MEMORY:
		return "out of memory";
	case CSV_ERR_OUT_OF_RANGE:
		return "index out of range";
	case CSV_ERR_IO_READ:
		return "read error";
	case CSV_ERR_IO_WRITE:
		return "write error";
	default:
		return "unknown error";
	}
	// NOT REACHED
}