#ifndef lint
static char *RCSid() { return RCSid("$Id: datafile.c,v 1.16.2.3 2000/06/14 00:38:37 joze Exp $"); }
#endif
/* GNUPLOT - datafile.c */
/*[
* Copyright 1986 - 1993, 1998 Thomas Williams, Colin Kelley
*
* Permission to use, copy, and distribute this software and its
* documentation for any purpose with or without fee is hereby granted,
* provided that the above copyright notice appear in all copies and
* that both that copyright notice and this permission notice appear
* in supporting documentation.
*
* Permission to modify the software is granted, but not the right to
* distribute the complete modified source code. Modifications are to
* be distributed as patches to the released version. Permission to
* distribute binaries produced by compiling modified sources is granted,
* provided you
* 1. distribute the corresponding source modifications from the
* released version in the form of a patch file along with the binaries,
* 2. add special version identification to distinguish your version
* in addition to the base release version number,
* 3. provide your name and address as the primary contact for the
* support of your modified version, and
* 4. retain our contact information in regard to use of the base
* software.
* Permission to distribute the released version of the source code along
* with corresponding source modifications in the form of a patch file is
* granted with same provisions 2 through 4 for binary distributions.
*
* This software is provided "as is" without express or implied warranty
* to the extent permitted by applicable law.
]*/
/* AUTHOR : David Denholm */
/*
* this file provides the functions to handle data-file reading..
* takes care of all the pipe / stdin / index / using worries
*/
/*{{{ notes */
/* couldn't decide how to implement 'thru' only for 2d and 'index'
* for only 3d, so I did them for both - I can see a use for
* index in 2d, especially for fit.
*
* I keep thru for backwards compatibility, and extend it to allow
* more natural plot 'data' thru f(y) - I (personally) prefer
* my syntax, but then I'm biased...
*
* - because I needed it, I have added a range of indexes...
* (s)plot 'data' [index i[:j]]
*
* also every a:b:c:d:e:f - plot every a'th point from c to e,
* in every b lines from d to f
* ie for (line=d; line=e; point+=a)
*
*
* I dont like mixing this with the time series hack... I am
* very into modular code, so I would prefer to not have to
* have _anything_ to do with time series... for example,
* we just look at columns in file, and that is independent
* of 2d/3d. I really dont want to have to pass a flag to
* this is plot or splot.
*
* use a global array df_timecol[] - cleared by df_open, then
* columns needing time set by client.
*
* Now that df_2dbinary() and df_3dbinary() are here, I am seriously
* tempted to move get_data() and get_3ddata() in here too
*
* public variables declared in this file.
* int df_no_use_specs - number of columns specified with 'using'
* int df_line_number - for error reporting
* int df_datum - increases with each data point
* TBOOLEAN df_binary - it's a binary file
* [ might change this to return value from df_open() ]
* int df_eof - end of file
* int df_timecol[] - client controls which cols read as time
*
* functions
* int df_open(int max_using)
* parses thru / index / using on command line
* max_using is max no of 'using' columns allowed
* returns number of 'using' cols specified, or -1 on error (?)
*
* int df_readline(double vector[], int max)
* reads a line, does all the 'index' and 'using' manipulation
* deposits values into vector[]
* returns
* number of columns parsed [0=not blank line, but no valid data],
* DF_EOF for EOF
* DF_UNDEFINED - undefined result during eval of extended using spec
* DF_FIRST_BLANK for first consecutive blank line
* DF_SECOND_BLANK for second consecutive blank line
* will return FIRST before SECOND
*
* if a using spec was given, lines not fulfilling spec are ignored.
* we will always return exactly the number of items specified
*
* if no spec given, we return number of consecutive columns we parsed.
*
* if we are processing indexes, separated by 'n' blank lines,
* we will return n-1 blank lines before noticing the index change
*
* void df_close()
* closes a currently open file.
*
* void f_dollars(x)
* void f_column() actions for expressions using $i, column(j), etc
* void f_valid()
*
*
* line parsing slightly differently from previous versions of gnuplot...
* given a line containing fewer columns than asked for, gnuplot used to make
* up values... I say that if I have explicitly said 'using 1:2:3', then if
* column 3 doesn't exist, I dont want this point...
*
* a column number of 0 means generate a value... as before, this value
* is useful in 2d as an x value, and is reset at blank lines.
* a column number of -1 means the (data) line number (not the file line
* number). splot 'file' using 1 is equivalent to
* splot 'file' using 0:-1:1
* column number -2 is the index. It was put in to kludge multi-branch
* fitting.
*
* 20/5/95 : accept 1.23d4 in place of e (but not in scanf string)
* : autoextend data line buffer and MAX_COLS
*
* 11/8/96 : add 'columns' -1 for suggested y value, and -2 for
* current index.
* using 1:-1:-2 and column(-1) are supported.
* $-1 and $-2 are not yet supported, because of the
* way the parser works
*
*/
/*}}} */
#include "datafile.h"
#include "alloc.h"
#include "command.h"
#include "binary.h"
#include "gp_time.h"
#include "graphics.h"
#include "internal.h"
#include "misc.h"
#include "parse.h"
#include "setshow.h"
#include "util.h"
/* if you change this, change the scanf in readline */
#define NCOL 7 /* max using specs */
/*{{{ static fns */
#if 0 /* not used */
static int get_time_cols __PROTO((char *fmt));
static void mod_def_usespec __PROTO((int specno, int jump));
#endif
static int check_missing __PROTO((char *s));
static char *df_gets __PROTO((void));
static int df_tokenise __PROTO((char *s));
static float **df_read_matrix __PROTO((int *rows, int *columns));
static void plot_option_every __PROTO((void));
static void plot_option_index __PROTO((void));
static void plot_option_thru __PROTO((void));
static void plot_option_using __PROTO((int));
static TBOOLEAN valid_format __PROTO((const char *));
/*}}} */
/*{{{ variables */
struct use_spec_s {
int column;
struct at_type *at;
};
/* public variables client might access */
int df_no_use_specs; /* how many using columns were specified */
int df_line_number;
int df_datum; /* suggested x value if none given */
TBOOLEAN df_matrix = FALSE; /* is this a matrix splot */
int df_eof = 0;
int df_timecol[NCOL];
TBOOLEAN df_binary = FALSE; /* this is a binary file */
/* jev -- the 'thru' function --- NULL means no dummy vars active */
/* HBB 990829: moved this here, from command.c */
struct udft_entry ydata_func;
/* private variables */
/* in order to allow arbitrary data line length, we need to use the heap
* might consider free-ing it in df_close, especially for small systems
*/
static char *line = NULL;
static size_t max_line_len = 0;
#define DATA_LINE_BUFSIZ 160
static FILE *data_fp = NULL;
static TBOOLEAN pipe_open = FALSE;
static TBOOLEAN mixed_data_fp = FALSE;
#ifndef MAXINT /* should there be one already defined ? */
# ifdef INT_MAX /* in limits.h ? */
# define MAXINT INT_MAX
# else
# define MAXINT ((~0)>>1)
# endif
#endif
/* stuff for implementing index */
static int blank_count = 0; /* how many blank lines recently */
static int df_lower_index = 0; /* first mesh required */
static int df_upper_index = MAXINT;
static int df_index_step = 1; /* 'every' for indices */
static int df_current_index; /* current mesh */
/* stuff for every point:line */
static int everypoint = 1;
static int firstpoint = 0;
static int lastpoint = MAXINT;
static int everyline = 1;
static int firstline = 0;
static int lastline = MAXINT;
static int point_count = -1; /* point counter - preincrement and test 0 */
static int line_count = 0; /* line counter */
/* parsing stuff */
static struct use_spec_s use_spec[NCOL];
static char df_format[MAX_LINE_LEN + 1];
/* rather than three arrays which all grow dynamically, make one
* dynamic array of this structure
*/
typedef struct df_column_struct {
double datum;
enum {
DF_MISSING, DF_BAD, DF_GOOD
} good;
char *position;
} df_column_struct;
static df_column_struct *df_column = NULL; /* we'll allocate space as needed */
static int df_max_cols = 0; /* space allocated */
static int df_no_cols; /* cols read */
static int fast_columns; /* corey@cac optimization */
/* columns needing timefmt are passed in df_timecol[] after df_open */
/*}}} */
/*{{{ static char *df_gets() */
static char *
df_gets()
{
int len = 0;
/* HBB 20000526: prompt user for inline data, if in interactive mode */
if (mixed_data_fp && interactive)
fputs("input data ('e' ends) > ", stderr);
if (!fgets(line, max_line_len, data_fp))
return NULL;
if (mixed_data_fp)
++inline_num;
for (;;) {
len += strlen(line + len);
if (len > 0 && line[len - 1] == '\n') {
/* we have read an entire text-file line.
* Strip the trailing linefeed and return
*/
line[len - 1] = 0;
return line;
}
/* buffer we provided may not be full - dont grab extra
* memory un-necessarily. This may trap a problem with last
* line in file not being properly terminated - each time
* through a replot loop, it was doubling buffer size
*/
if ((max_line_len - len) < 32)
line = gp_realloc(line, max_line_len *= 2, "datafile line buffer");
if (!fgets(line + len, max_line_len - len, data_fp))
return line; /* unexpected end of file, but we have something to do */
}
/* NOTREACHED */
return NULL;
}
/*}}} */
/*{{{ static int df_tokenise(s) */
static int
df_tokenise(s)
char *s;
{
/* implement our own sscanf that takes 'missing' into account,
* and can understand fortran quad format
*/
df_no_cols = 0;
while (*s) {
/* check store - double max cols or add 20, whichever is greater */
if (df_max_cols 0)
&& (use_spec[0].column == dfncp1
|| (df_no_use_specs > 1
&& (use_spec[1].column == dfncp1
|| (df_no_use_specs > 2
&& (use_spec[2].column == dfncp1
|| (df_no_use_specs > 3
&& (use_spec[3].column == dfncp1
|| (df_no_use_specs > 4 && (use_spec[4].column == dfncp1 || df_no_use_specs > 5)
)
)
)
)
)
)
)
)
)
) {
#ifndef NO_FORTRAN_NUMS
count = sscanf(s, "%lf%n", &df_column[df_no_cols].datum, &used);
#else
while (isspace(*s))
++s;
count = *s ? 1 : 0;
df_column[df_no_cols].datum = atof(s);
#endif /* NO_FORTRAN_NUMS */
} else {
/* skip any space at start of column */
/* HBB tells me that the cast must be to
* unsigned char instead of int. */
while (isspace((unsigned char) *s))
++s;
count = *s ? 1 : 0;
/* skip chars to end of column */
used = 0;
while (!isspace((unsigned char) *s) && (*s != NUL))
++s;
}
/* it might be a fortran double or quad precision.
* 'used' is only safe if count is 1
*/
#ifndef NO_FORTRAN_NUMS
if (count == 1 &&
(s[used] == 'd' || s[used] == 'D' ||
s[used] == 'q' || s[used] == 'Q')) {
/* might be fortran double */
s[used] = 'e';
/* and try again */
count = sscanf(s, "%lf", &df_column[df_no_cols].datum);
}
#endif /* NO_FORTRAN_NUMS */
#endif /* OSK */
df_column[df_no_cols].good = count == 1 ? DF_GOOD : DF_BAD;
}
++df_no_cols;
/*{{{ skip chars to end of column */
while ((!isspace((int) *s)) && (*s != '\0'))
++s;
/*}}} */
/*{{{ skip spaces to start of next column */
while (isspace((int) *s))
++s;
/*}}} */
}
return df_no_cols;
}
/*}}} */
/*{{{ static float **df_read_matrix() */
/* reads a matrix from a text file
* stores in same storage format as fread_matrix
*/
static float **
df_read_matrix(rows, cols)
int *rows, *cols;
{
int max_rows = 0;
int c;
float **rmatrix = NULL;
char *s;
*rows = 0;
*cols = 0;
for (;;) {
if (!(s = df_gets())) {
df_eof = 1;
return rmatrix; /* NULL if we have not read anything yet */
}
while (isspace((int) *s))
++s;
if (!*s || is_comment(*s)) {
if (rmatrix)
return rmatrix;
else
continue;
}
if (mixed_data_fp && is_EOF(*s)) {
df_eof = 1;
return rmatrix;
}
c = df_tokenise(s);
if (!c)
return rmatrix;
if (*cols && c != *cols) {
/* its not regular */
int_error(NO_CARET, "Matrix does not represent a grid");
}
*cols = c;
if (*rows >= max_rows) {
rmatrix = gp_realloc(rmatrix, (max_rows += 10) * sizeof(float *), "df_matrix");
}
/* allocate a row and store data */
{
int i;
float *row = rmatrix[*rows] = (float *) gp_alloc(c * sizeof(float),
"df_matrix row");
for (i = 0; i < c; ++i) {
if (df_column[i].good != DF_GOOD)
int_error(NO_CARET, "Bad number in matrix");
row[i] = (float) df_column[i].datum;
}
++*rows;
}
}
}
/*}}} */
/*{{{ int df_open(max_using) */
int
df_open(max_using)
int max_using;
/* open file, parsing using/thru/index stuff
* return number of using specs [well, we have to return something !]
*/
{
/* now allocated dynamically */
static char *filename = NULL;
int i;
int name_token;
fast_columns = 1; /* corey@cac */
/*{{{ close file if necessary */
if (data_fp)
df_close();
/*}}} */
/*{{{ initialise static variables */
df_format[0] = NUL; /* no format string */
df_no_use_specs = 0;
for (i = 0; i < NCOL; ++i) {
use_spec[i].column = i + 1; /* default column */
use_spec[i].at = NULL; /* no expression */
}
if (max_using > NCOL)
max_using = NCOL;
df_datum = -1; /* it will be preincremented before use */
df_line_number = 0; /* ditto */
df_lower_index = 0;
df_index_step = 1;
df_upper_index = MAXINT;
df_current_index = 0;
blank_count = 2;
/* by initialising blank_count, leading blanks will be ignored */
everypoint = everyline = 1; /* unless there is an every spec */
firstpoint = firstline = 0;
lastpoint = lastline = MAXINT;
df_eof = 0;
memset(df_timecol, 0, sizeof(df_timecol));
df_binary = 1;
/*}}} */
assert(max_using = -2");
use_spec[df_no_use_specs++].column = col;
}
} while (equals(c_token, ":") && ++c_token);
}
if (!END_OF_COMMAND && isstring(c_token)) {
if (df_binary)
int_error(NO_CARET, "Format string meaningless with binary data");
quote_str(df_format, c_token, MAX_LINE_LEN);
if (!valid_format(df_format))
int_error(c_token, "Please use a double conversion %lf");
c_token++; /* skip format */
}
}
/*{{{ int df_readline(v, max) */
/* do the hard work... read lines from file,
* - use blanks to get index number
* - ignore lines outside range of indices required
* - fill v[] based on using spec if given
*/
int
df_readline(v, max)
double v[];
int max;
{
char *s;
assert(data_fp != NULL);
assert(!df_binary);
assert(max_line_len); /* alloc-ed in df_open() */
assert(max