<?php
// +----------------------------------------------------------------------+
// | PHP Version 4                                                        |
// +----------------------------------------------------------------------+
// | Copyright (c) 2002-2003 Tomas Von Veschler Cox                       |
// +----------------------------------------------------------------------+
// | This source file is subject to version 2.0 of the PHP license,       |
// | that is bundled with this package in the file LICENSE, and is        |
// | available at through the world-wide-web at                           |
// | http://www.php.net/license/2_02.txt.                                 |
// | If you did not receive a copy of the PHP license and are unable to   |
// | obtain it through the world-wide-web, please send a note to          |
// | license@php.net so we can mail you a copy immediately.               |
// +----------------------------------------------------------------------+
// | Authors: Tomas V.V.Cox <cox@idecnet.com>                             |
// | Authors: Paul M. Jones <pjones@ciaweb.net>                           |
// +----------------------------------------------------------------------+
//
// $Id: CSV.php,v 1.13 2003/01/04 11:54:55 mj Exp $

require_once 'PEAR.php';
require_once 'File.php';

/**
* File class for handling CSV files (Comma Separated Values), a common format
* for exchanging data.
*
* TODO:
*  - Usage example and Doc
*  - Use getPointer() in discoverFormat
*  - Add a line counter for being able to output better error reports
*  - Store the last error in GLOBALS and add File_CSV::getLastError()
*
* Wish:
*  - Support Mac EOL format
*  - Other methods like readAll(), writeAll(), numFields(), numRows()
*  - Try to detect if a CSV has header or not in discoverFormat()
*
* Known Bugs:
* (they has been analyzed but for the moment the impact in the speed for
*  properly handle this uncommon cases is too high and won't be supported)
*  - A field which is composed only by a single quoted separator (ie -> ;";";)
*    is not handled properly
*  - When there is exactly one field minus than the expected number and there
*    is a field with a separator inside, the parser will throw the "wrong count" error
*
* @author Tomas V.V.Cox <cox@idecnet.com>
* @package File
*/
class File_CSV
{
    /**
    * This raiseError method works in a different way. It will always return
    * false (an error occurred) but it will call PEAR::raiseError() before
    * it. If no default PEAR global handler is set, will trigger an error.
    *
    * @param string $error The error message
    * @return bool always false
    */
    function raiseError($error)
    {
        // If a default PEAR Error handler is not set trigger the error
        // XXX Add a PEAR::isSetHandler() method?
        if ($GLOBALS['_PEAR_default_error_mode'] == PEAR_ERROR_RETURN) {
            PEAR::raiseError($error, null, PEAR_ERROR_TRIGGER, E_USER_WARNING);
        } else {
            PEAR::raiseError($error);
        }
        return false;
    }

    /**
    * Checks the configuration given by the user
    *
    * @param array  &$conf  The configuration assoc array
    * @param string &$error The error will be written here if any
    */
    function _conf(&$conf, &$error)
    {
        // check conf
        if (!is_array($conf)) {
            return $error = "Invalid configuration";
        }
        if (isset($conf['sep'])) {
            if (strlen($conf['sep']) != 1) {
                return $error = 'Separator can only be one char';
            }
        } else {
            return $error = 'Missing separator (the "sep" key)';
        }
        if (!isset($conf['fields']) || !is_numeric($conf['fields'])) {
            return $error = 'The number of fields must be numeric (the "fields" key)';
        }
        if (isset($conf['quote'])) {
            if (strlen($conf['quote']) != 1) {
                return $error = 'The quote char must be one char (the "quote" key)';
            }
        } else {
            $conf['quote'] = null;
        }
        if (!isset($conf['crlf'])) {
            $conf['crlf'] = "\n";
        }
    }

    /**
    * Return or create the file descriptor associated with a file
    *
    * @param string $file The name of the file
    * @param array  &$conf The configuration
    * @param string $mode The open node (ex: FILE_MODE_READ or FILE_MODE_WRITE)
    *
    * @return mixed A file resource or false
    */
    function getPointer($file, &$conf, $mode = FILE_MODE_READ)
    {
        static $resources  = array();
        static $config;
        if (isset($resources[$file])) {
            $conf = $config;
            return $resources[$file];
        }
        File_CSV::_conf($conf, $error);
        if ($error) {
            return File_CSV::raiseError($error);
        }
        $config = $conf;
        PEAR::pushErrorHandling(PEAR_ERROR_RETURN);
        $fp = &File::_getFilePointer($file, $mode);
        PEAR::popErrorHandling();
        if (PEAR::isError($fp)) {
            return File_CSV::raiseError($fp);
        }
        $resources[$file] = $fp;

        if ($mode == FILE_MODE_READ && !empty($conf['header'])) {
            if (!File_CSV::read($file, $conf)) {
                return false;
            }
        }
        return $fp;
    }

    /**
    * Unquote data
    *
    * @param string $field The data to unquote
    * @param string $quote The quote char
    * @return string the unquoted data
    */
    function unquote($field, $quote)
    {
        // Incase null fields (form: ;;)
        if (!strlen($field)) {
            return $field;
        }
        if ($quote && $field{0} == $quote && $field{strlen($field)-1} == $quote) {
            return substr($field, 1, -1);
        }
        return $field;
    }

    /**
    * Reads a row of data as an array from a CSV file. It's able to
    * read memo fields with multiline data.
    *
    * @param string $file   The filename where to write the data
    * @param array  &$conf   The configuration of the dest CSV
    *
    * @return mixed Array with the data read or false on error/no more data
    */
    function readQuoted($file, &$conf)
    {
        if (!$fp = File_CSV::getPointer($file, $conf, FILE_MODE_READ)) {
            return false;
        }
        $buff = $c = null;
        $ret  = array();
        $i = 1;
        $in_quote = false;
        $quote = $conf['quote'];
        $f = $conf['fields'];
        while (($ch = fgetc($fp)) !== false) {
            $prev = $c;
            $c = $ch;
            // Common case
            if ($c != $quote && $c != $conf['sep'] && $c != "\n") {
                $buff .= $c;
                continue;
            }
            if ($c == $quote && $quote &&
                ($prev == $conf['sep'] || $prev == "\n" || $prev === null))
            {
                $in_quote = true;
            } elseif ($in_quote) {
                // When ends quote
                if ($c == $conf['sep'] && $prev == $conf['quote']) {
                    $in_quote = false;
                } elseif ($c == "\n") {
                    $sub = ($prev == "\r") ? 2 : 1;
                    if ((strlen($buff) >= $sub) &&
                        ($buff{strlen($buff) - $sub} == $quote))
                    {
                        $in_quote = false;
                    }
                }
            }
            if (!$in_quote && ($c == $conf['sep'] || $c == "\n")) {
                // More fields than expected
                if (($c == $conf['sep']) && ((count($ret) + 1) == $f)) {
                    while ($c != "\n") {
                        $c = fgetc($fp);
                    }
                    File_CSV::raiseError("Read more fields than the ".
                                         "expected ".$conf['fields']);
                    return true;
                }
                // Less fields than expected
                if (($c == "\n") && ($i != $f)) {
                    File_CSV::raiseError("Read wrong fields number count: '". $i .
                                         "' expected ".$conf['fields']);
                    return true;
                }
                if ($prev == "\r") {
                    $buff = substr($buff, 0, -1);
                }
                $ret[] = File_CSV::unquote($buff, $quote);
                if (count($ret) == $f) {
                    return $ret;
                }
                $buff = '';
                $i++;
                continue;
            }
            $buff .= $c;
        }
        return !feof($fp) ? $ret : false;
    }

    /**
    * Reads a "row" from a CSV file and return it as an array
    *
    * @param string $file The CSV file
    * @param array  &$conf The configuration of the dest CSV
    *
    * @return mixed Array or false
    */
    function read($file, &$conf)
    {
        if (!$fp = File_CSV::getPointer($file, $conf, FILE_MODE_READ)) {
            return false;
        }
        // The size is limited to 4K
        if (!$line   = fgets($fp, 4096)) {
            return false;
        }
        $fields = explode($conf['sep'], $line);
        if ($conf['quote']) {
            $last =& $fields[count($fields) - 1];
            // Fallback to read the line with readQuoted when guess
            // that the simple explode won't work right
            if (($last{strlen($last) - 1} == "\n"
                && $last{0} == $conf['quote']
                && $last{strlen(rtrim($last)) - 1} != $conf['quote'])
                ||
                (count($fields) != $conf['fields'])
                // XXX perhaps there is a separator inside a quoted field
                //preg_match("|{$conf['quote']}.*{$conf['sep']}.*{$conf['quote']}|U", $line)
                )
            {
                $len = strlen($line);
                fseek($fp, -1 * strlen($line), SEEK_CUR);
                return File_CSV::readQuoted($file, $conf);
            } else {
                $last = rtrim($last);
                foreach ($fields as $k => $v) {
                    $fields[$k] = File_CSV::unquote($v, $conf['quote']);
                }
            }
        }
        if (count($fields) != $conf['fields']) {
            File_CSV::raiseError("Read wrong fields number count: '". count($fields) .
                                  "' expected ".$conf['fields']);
            return true;
        }
        return $fields;
    }
    

    /**
    * Internal use only, will be removed in the future
    *
    * @param string $str The string to debug
    * @access private
    */
    function _dbgBuff($str)
    {
        if (strpos($str, "\r") !== false) {
            $str = str_replace("\r", "_r_", $str);
        }
        if (strpos($str, "\n") !== false) {
            $str = str_replace("\n", "_n_", $str);
        }
        if (strpos($str, "\t") !== false) {
            $str = str_replace("\t", "_t_", $str);
        }
        echo "buff: ($str)\n";
    }

    /**
    * Writes a struc (array) in a file as CSV
    *
    * @param string $file   The filename where to write the data
    * @param array  $fields Ordered array with the data
    * @param array  &$conf   The configuration of the dest CSV
    *
    * @return bool True on success false otherwise
    */
    function write($file, $fields, &$conf)
    {
        if (!$fp = File_CSV::getPointer($file, $conf, FILE_MODE_WRITE)) {
            return false;
        }
        if (count($fields) != $conf['fields']) {
            File_CSV::raiseError("Wrong fields number count: '". count($fields) .
                                  "' expected ".$conf['fields']);
            return true;
        }
        $write = '';
        for ($i = 0; $i < count($fields); $i++) {
            if (!is_numeric($fields[$i]) && $conf['quote']) {
                $write .= $conf['quote'] . $fields[$i] . $conf['quote'];
            } else {
                $write .= $fields[$i];
            }
            if ($i < (count($fields) - 1)) {
                $write .= $conf['sep'];
            } else {
                $write .= $conf['crlf'];
            }
        }
        if (!fwrite($fp, $write)) {
            return File_CSV::raiseError('Can not write to file');
        }
        return true;
    }

    /**
    * Discover the format of a CSV file (the number of fields, the separator
    * and if it quote string fields)
    *
    * @param string the CSV file name
    * @return mixed Assoc array or false
    */
    function discoverFormat($file)
    {
        if (!$fp = @fopen($file, 'r')) {
            return File_CSV::raiseError("Could not open file: $file");
        }
        $seps = array("\t", ';', ':', ',');
        $matches = array();
        // Take the first 10 lines and store the number of ocurrences
        // for each separator in each line
        for ($i = 0; ($i < 10) && ($line = fgets($fp, 4096)); $i++) {
            foreach ($seps as $sep) {
                $matches[$sep][$i] = substr_count($line, $sep);
            }
        }
        $final = array();
        // Group the results by amount of equal ocurrences
        foreach ($matches as $sep => $res) {
            $times = array();
            $times[0] = 0;
            foreach ($res as $k => $num) {
                if ($num > 0) {
                    $times[$num] = (isset($times[$num])) ? $times[$num] + 1 : 1;
                }
            }
            arsort($times);
            $fields[$sep] = key($times);
            $amount[$sep] = $times[key($times)];
        }
        arsort($amount);
        $sep    = key($amount);
        $fields = $fields[$sep];
        if (empty($fields)) {
            return File_CSV::raiseError('Could not discover the separator');
        }
        $conf['fields'] = $fields + 1;
        $conf['sep']    = $sep;
        // Test if there are fields with quotes arround in the first 5 lines
        $quotes = '"\'';
        $quote  = null;
        rewind($fp);
        for ($i = 0; ($i < 5) && ($line = fgets($fp, 4096)); $i++) {
            if (preg_match("|$sep([$quotes]).*([$quotes])$sep|U", $line, $match)) {
                if ($match[1] == $match[2]) {
                    $quote = $match[1];
                    break;
                }
            }
            if (preg_match("|^([$quotes]).*([$quotes])$sep|", $line, $match)
                || preg_match("|([$quotes]).*([$quotes])$sep\s$|Us", $line, $match))
            {
                if ($match[1] == $match[2]) {
                    $quote = $match[1];
                    break;
                }
            }
        }
        $conf['quote'] = $quote;
        fclose($fp);
        // XXX What about trying to discover the "header"?
        return $conf;
    }
    
    
    /**
    *
    * Parses a block of CSV-formatted text and returns an array of all
    * rows and fields.
    * 
    * This method is very heavily based on readQuoted() by Tomas V. V.
    * Cox, above.  Instead of going character by character through a
    * file, it goes character-by-character through a block of text.
    * While readQuoted() returns only one row from a file, parse()
    * returns all rows.
    * 
    * @author Paul M. Jones <pjones@ciawb.net>
    * 
    * @access public
    * 
    * @static
    * 
    * @param string $text The text to parse.
    * 
    * @param array &$conf The configuration of how to read the CSV text.
    * 
    * @return mixed Array of all rows and fields, or boolean false on
    * error.
    * 
    * @see File_CSV::readQuoted()
    * 
    */
    
    function parse($text, &$conf)
    {
        // -------------------------------------------------------------
        //
        // Initialization.
        //
        
        // ugly hack: force a \n on the end of the text so that the
        // parser reads the final line properly.  can't tell why
        // readQuoted() works properly in this respect.
        if (substr($text, -1) != "\n") {
            $text .= "\n";
        }
        
        // buffer for keeping field data, until the field is complete
        // and ready for including the row results
        $buff = null;
        
        // the current character the parser is working with
        $c = null;
        
        // the array of all data rows
        $data = array();
        
        // an array of fields from the current row
        $row = array();
        
        // the count of fields discovered in the current row
        $i = 1;
        
        // is our current "state" in the iteration as being inside a
        // quoted field, or not?
        $in_quote = false;
        
        // the length of the source text
        $text_len = strlen($text);
        
        // the current character number pointer
        $pointer = 0;
        
        
        // -------------------------------------------------------------
        //
        // The parsing mechanism.  Iterate through each character in the
        // source text and keep track of "state" (in-quote or not-in-quote,
        // when we hit a field-separator that is not in quotes, and when
        // we hit a row-separator that is not in quotes).
        //
        
        // check to see that the pointer value is not past the end
        // of the text.  then read the current-pointer character into
        // $ch and post-increment the pointer.
        //
        // (The $string{n} notation addresses character number n in
        // the string.)
        while ($pointer < $text_len && $ch = $text{$pointer++})
        {
            // remember the previous character (i.e., the character from
            // the last iteration)
            $prev = $c;
            
            // reset the current character to the character just read
            // from the source text (in the while() directive)
            $c = $ch;
            
            
            // Check the character to see how we should continue.
            //
            // Common case: the character is not a quote, it's not a
            // field separator, and it's not a row separator, so keep it
            // in the buffer and move on to the next character.
            if ($c != $conf['quote']
                && $c != $conf['sep']
                && $c != "\n"
            ) {
                
                // add the character to the buffer, then loop to the
                // next character in the source text.
                $buff .= $c;
                continue;
                
            }
            
            // Check the "in-quote" state.
            if ($c == $conf['quote']
                && $conf['quote']
                && ($prev == $conf['sep']
                    || $prev == "\n"
                    || $prev === null
                )
            ) {
                
                // The current character is a quote, and there is a
                // quote-character set, and the previous character was
                // one of: a field-separator, a row-separator, or null
                // (indicating that there was no previous character,
                // i.e., the beginning of the text).
                // 
                // Change the state to indicate that subsequent
                // characters read from the text are in quotes, so
                // field-sep and row-sep characters are treated as
                // escaped data.
                $in_quote = true;
                
            } elseif ($in_quote) {
            
                // We are currently in a quoted state; should we change
                // to the not-in-quote state?
                if ($c == $conf['sep']
                    && $prev == $conf['quote']
                ) {
                    
                    // The current character is a field-separator, and
                    // the previous character was a quote.  (This means
                    // we have skipped to the next field and should dump
                    // the buffer?)
                    $in_quote = false;
                    
                } elseif ($c == "\n") {
                    
                    // The current character is a row-separator.
                    
                    // Create a substring check-length comparison.
                    
                    // The current char is \n; if the previous char is \r,
                    // set the length to 2.  Default length is 1.
                    $sub = ($prev == "\r") ? 2 : 1;
                    
                    // !!! Why do this?
                    if ((strlen($buff) >= $sub)
                        && ($buff{strlen($buff) - $sub} == $conf['quote'])
                    ) {
                        // The length of the buffer is same as or longer
                        // than the substring check-length.  Also, the
                        // last character in the buffer, just before the
                        // substring check-length, is a quote. Change
                        // state to not-in-quotes.
                        $in_quote = false;
                    }
                }
            }
            
            
            // We've checked the quote-state (and changed it as necessary), so
            // continue parsing with the current character.
            //
            // If we are not in a quote-state, and the current character
            // is either a field-separator or a row-separator.
            if (! $in_quote
                && ($c == $conf['sep'] || $c == "\n")
            ) {
                
                // Are there more fields than expected?
                if ($c == $conf['sep']
                    && count($row) + 1 == $conf['fields']
                ) {
                    // The current character is a field-separator, but the 
                    // count of the returned fields is already at the expected
                    // level (i.e., there are more fields to read but we only
                    // expected as many as we already have).
                    //
                    // Read to the end of the line (to put the pointer at the
                    // beginning of the next line...
                    
                    //while ($c != "\n") {
                    //    $c = fgetc($fp);
                    //}
                    
                    // ... and error out.
                    File_CSV::raiseError("Read more fields than the expected '".
                        $conf['fields'] . "'");
                    
                    // Done!
                    return false;
                }
                
                // Are there fewer fields than expected?
                if ($c == "\n"
                    && $i != $conf['fields'] // !!! should this be less-than?
                ) {
                    // The current character is a row-separator, but the 
                    // current field-number is not the same as the the
                    // expected count.
                    File_CSV::raiseError("Read wrong fields number, count '"
                        . $i . "' but expected '" . $conf['fields'] . "'");
                    
                    // Done!
                    return false;
                }
                
                // The previous character was a \r, so it is the last character
                // in the buffer.  Delete it out of the buffer.
                // !!! Why?
                if ($prev == "\r") {
                    $buff = substr($buff, 0, -1);
                }
                
                // Unquote the buffer (i.e., the contents of the current field)
                // and keep it in the row-return results.
                $row[] = File_CSV::unquote($buff, $conf['quote']);
                
                // are we finished yet?
                if (count($row) == $conf['fields']) {
                    
                    // the count of fields to be returned matches the
                    // expected number of fields; we're done with the
                    // current row.
                    
                    // add the current row data to the set of all
                    // rows.
                    $data[] = $row;
                    
                    // reset the current row to nothing so we can start
                    // capturing new field content.
                    $row = array();
                    
                    // reset the current field count.
                    $i = 1;
                    
                    // reset the buffer
                    $buff = '';
                    
                    // iterate to the next character in the text.
                    continue;
                    
                } else {
                    
                    // otherwise, reset the buffer, increase the field-count,
                    // and iterate to the next character in the text.
                    $buff = '';
                    $i++;
                    continue;
                }
            }
            
            // default action after all this:  add the current character to the
            // field contents buffer.
            $buff .= $c;
            
        } // end while
        
        return $data;
    }
}
?>