<?php

    /**
    *
    * Parses a block of CSV-formatted text and returns an array of all
    * rows and fields.
    * 
    * This method is very heavily based on readQuoted() by Tomas V. V.
    * Cox, above.  Instead of going character by character through a
    * file, it goes character-by-character through a block of text.
    * While readQuoted() returns only one row from a file, parse()
    * returns all rows.
    * 
    * @author Paul M. Jones <pmjones@ciawb.net>
    * 
    * @access public
    * 
    * @static
    * 
    * @param string $text The text to parse.
    * 
    * @param array &$conf The configuration of how to read the CSV text.
    * 
    * @return mixed Array of all rows and fields, or boolean false on
    * error.
    * 
    * @see File_CSV::readQuoted()
    * 
    */
    
    function parse($text, &$conf)
    {
        // -------------------------------------------------------------
        //
        // Initialization.
        //
        
        // ugly hack: force a \n on the end of the text so that the
        // parser reads the final line properly.  can't tell why
        // readQuoted() works properly in this respect.
        if (substr($text, -1) != "\n") {
            $text .= "\n";
        }
        
        // buffer for keeping field data, until the field is complete
        // and ready for including the row results
        $buff = null;
        
        // the current character the parser is working with
        $c = null;
        
        // the array of all data rows
        $data = array();
        
        // an array of fields from the current row
        $row = array();
        
        // the count of fields discovered in the current row
        $i = 1;
        
        // is our current "state" in the iteration as being inside a
        // quoted field, or not?
        $in_quote = false;
        
        // the length of the source text
        $text_len = strlen($text);
        
        // the current character number pointer
        $pointer = 0;
        
        
        // -------------------------------------------------------------
        //
        // The parsing mechanism.  Iterate through each character in the
        // source text and keep track of "state" (in-quote or not-in-quote,
        // when we hit a field-separator that is not in quotes, and when
        // we hit a row-separator that is not in quotes).
        //
        
        // check to see that the pointer value is not past the end
        // of the text.  then read the current-pointer character into
        // $ch and post-increment the pointer.
        //
        // (The $string{n} notation addresses character number n in
        // the string.)
        while ($pointer < $text_len && $ch = $text{$pointer++})
        {
            // remember the previous character (i.e., the character from
            // the last iteration)
            $prev = $c;
            
            // reset the current character to the character just read
            // from the source text (in the while() directive)
            $c = $ch;
            
            
            // Check the character to see how we should continue.
            //
            // Common case: the character is not a quote, it's not a
            // field separator, and it's not a row separator, so keep it
            // in the buffer and move on to the next character.
            if ($c != $conf['quote']
                && $c != $conf['sep']
                && $c != "\n"
            ) {
                
                // add the character to the buffer, then loop to the
                // next character in the source text.
                $buff .= $c;
                continue;
                
            }
            
            // Check the "in-quote" state.
            if ($c == $conf['quote']
                && $conf['quote']
                && ($prev == $conf['sep']
                    || $prev == "\n"
                    || $prev === null
                )
            ) {
                
                // The current character is a quote, and there is a
                // quote-character set, and the previous character was
                // one of: a field-separator, a row-separator, or null
                // (indicating that there was no previous character,
                // i.e., the beginning of the text).
                // 
                // Change the state to indicate that subsequent
                // characters read from the text are in quotes, so
                // field-sep and row-sep characters are treated as
                // escaped data.
                $in_quote = true;
                
            } elseif ($in_quote) {
            
                // We are currently in a quoted state; should we change
                // to the not-in-quote state?
                if ($c == $conf['sep']
                    && $prev == $conf['quote']
                ) {
                    
                    // The current character is a field-separator, and
                    // the previous character was a quote.  (This means
                    // we have skipped to the next field and should dump
                    // the buffer?)
                    $in_quote = false;
                    
                } elseif ($c == "\n") {
                    
                    // The current character is a row-separator.
                    
                    // Create a substring check-length comparison.
                    
                    // The current char is \n; if the previous char is \r,
                    // set the length to 2.  Default length is 1.
                    $sub = ($prev == "\r") ? 2 : 1;
                    
                    // !!! Why do this?
                    if ((strlen($buff) >= $sub)
                        && ($buff{strlen($buff) - $sub} == $conf['quote'])
                    ) {
                        // The length of the buffer is same as or longer
                        // than the substring check-length.  Also, the
                        // last character in the buffer, just before the
                        // substring check-length, is a quote. Change
                        // state to not-in-quotes.
                        $in_quote = false;
                    }
                }
            }
            
            
            // We've checked the quote-state (and changed it as necessary), so
            // continue parsing with the current character.
            //
            // If we are not in a quote-state, and the current character
            // is either a field-separator or a row-separator.
            if (! $in_quote
                && ($c == $conf['sep'] || $c == "\n")
            ) {
                
                // Are there more fields than expected?
                if ($c == $conf['sep']
                    && count($row) + 1 == $conf['fields']
                ) {
                    // The current character is a field-separator, but the 
                    // count of the returned fields is already at the expected
                    // level (i.e., there are more fields to read but we only
                    // expected as many as we already have).
                    //
                    // Read to the end of the line (to put the pointer at the
                    // beginning of the next line...
                    
                    //while ($c != "\n") {
                    //    $c = fgetc($fp);
                    //}
                    
                    // ... and error out.
                    File_CSV::raiseError("Read more fields than the expected '".
                        $conf['fields'] . "'");
                    
                    // Done!
                    return false;
                }
                
                // Are there fewer fields than expected?
                if ($c == "\n"
                    && $i != $conf['fields'] // !!! should this be less-than?
                ) {
                    // The current character is a row-separator, but the 
                    // current field-number is not the same as the the
                    // expected count.
                    File_CSV::raiseError("Read wrong fields number, count '"
                        . $i . "' but expected '" . $conf['fields'] . "'");
                    
                    // Done!
                    return false;
                }
                
                // The previous character was a \r, so it is the last character
                // in the buffer.  Delete it out of the buffer.
                // !!! Why?
                if ($prev == "\r") {
                    $buff = substr($buff, 0, -1);
                }
                
                // Unquote the buffer (i.e., the contents of the current field)
                // and keep it in the row-return results.
                $row[] = File_CSV::unquote($buff, $conf['quote']);
                
                // are we finished yet?
                if (count($row) == $conf['fields']) {
                    
                    // the count of fields to be returned matches the
                    // expected number of fields; we're done with the
                    // current row.
                    
                    // add the current row data to the set of all
                    // rows.
                    $data[] = $row;
                    
                    // reset the current row to nothing so we can start
                    // capturing new field content.
                    $row = array();
                    
                    // reset the current field count.
                    $i = 1;
                    
                    // reset the buffer
                    $buff = '';
                    
                    // iterate to the next character in the text.
                    continue;
                    
                } else {
                    
                    // otherwise, reset the buffer, increase the field-count,
                    // and iterate to the next character in the text.
                    $buff = '';
                    $i++;
                    continue;
                }
            }
            
            // default action after all this:  add the current character to the
            // field contents buffer.
            $buff .= $c;
            
        } // end while
        
        return $data;
    }

?>