diff --git a/ext/csv/LICENSE b/ext/csv/LICENSE new file mode 100644 index 000000000000..531871e072f6 --- /dev/null +++ b/ext/csv/LICENSE @@ -0,0 +1,27 @@ +Copyright © 2020-2026, Gina Peter Banyard and Contributors. +Copyright © 1999-2019, The PHP Group. + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are met: + +1. Redistributions of source code must retain the above copyright notice, this + list of conditions and the following disclaimer. + +2. Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + +3. Neither the name of the copyright holder nor the names of its + contributors may be used to endorse or promote products derived from + this software without specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" +AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE +IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE +DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE +FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL +DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR +SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER +CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, +OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/ext/csv/config.m4 b/ext/csv/config.m4 new file mode 100644 index 000000000000..f5ff7623635d --- /dev/null +++ b/ext/csv/config.m4 @@ -0,0 +1,14 @@ +dnl config.m4 for extension csv + +PHP_ARG_ENABLE([csv], + [whether to enable CSV support], + [AS_HELP_STRING([--disable-csv], + [Disable CSV support])], + [yes]) + +if test "$PHP_CSV" != "no"; then + AC_DEFINE([HAVE_CSV], [1], + [Define to 1 if the PHP extension 'csv' is available.]) + PHP_NEW_EXTENSION([csv], [csv.c], [$ext_shared]) + PHP_INSTALL_HEADERS([ext/csv], [php_csv.h]) +fi diff --git a/ext/csv/config.w32 b/ext/csv/config.w32 new file mode 100644 index 000000000000..f02ecfc2cf8c --- /dev/null +++ b/ext/csv/config.w32 @@ -0,0 +1,9 @@ +// vim:ft=javascript + +ARG_ENABLE("csv", "CSV support", "yes"); + +if (PHP_CSV != "no") { + EXTENSION("csv", "csv.c"); + AC_DEFINE("HAVE_CSV", 1, "Define to 1 if the PHP extension 'csv' is available."); + PHP_INSTALL_HEADERS("ext/csv", "php_csv.h"); +} diff --git a/ext/csv/csv.c b/ext/csv/csv.c new file mode 100644 index 000000000000..04fba9ec0b66 --- /dev/null +++ b/ext/csv/csv.c @@ -0,0 +1,1517 @@ +/* + +----------------------------------------------------------------------+ + | Copyright © 2020-2026, Gina Peter Banyard and Contributors. | + +----------------------------------------------------------------------+ + | This source file is subject to the Modified BSD License that is | + | bundled with this package in the file LICENSE, and is available | + | through the World Wide Web at . | + | | + | SPDX-License-Identifier: BSD-3-Clause | + +----------------------------------------------------------------------+ + | Authors: Gina Peter Banyard | + | Damian Jóźwiak | + +----------------------------------------------------------------------+ +*/ +/* csv extension for PHP */ + +#ifdef HAVE_CONFIG_H +# include "config.h" +#endif + +#include "php.h" +#include "php_streams.h" +#include "ext/standard/info.h" +#include "ext/standard/file.h" /* For the default stream context */ +#include "ext/standard/php_string.h" /* For php_str_to_str() */ +#include "php_csv.h" +#include "csv_arginfo.h" + +#include +#include "zend_smart_str.h" +#include "zend_interfaces.h" +#include "zend_exceptions.h" + +PHP_MINFO_FUNCTION(csv) +{ + php_info_print_table_start(); + php_info_print_table_row(2, "CSV support", "enabled"); + php_info_print_table_end(); +} + +#define EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(delimiter_arg_num, enclosure_arg_num, eol_arg_num) { \ + if (eol_sequence) { \ + /* Make sure that there is at least one character in string */ \ + if (UNEXPECTED(ZSTR_LEN(eol_sequence) == 0)) { \ + zend_argument_must_not_be_empty_error(eol_arg_num); \ + RETURN_THROWS(); \ + } \ + zend_string_addref(eol_sequence); \ + } else { \ + eol_sequence = zend_string_init(ZEND_STRL("\r\n"), 0); \ + } \ + \ + if (delimiter) { \ + /* Make sure that there is at least one character in string */ \ + if (UNEXPECTED(ZSTR_LEN(delimiter) == 0)) { \ + zend_argument_must_not_be_empty_error(delimiter_arg_num); \ + zend_string_release(eol_sequence); \ + RETURN_THROWS(); \ + } \ + if (UNEXPECTED(zend_string_equals(delimiter, eol_sequence))) { \ + zend_argument_value_error( \ + eol_arg_num, \ + "must not be identical to argument #%"PRIu32" ($delimiter)", \ + delimiter_arg_num); \ + zend_string_release(eol_sequence); \ + RETURN_THROWS(); \ + } \ + zend_string_addref(delimiter); \ + } else { \ + delimiter = ZSTR_CHAR(','); \ + } \ + \ + if (enclosure) { \ + if (UNEXPECTED(ZSTR_LEN(enclosure) == 0)) { \ + zend_argument_must_not_be_empty_error(enclosure_arg_num); \ + zend_string_release(delimiter); \ + zend_string_release(eol_sequence); \ + RETURN_THROWS(); \ + } \ + if (UNEXPECTED(zend_string_equals(enclosure, eol_sequence))) { \ + zend_argument_value_error( \ + eol_arg_num, \ + "must not be identical to argument #%"PRIu32" ($enclosure)", \ + enclosure_arg_num); \ + zend_string_release(eol_sequence); \ + zend_string_release(delimiter); \ + RETURN_THROWS(); \ + } \ + zend_string_addref(enclosure); \ + } else { \ + enclosure = ZSTR_CHAR('"'); \ + } \ + \ + /* Ensure delimiter and enclosure are different */ \ + if (UNEXPECTED(zend_string_equals(delimiter, enclosure))) { \ + zend_argument_value_error( \ + enclosure_arg_num, \ + "must not be identical to argument #%"PRIu32" ($delimiter)", \ + delimiter_arg_num); \ + zend_string_release(eol_sequence); \ + zend_string_release(delimiter); \ + zend_string_release(enclosure); \ + RETURN_THROWS(); \ + } \ +} + +static bool zend_string_contains(const zend_string *haystack, const zend_string *needle) { + return php_memnstr(ZSTR_VAL(haystack), ZSTR_VAL(needle), ZSTR_LEN(needle), ZSTR_VAL(haystack) + ZSTR_LEN(haystack)); +} + +static zend_always_inline bool buffer_starts_with_zend_string(const char *buffer, const char *end, const zend_string *needle) { + size_t buffer_len = end-buffer; + size_t needle_len = ZSTR_LEN(needle); + /* Tokens are not empty; comparing the first byte inline avoids a memcmp() call per byte */ + return buffer_len >= needle_len + && buffer[0] == ZSTR_VAL(needle)[0] + && (needle_len == 1 || !memcmp(buffer + 1, ZSTR_VAL(needle) + 1, needle_len - 1)); +} + +/** + * Whether a proper prefix of the enclosure sequence is also a suffix of it (e.g. "aa", "--"). + * With such an enclosure, the bytes before the end of an enclosed field are ambiguous: the field + * "a" enclosed with "aa" is written as "aaaaa", where the enclosure sequence also occurs one byte + * before the closing one. The closing enclosure is then the one that is followed by the delimiter, + * the EOL sequence or the end of the input. + */ +static bool csv_enclosure_overlaps_itself(const zend_string *enclosure) { + for (size_t k = 1; k < ZSTR_LEN(enclosure); k++) { + if (!memcmp(ZSTR_VAL(enclosure), ZSTR_VAL(enclosure) + ZSTR_LEN(enclosure) - k, k)) { + return true; + } + } + return false; +} + +static bool csv_is_at_end_of_field( + const char *position, + const char *end, + const zend_string *delimiter, + const zend_string *eol_sequence +) { + return position == end + || buffer_starts_with_zend_string(position, end, delimiter) + || buffer_starts_with_zend_string(position, end, eol_sequence); +} + +/* The tokens of a CSV dialect, with what the parser and the row boundary scanner derive from them */ +typedef struct csv_dialect { + const zend_string *delimiter; + const zend_string *enclosure; + const zend_string *eol_sequence; + bool enclosure_overlaps_itself; + char enclosure_first_byte; + /* Outside of an escaped field, the bytes that may start a token or must be rejected (CR and + * LF); all other bytes are data and are skipped without matching any token */ + bool is_special_byte[256]; +} csv_dialect; + +static void csv_dialect_init( + csv_dialect *dialect, + const zend_string *delimiter, + const zend_string *enclosure, + const zend_string *eol_sequence +) { + dialect->delimiter = delimiter; + dialect->enclosure = enclosure; + dialect->eol_sequence = eol_sequence; + dialect->enclosure_overlaps_itself = csv_enclosure_overlaps_itself(enclosure); + dialect->enclosure_first_byte = ZSTR_VAL(enclosure)[0]; + memset(dialect->is_special_byte, 0, sizeof(dialect->is_special_byte)); + dialect->is_special_byte[(unsigned char) ZSTR_VAL(delimiter)[0]] = true; + dialect->is_special_byte[(unsigned char) ZSTR_VAL(enclosure)[0]] = true; + dialect->is_special_byte[(unsigned char) ZSTR_VAL(eol_sequence)[0]] = true; + dialect->is_special_byte['\r'] = true; + dialect->is_special_byte['\n'] = true; +} + +/* Append the field made of field_value followed by [start, end) to the row */ +static zend_always_inline void csv_add_field(HashTable *row, smart_str *field_value, const char *start, const char *end) { + zval tmp; + if (field_value->s == NULL) { + ZVAL_STR(&tmp, zend_string_init_fast(start, end - start)); + } else { + smart_str_appendl(field_value, start, end - start); + ZVAL_STR(&tmp, smart_str_extract(field_value)); + } + zend_hash_next_index_insert_new(row, &tmp); +} + +/** + * Follows RFC4180 https://tools.ietf.org/html/rfc4180 + * The terminology can be slightly confusing here as what PHP considers the 'enclosure' is what is used + * to escape a field in the RFC. + * + * This only formats ONE row of a CSV file. + * + * Returns a NULL pointer on error + */ +static zend_string* hashtable_to_rfc4180_string( + HashTable *fields, + const zend_string *delimiter, + const zend_string *enclosure, + const zend_string *eol_sequence +) { + uint32_t nb_fields; + uint32_t fields_iterated = 0; + zval *tmp_field_zval; + smart_str row = {0}; + + nb_fields = zend_hash_num_elements(fields); + ZEND_HASH_FOREACH_VAL(fields, tmp_field_zval) { + bool escape_field = false; + bool enclosure_within_filed = false; + zend_string *tmp_field_str; + zend_string *field_str = zval_try_get_tmp_string(tmp_field_zval, &tmp_field_str); + if (UNEXPECTED(field_str == NULL)) { + smart_str_free(&row); + return NULL; + } + + /* + * A field must be escaped (enclosed) if it contains the delimiter OR a Carriage Return (\r) + * OR a Line Feed (\n) OR the field escape sequence (i.e. enclosure parameter) OR the custom EOL sequence + */ + if ( + memchr(ZSTR_VAL(field_str), '\n', ZSTR_LEN(field_str)) + || memchr(ZSTR_VAL(field_str), '\r', ZSTR_LEN(field_str)) + || zend_string_contains(field_str, delimiter) + || zend_string_contains(field_str, eol_sequence) + ) { + escape_field = true; + } + /* If the field escape sequence (i.e. enclosure parameter) is within the field it needs to be + * duplicated within the field. */ + if (zend_string_contains(field_str, enclosure)) { + escape_field = true; + enclosure_within_filed = true; + } + + if (escape_field) { + smart_str_append(&row, enclosure); + + if (enclosure_within_filed) { + /* Create replace string (twice the field escape sequence) */ + smart_str escaped_enclosure = {0}; + smart_str_append(&escaped_enclosure, enclosure); + smart_str_append(&escaped_enclosure, enclosure); + smart_str_0(&escaped_enclosure); + + zend_string *replace = php_str_to_str(ZSTR_VAL(field_str), ZSTR_LEN(field_str), ZSTR_VAL(enclosure), + ZSTR_LEN(enclosure), ZSTR_VAL(escaped_enclosure.s), ZSTR_LEN(escaped_enclosure.s)); + + smart_str_append(&row, replace); + + smart_str_free(&escaped_enclosure); + zend_string_release(replace); + } else { + smart_str_append(&row, field_str); + } + + smart_str_append(&row, enclosure); + } else { + smart_str_append(&row, field_str); + } + + /* Only add the delimiter in between fields on the same row. */ + if (++fields_iterated != nb_fields) { + smart_str_append(&row, delimiter); + } + + /* Clear temporary variable */ + zend_tmp_string_release(tmp_field_str); + } ZEND_HASH_FOREACH_END(); + /* Add the EOL sequence to indicate the end of the row. */ + smart_str_append(&row, eol_sequence); + + return smart_str_extract(&row); +} + +static zend_object_iterator* csv_init_iterator_foreach(zval *iterator_zval, zval **first_iteration_value) { + ZEND_ASSERT(Z_TYPE_P(iterator_zval) == IS_OBJECT); + ZEND_ASSERT(first_iteration_value != NULL); + + zend_class_entry *ce = Z_OBJCE_P(iterator_zval); + ZEND_ASSERT(ce->iterator_funcs_ptr != NULL); + ZEND_ASSERT(ce->get_iterator != NULL); + + zend_object_iterator *it = ce->get_iterator(ce, iterator_zval, 0); + if (UNEXPECTED(EG(exception))) { + if (UNEXPECTED(it)) { + zend_iterator_dtor(it); + } + return NULL; + } + ZEND_ASSERT(it != NULL); + + /* Rewind iterator */ + it->index = 0; + if (it->funcs->rewind) { + it->funcs->rewind(it); + if (UNEXPECTED(EG(exception))) { + zend_iterator_dtor(it); + return NULL; + } + } + + if (it->funcs->valid(it) != SUCCESS || UNEXPECTED(EG(exception))) { + zend_iterator_dtor(it); + return NULL; + } + + *first_iteration_value = it->funcs->get_current_data(it); + if (*first_iteration_value == NULL || UNEXPECTED(EG(exception))) { + zend_iterator_dtor(it); + return NULL; + } + + return it; +} + +static zval* csv_advance_iterator_foreach(zend_object_iterator *it) { + ZEND_ASSERT(it != NULL); + + /* Move iterator forward */ + it->index++; + it->funcs->move_forward(it); + if (UNEXPECTED(EG(exception))) { + zend_iterator_dtor(it); + return NULL; + } + + if (it->funcs->valid(it) != SUCCESS || UNEXPECTED(EG(exception))) { + zend_iterator_dtor(it); + return NULL; + } + + zval *next_value = it->funcs->get_current_data(it); + if (next_value == NULL || UNEXPECTED(EG(exception))) { + zend_iterator_dtor(it); + return NULL; + } + + return next_value; +} + +#define CSV_ITERABLE_FOREACH_VAL(____iterable_zval, __value, __iterator_error_label) \ + do { \ + bool __is_array = true; \ + bool __is_foreach_loop_valid = true; \ + zval *__loop_value = NULL; \ + /* Needed for iterators */ \ + zend_object_iterator *__zend_iterator = NULL; \ + /* Copied from _ZEND_HASH_FOREACH_VAL */ \ + const HashTable *__ht = NULL; \ + uint32_t _count = 0; \ + size_t _size = 0; \ + if (Z_TYPE_P(____iterable_zval) == IS_ARRAY) { \ + __ht = Z_ARRVAL_P(____iterable_zval); \ + _count = __ht->nNumUsed; \ + _size = ZEND_HASH_ELEMENT_SIZE(__ht); \ + __loop_value = __ht->arPacked; \ + __is_foreach_loop_valid = _count > 0; \ + } else { \ + __is_array = false; \ + __zend_iterator = csv_init_iterator_foreach(____iterable_zval, &__loop_value); \ + if (UNEXPECTED(__zend_iterator == NULL)) { \ + goto __iterator_error_label; \ + } \ + } \ + while (__is_foreach_loop_valid) { \ + if (__is_array) { \ + /* Skip holes (e.g. after unset()) while advancing, otherwise the loop never progresses */ \ + while (__is_foreach_loop_valid && UNEXPECTED(Z_TYPE_P(__loop_value) == IS_UNDEF)) { \ + _count--; \ + __loop_value = ZEND_HASH_NEXT_ELEMENT(__loop_value, _size); \ + __is_foreach_loop_valid = _count > 0; \ + } \ + if (!__is_foreach_loop_valid) { \ + break; \ + } \ + } \ + __value = __loop_value; + +#define CSV_ITERABLE_FOREACH_END() \ + do { \ + if (__is_array) { \ + _count--; \ + __loop_value = ZEND_HASH_NEXT_ELEMENT(__loop_value, _size); \ + __is_foreach_loop_valid = _count > 0; \ + } else { \ + __loop_value = csv_advance_iterator_foreach(__zend_iterator); \ + if (__loop_value == NULL) { \ + __is_foreach_loop_valid = false; \ + } \ + } \ + } while (false); \ + } /* end while (__is_foreach_loop_valid) */ \ + if (!__is_array && __is_foreach_loop_valid) { \ + /* The loop was aborted with break: the iterator has not been destroyed yet */ \ + zend_iterator_dtor(__zend_iterator); \ + } \ + } while (false) + +/* Returns a NULL pointer on error */ +static HashTable* rfc4180_string_to_hashtable( + const char **buffer, + const char *end_buffer, + const csv_dialect *dialect +) { + HashTable *return_value = zend_new_array(8); + const zend_string *delimiter = dialect->delimiter; + const zend_string *enclosure = dialect->enclosure; + const zend_string *eol_sequence = dialect->eol_sequence; + + bool in_escaped_field = false; + bool at_start_of_field = true; + /* A closing enclosure must be followed by a delimiter, an EOL sequence or the end of input */ + bool after_closing_enclosure = false; + + /* Dereference buffer */ + const char *row = *buffer; + /* row_to_array('') passes an empty buffer, which is a single empty field */ + ZEND_ASSERT(row <= end_buffer); + + /* The value of the current field is field_value followed by the bytes [segment_start, + * segment_end) of the buffer. field_value is only used once a doubled enclosure sequence + * splits an escaped field into several segments; otherwise the field is created from the + * buffer directly. */ + smart_str field_value = {0}; + const char *segment_start = row; + const char *segment_end = row; + + /* Main loop to "tokenize" the row */ + while (row < end_buffer) { + if (in_escaped_field) { + /* Only an enclosure sequence ends or interrupts an escaped field */ + const char *next = memchr(row, dialect->enclosure_first_byte, end_buffer - row); + if (next == NULL) { + row = end_buffer; + break; + } + row = next; + if (!buffer_starts_with_zend_string(row, end_buffer, enclosure)) { + row++; + continue; + } + const char *enclosure_start = row; + row += ZSTR_LEN(enclosure); + + /* A doubled enclosure sequence stays in the field once. Consume the whole sequence + * at once, otherwise a multibyte enclosure would be re-matched against its own tail + * bytes. */ + if (buffer_starts_with_zend_string(row, end_buffer, enclosure)) { + smart_str_appendl(&field_value, segment_start, row - segment_start); + row += ZSTR_LEN(enclosure); + segment_start = row; + continue; + } + + /* See csv_enclosure_overlaps_itself(): the enclosure sequence starting one byte later + * may be the closing one, so this byte is data. */ + if (dialect->enclosure_overlaps_itself + && !csv_is_at_end_of_field(row, end_buffer, delimiter, eol_sequence)) { + row = enclosure_start + 1; + continue; + } + + in_escaped_field = false; + after_closing_enclosure = true; + segment_end = enclosure_start; + continue; + } + + /* Bytes that cannot start a token are data */ + if (!dialect->is_special_byte[(unsigned char) *row]) { + if (UNEXPECTED(after_closing_enclosure)) { + zend_value_error("Closing enclosure sequence must be followed by the delimiter or the EOL sequence"); + goto error; + } + do { + row++; + } while (row < end_buffer && !dialect->is_special_byte[(unsigned char) *row]); + at_start_of_field = false; + segment_end = row; + continue; + } + + /* Check for field escape sequence (i.e. enclosure) */ + if (buffer_starts_with_zend_string(row, end_buffer, enclosure)) { + if (UNEXPECTED(after_closing_enclosure)) { + zend_value_error("Closing enclosure sequence must be followed by the delimiter or the EOL sequence"); + goto error; + } + if (UNEXPECTED(!at_start_of_field)) { + zend_value_error("Enclosure sequence is used in a non escaped field"); + goto error; + } + row += ZSTR_LEN(enclosure); + in_escaped_field = true; + at_start_of_field = false; + segment_start = row; + continue; + } + + /* Check for delimiter */ + if (buffer_starts_with_zend_string(row, end_buffer, delimiter)) { + csv_add_field(return_value, &field_value, segment_start, segment_end); + row += ZSTR_LEN(delimiter); + at_start_of_field = true; + after_closing_enclosure = false; + segment_start = row; + segment_end = row; + continue; + } + + /* Check for End Of Line sequence */ + if (buffer_starts_with_zend_string(row, end_buffer, eol_sequence)) { + row += ZSTR_LEN(eol_sequence); + goto eol; + } + + /* A byte that may start a token but does not */ + if (UNEXPECTED(after_closing_enclosure)) { + zend_value_error("Closing enclosure sequence must be followed by the delimiter or the EOL sequence"); + goto error; + } + /* RFC 4180 only allows CR and LF inside enclosed fields; outside of them they are + * either part of the EOL sequence or a sign that the input uses a different one */ + if (UNEXPECTED(*row == '\r' || *row == '\n')) { + zend_value_error("A non escaped field must not contain CR or LF characters that are not part of the EOL sequence"); + goto error; + } + at_start_of_field = false; + row++; + segment_end = row; + } + + if (UNEXPECTED(in_escaped_field)) { + zend_value_error("Enclosure sequence is not closed"); + goto error; + } + + eol:; + csv_add_field(return_value, &field_value, segment_start, segment_end); + + /* Update outer buffer position */ + *buffer = row; + + return return_value; + +error: + smart_str_free(&field_value); + zend_array_destroy(return_value); + return NULL; +} + +/* Returns a NULL pointer on error */ +static HashTable* rfc4180_buffer_to_hashtable_collection( + const zend_string *buffer, + const zend_string *delimiter, + const zend_string *enclosure, + const zend_string *eol_sequence, + bool is_lax +) { + HashTable *return_value = zend_new_array(8); + const char *start_position = ZSTR_VAL(buffer); + const char *current_position = ZSTR_VAL(buffer); + const char *end_buffer = ZSTR_VAL(buffer) + ZSTR_LEN(buffer); + size_t length = ZSTR_LEN(buffer); + size_t buffer_row_nb = 1; + uint32_t nb_fields = 0; + csv_dialect dialect; + csv_dialect_init(&dialect, delimiter, enclosure, eol_sequence); + + while (current_position - start_position < (ptrdiff_t) length) { + uint32_t new_nb_fields = 0; + HashTable *row = rfc4180_string_to_hashtable(¤t_position, end_buffer, &dialect); + + /* Issue with parsing row */ + if (row == NULL) { + zend_array_destroy(return_value); + return NULL; + } + + /* used to check if the number of fields is equal in each iteration */ + new_nb_fields = zend_hash_num_elements(row); + /* buffer_row_nb == 1 means we are at the first iteration so don't check */ + if (new_nb_fields != nb_fields && buffer_row_nb != 1 && !is_lax) { + zend_value_error("Buffer row %zu contains %"PRIu32" fields compared to %"PRIu32" fields on previous rows", + buffer_row_nb, new_nb_fields, nb_fields); + zend_array_destroy(row); + zend_array_destroy(return_value); + return NULL; + } + nb_fields = new_nb_fields; + buffer_row_nb++; + + zval tmp; + ZVAL_ARR(&tmp, row); + zend_hash_next_index_insert(return_value, &tmp); + } + + return return_value; +} + +PHP_FUNCTION(Csv_array_to_row) +{ + zend_string *delimiter = NULL; + zend_string *enclosure = NULL; + zend_string *eol_sequence = NULL; + HashTable *fields; + + if (zend_parse_parameters(ZEND_NUM_ARGS(), "h|SSS", &fields, &delimiter, &enclosure, &eol_sequence) == FAILURE) { + RETURN_THROWS(); + } + + EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(2, 3, 4); + + zend_string *result = hashtable_to_rfc4180_string(fields, delimiter, enclosure, eol_sequence); + zend_string_release(eol_sequence); + zend_string_release(delimiter); + zend_string_release(enclosure); + + if (UNEXPECTED(result == NULL)) { + RETURN_THROWS(); + } + + RETURN_STR(result); +} + +PHP_FUNCTION(Csv_collection_to_buffer) +{ + zend_string *delimiter = NULL; + zend_string *enclosure = NULL; + zend_string *eol_sequence = NULL; + zval *collection; + + ZEND_PARSE_PARAMETERS_START(1, 4) + Z_PARAM_ITERABLE(collection) + Z_PARAM_OPTIONAL + Z_PARAM_STR(delimiter) + Z_PARAM_STR(enclosure) + Z_PARAM_STR(eol_sequence) + ZEND_PARSE_PARAMETERS_END(); + + EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(2, 3, 4); + + zval *fields; + size_t collection_index = 0; + uint32_t nb_fields = 0; + smart_str buffer = {0}; + bool has_errors = false; + CSV_ITERABLE_FOREACH_VAL(collection, fields, end) { + /* Check that fields is an array */ + ZVAL_DEREF(fields); + if (UNEXPECTED(Z_TYPE_P(fields) != IS_ARRAY)) { + zend_type_error("Element %zu of the collection must be an array", collection_index); + has_errors = true; + break; + } + + /* used to check if the number of fields is equal in each iteration */ + uint32_t new_nb_fields = zend_hash_num_elements(Z_ARRVAL_P(fields)); + /* collection_index == 0 means we are at the first iteration so don't check */ + if (nb_fields != new_nb_fields && collection_index != 0) { + zend_value_error("Element %zu of the collection contains %"PRIu32 + " fields compared to %"PRIu32" fields on previous rows", + collection_index, new_nb_fields, nb_fields); + has_errors = true; + break; + } + nb_fields = new_nb_fields; + collection_index++; + + /* Hold a reference to the row while formatting it: converting a field to string can run + * userland code (__toString()) that resumes a generator or overwrites the element of the + * collection, which would otherwise free the row while it is being iterated. */ + zval row; + ZVAL_COPY(&row, fields); + zend_string *result = hashtable_to_rfc4180_string(Z_ARRVAL(row), delimiter, enclosure, eol_sequence); + zval_ptr_dtor(&row); + if (UNEXPECTED(result == NULL)) { + has_errors = true; + break; + } + smart_str_append(&buffer, result); + zend_string_release(result); + } CSV_ITERABLE_FOREACH_END(); + + end: + /* Release strings */ + zend_string_release(eol_sequence); + zend_string_release(delimiter); + zend_string_release(enclosure); + /* If an Error/Exception has been thrown */ + if (has_errors || UNEXPECTED(EG(exception))) { + smart_str_free(&buffer); + RETURN_THROWS(); + } + + RETURN_STR(smart_str_extract(&buffer)); +} + +/* Open a stream the way SplFileObject does: through the default stream context (so that + * stream_context_set_default() applies), with the reason of a failure (e.g. "No such file or + * directory") reported by the wrapper turned into the message of the thrown Error. + * + * The stream stays registered as a resource, so the resource list closes it at shutdown if the + * owner never got to, but userland must not close it behind the owner's back (it can reach it + * through get_resources()): PHP_STREAM_FLAG_NO_FCLOSE makes fclose() refuse to do that. */ +static php_stream *php_csv_stream_open(const zend_string *file, const char *mode, const char *purpose) +{ + zend_error_handling error_handling; + zend_replace_error_handling(EH_THROW, zend_ce_error, &error_handling); + php_stream *stream = php_stream_open_wrapper_ex(ZSTR_VAL(file), mode, REPORT_ERRORS, NULL, + php_stream_context_from_zval(NULL, 0)); + zend_restore_error_handling(&error_handling); + + if (UNEXPECTED(stream == NULL)) { + if (!EG(exception)) { + zend_throw_error(NULL, "Failed to open \"%s\" for %s", ZSTR_VAL(file), purpose); + } + return NULL; + } + if (UNEXPECTED(EG(exception))) { + /* E.g. a userspace wrapper opened the stream but raised a warning while doing so */ + php_stream_close(stream); + return NULL; + } + + stream->flags |= PHP_STREAM_FLAG_NO_FCLOSE; + return stream; +} + +PHP_FUNCTION(Csv_collection_to_file) +{ + zend_string *file = NULL; + zend_string *delimiter = NULL; + zend_string *enclosure = NULL; + zend_string *eol_sequence = NULL; + zval *collection; + + ZEND_PARSE_PARAMETERS_START(2, 5) + Z_PARAM_PATH_STR(file) + Z_PARAM_ITERABLE(collection) + Z_PARAM_OPTIONAL + Z_PARAM_STR(delimiter) + Z_PARAM_STR(enclosure) + Z_PARAM_STR(eol_sequence) + ZEND_PARSE_PARAMETERS_END(); + + EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(3, 4, 5); + + php_stream *stream = php_csv_stream_open(file, "wb", "writing"); + if (UNEXPECTED(stream == NULL)) { + zend_string_release(eol_sequence); + zend_string_release(delimiter); + zend_string_release(enclosure); + RETURN_THROWS(); + } + + zval *fields; + size_t collection_index = 0; + uint32_t nb_fields = 0; + bool has_errors = false; + CSV_ITERABLE_FOREACH_VAL(collection, fields, end) { + /* Check that fields is an array */ + ZVAL_DEREF(fields); + if (UNEXPECTED(Z_TYPE_P(fields) != IS_ARRAY)) { + zend_type_error("Element %zu of the collection must be an array", collection_index); + has_errors = true; + break; + } + + /* used to check if the number of fields is equal in each iteration */ + uint32_t new_nb_fields = zend_hash_num_elements(Z_ARRVAL_P(fields)); + /* collection_index == 0 means we are at the first iteration so don't check */ + if (nb_fields != new_nb_fields && collection_index != 0) { + zend_value_error("Element %zu of the collection contains %"PRIu32 + " fields compared to %"PRIu32" fields on previous rows", + collection_index, new_nb_fields, nb_fields); + has_errors = true; + break; + } + nb_fields = new_nb_fields; + collection_index++; + + /* Hold a reference to the row while formatting it: converting a field to string can run + * userland code (__toString()) that resumes a generator or overwrites the element of the + * collection, which would otherwise free the row while it is being iterated. */ + zval row; + ZVAL_COPY(&row, fields); + zend_string *result = hashtable_to_rfc4180_string(Z_ARRVAL(row), delimiter, enclosure, eol_sequence); + zval_ptr_dtor(&row); + if (UNEXPECTED(result == NULL)) { + has_errors = true; + break; + } + + ssize_t bytes_written = php_stream_write(stream, ZSTR_VAL(result), ZSTR_LEN(result)); + bool is_short_write = bytes_written < 0 || (size_t) bytes_written != ZSTR_LEN(result); + zend_string_release(result); + if (UNEXPECTED(is_short_write)) { + zend_throw_error(NULL, "Failed to write element %zu of the collection to \"%s\"", + collection_index - 1, ZSTR_VAL(file)); + has_errors = true; + break; + } + } CSV_ITERABLE_FOREACH_END(); + + end:; + /* Flush explicitly: php_stream_free() flushes too but discards the result, so e.g. a + * userspace wrapper returning false from stream_flush() would go unnoticed. Closing + * can also fail on its own (e.g. a compressed stream finishing its deflate buffer); + * either failure means the file is incomplete even though every write succeeded. + * Known limitation, shared with userland fclose(): a failure signalled by a write + * filter only during its final ($closing) flush is not observable through the + * streams API and is reported as success. */ + int flush_status = has_errors || EG(exception) ? 0 : php_stream_flush(stream); + int close_status = php_stream_close(stream); + /* Release strings */ + zend_string_release(eol_sequence); + zend_string_release(delimiter); + zend_string_release(enclosure); + /* If an Error/Exception has been thrown */ + if (has_errors || UNEXPECTED(EG(exception))) { + RETURN_THROWS(); + } + if (UNEXPECTED(flush_status != 0 || close_status != 0)) { + zend_throw_error(NULL, "Failed to finish writing to \"%s\"", ZSTR_VAL(file)); + RETURN_THROWS(); + } +} + +PHP_FUNCTION(Csv_row_to_array) +{ + zend_string *delimiter = NULL; + zend_string *enclosure = NULL; + zend_string *eol_sequence = NULL; + zend_string *row; + HashTable *fields = NULL; + + if (zend_parse_parameters(ZEND_NUM_ARGS(), "S|SSS", &row, &delimiter, &enclosure, &eol_sequence) == FAILURE) { + RETURN_THROWS(); + } + + EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(2, 3, 4); + + const char *row_c = ZSTR_VAL(row); + const char *end_row = ZSTR_VAL(row) + ZSTR_LEN(row); + csv_dialect dialect; + csv_dialect_init(&dialect, delimiter, enclosure, eol_sequence); + fields = rfc4180_string_to_hashtable(&row_c, end_row, &dialect); + zend_string_release(eol_sequence); + zend_string_release(delimiter); + zend_string_release(enclosure); + + if (UNEXPECTED(fields == NULL)) { + RETURN_THROWS(); + } + + /* The row may end with the EOL sequence, but must not be followed by another one */ + if (UNEXPECTED(row_c != end_row)) { + zend_array_destroy(fields); + zend_argument_value_error(1, "must contain a single row"); + RETURN_THROWS(); + } + + RETURN_ARR(fields); +} + +static void buffer_to_collection_generic(INTERNAL_FUNCTION_PARAMETERS, bool is_lax) +{ + zend_string *delimiter = NULL; + zend_string *enclosure = NULL; + zend_string *eol_sequence = NULL; + zend_string *buffer; + + if (zend_parse_parameters(ZEND_NUM_ARGS(), "S|SSS", &buffer, &delimiter, &enclosure, &eol_sequence) == FAILURE) { + RETURN_THROWS(); + } + + EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(2, 3, 4); + + HashTable *collection = rfc4180_buffer_to_hashtable_collection(buffer, delimiter, enclosure, eol_sequence, is_lax); + + /* Release strings */ + zend_string_release(eol_sequence); + zend_string_release(delimiter); + zend_string_release(enclosure); + /* If an Error has been thrown */ + if (UNEXPECTED(collection == NULL)) { + RETURN_THROWS(); + } + RETURN_ARR(collection); +} + +PHP_FUNCTION(Csv_buffer_to_collection) +{ + buffer_to_collection_generic(INTERNAL_FUNCTION_PARAM_PASSTHRU, false); +} +PHP_FUNCTION(Csv_buffer_to_collection_lax) +{ + buffer_to_collection_generic(INTERNAL_FUNCTION_PARAM_PASSTHRU, true); +} + +static zend_class_entry *php_csv_lazy_collection_ce = NULL; +static zend_object_handlers php_csv_lazy_collection_object_handlers; + +/* How much data is read from the stream at once while looking for the end of a row */ +#define PHP_CSV_STREAM_CHUNK_SIZE 8192 +/* Once this many consumed bytes accumulate at the front of the stream buffer it is compacted, + * which is what keeps memory usage bounded for createFromFile() */ +#define PHP_CSV_STREAM_BUFFER_COMPACT_THRESHOLD (64 * 1024) + +typedef struct php_csv_lazy_collection_object { + /* Buffer mode (createFromBuffer): the whole CSV document; NULL in stream mode */ + zend_string *buffer; + /* Stream mode (createFromFile): the stream and a sliding window of not yet parsed data. The + * stream is closed (and set to NULL) when the object is destroyed. As a stream can only be + * read from one place, at most one iterator may be active at a time. */ + bool is_stream_mode; + bool stream_has_active_iterator; + bool stream_iteration_started; + php_stream *stream; + smart_str stream_buffer; + size_t stream_buffer_position; + /* Where csv_find_end_of_row() resumes scanning the row starting at stream_buffer_position + * after more data has been read, so that a long row is not rescanned from its start */ + size_t stream_scan_position; + bool stream_scan_in_escaped_field; + /* Common */ + zend_string *delimiter; + zend_string *enclosure; + zend_string *eol_sequence; + csv_dialect dialect; + zend_object std; +} php_csv_lazy_collection_object; + +static php_csv_lazy_collection_object *php_csv_lazy_collection_object_from(zend_object *object) { + return (php_csv_lazy_collection_object *)(((char*) object) - offsetof(php_csv_lazy_collection_object, std)); +} + +static php_csv_lazy_collection_object *php_csv_lazy_collection_object_fetch(zval *obj) { + return php_csv_lazy_collection_object_from(Z_OBJ_P(obj)); +} + +static zend_object *php_csv_lazy_collection_object_new(zend_class_entry *ce) +{ + php_csv_lazy_collection_object *lazy_collection = zend_object_alloc(sizeof(php_csv_lazy_collection_object), ce); + zend_object_std_init(&lazy_collection->std, ce); + object_properties_init(&lazy_collection->std, ce); + + return &lazy_collection->std; +} + +/* The stream is closed here rather than in free_obj: destructors run while the executor is still + * fully functional, so closing a userspace-wrapper or filtered stream can safely call back into + * userland (stream_close(), onClose()). If destructors are skipped (e.g. after a fatal error), the + * stream is left to the resource list, which closes every stream that is still open. */ +static void php_csv_lazy_collection_object_dtor(zend_object *std) +{ + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_from(std); + + zend_objects_destroy_object(std); + + if (lazy_collection->stream != NULL) { + php_stream *stream = lazy_collection->stream; + lazy_collection->stream = NULL; + php_stream_close(stream); + } +} + +static void php_csv_lazy_collection_object_free(zend_object *std) +{ + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_from(std); + + /* PHP Will create an object even if php_csv_lazy_collection_get_constructor() throws, + * So we cannot assume the buffers and various CSV settings are set */ + if (lazy_collection->buffer != NULL) { + zend_string_release(lazy_collection->buffer); + lazy_collection->buffer = NULL; + } + + /* A stream that is still set was not closed by the destructor; it belongs to the resource + * list then (see php_csv_lazy_collection_object_dtor()) and may already have been freed. */ + lazy_collection->stream = NULL; + smart_str_free(&lazy_collection->stream_buffer); + + if (lazy_collection->delimiter != NULL) { + zend_string_release(lazy_collection->delimiter); + lazy_collection->delimiter = NULL; + } + + if (lazy_collection->enclosure != NULL) { + zend_string_release(lazy_collection->enclosure); + lazy_collection->enclosure = NULL; + } + + if (lazy_collection->eol_sequence != NULL) { + zend_string_release(lazy_collection->eol_sequence); + lazy_collection->eol_sequence = NULL; + } + + zend_object_std_dtor(&lazy_collection->std); +} + +static zend_function *php_csv_lazy_collection_get_constructor(zend_object *obj) { + zend_throw_error(NULL, "Cannot instantiate Csv\\LazyLaxCollection class"); + return NULL; +} + +/** + * Find the end of the first complete row in [row_start, end), honouring the enclosure rules of + * rfc4180_string_to_hashtable(): an EOL sequence inside an escaped (enclosed) field does not + * terminate the row, and a doubled enclosure sequence stays inside the field. + * + * Returns a pointer one past the row's EOL sequence, or NULL when no complete row is available + * yet (i.e. more data must be read from the stream). + * + * Scanning starts at *resume_position in the *resume_in_escaped_field state (the start of the row + * and false for a new row), and the furthest position up to which every decision was final is + * stored back with the state at that point, so that after reading more data the caller resumes + * there and a row is scanned in linear time however many reads it takes. Skipping bytes that + * cannot start a token is always final; a decision on a token depends on at most stable_len bytes, + * so it is final when that many bytes are available. Decisions closer to the end (e.g. a multibyte + * EOL split by a read boundary) are made again once more data is available. + * + * The scanner does not reproduce the parser's errors on malformed input; on such input it may + * over-approximate the row length, and the parser then reports the error. + */ +static const char *csv_find_end_of_row( + const char **resume_position, + bool *resume_in_escaped_field, + const char *end, + const csv_dialect *dialect +) { + const zend_string *delimiter = dialect->delimiter; + const zend_string *enclosure = dialect->enclosure; + const zend_string *eol_sequence = dialect->eol_sequence; + /* Deciding whether an enclosure closes the field reads it, a possible doubled one, and for a + * self-overlapping enclosure the delimiter or EOL sequence following it */ + const size_t stable_len = ZSTR_LEN(enclosure) + + MAX(ZSTR_LEN(enclosure), MAX(ZSTR_LEN(delimiter), ZSTR_LEN(eol_sequence))); + const char *position = *resume_position; + bool in_escaped_field = *resume_in_escaped_field; + bool decisions_are_final = true; + + while (position < end) { + /* Skip the bytes that cannot start a token, in the same way as the parser */ + if (in_escaped_field) { + const char *next = memchr(position, dialect->enclosure_first_byte, end - position); + position = next != NULL ? next : end; + } else { + while (position < end && !dialect->is_special_byte[(unsigned char) *position]) { + position++; + } + } + + if (decisions_are_final) { + *resume_position = position; + *resume_in_escaped_field = in_escaped_field; + decisions_are_final = (size_t) (end - position) >= stable_len; + } + if (position == end) { + break; + } + + if (buffer_starts_with_zend_string(position, end, enclosure)) { + position += ZSTR_LEN(enclosure); + if (in_escaped_field) { + if (buffer_starts_with_zend_string(position, end, enclosure)) { + /* Doubled enclosure: still inside the field */ + position += ZSTR_LEN(enclosure); + } else if (dialect->enclosure_overlaps_itself + && !csv_is_at_end_of_field(position, end, delimiter, eol_sequence)) { + /* Same rule as the parser: the first byte is data */ + position -= ZSTR_LEN(enclosure) - 1; + } else { + in_escaped_field = false; + } + } else { + in_escaped_field = true; + } + continue; + } + + if (!in_escaped_field) { + /* Tokens must be matched in the same order as the parser does: a delimiter is + * consumed as a whole so that an EOL sequence occurring inside it (e.g. + * delimiter "||" with EOL "|") does not split the row. */ + if (buffer_starts_with_zend_string(position, end, delimiter)) { + position += ZSTR_LEN(delimiter); + continue; + } + if (buffer_starts_with_zend_string(position, end, eol_sequence)) { + return position + ZSTR_LEN(eol_sequence); + } + } + + position++; + } + + return NULL; +} + +/* Parse one row from the stream buffer window [start, end) into current_row and advance the + * consumed position. */ +static void php_csv_lazy_collection_parse_buffered_row( + php_csv_lazy_collection_object *lazy_collection, + zval *current_row, + const char *start, + const char *end +) { + const char *position = start; + HashTable *row_ht = rfc4180_string_to_hashtable(&position, end, &lazy_collection->dialect); + if (UNEXPECTED(row_ht == NULL)) { + /* An Error has been thrown; leaving current_row undef ends the iteration */ + return; + } + + lazy_collection->stream_buffer_position += (size_t) (position - start); + ZVAL_ARR(current_row, row_ht); + + /* Compact the buffer so memory stays proportional to the longest row, not the file */ + if (lazy_collection->stream_buffer_position >= PHP_CSV_STREAM_BUFFER_COMPACT_THRESHOLD) { + zend_string *s = lazy_collection->stream_buffer.s; + size_t remaining = ZSTR_LEN(s) - lazy_collection->stream_buffer_position; + memmove(ZSTR_VAL(s), ZSTR_VAL(s) + lazy_collection->stream_buffer_position, remaining); + ZSTR_LEN(s) = remaining; + lazy_collection->stream_buffer_position = 0; + } + + /* The next row is scanned from its start */ + lazy_collection->stream_scan_position = lazy_collection->stream_buffer_position; + lazy_collection->stream_scan_in_escaped_field = false; +} + +/* How many bytes of lookahead past a candidate row end the scanner needs before the row + * boundary can be trusted. Two situations make a boundary decision depend on bytes that may + * not have been read yet: + * - a token occurring inside a longer token (as its prefix, like EOL "\n" in enclosure + * "\nX", or in its interior, like EOL "\n" in delimiter "x\ny"): the shorter token could + * be matched where the longer one was cut by the read boundary. Every decision the + * scanner makes at a position q consumes a token ending at or before the row end, and + * the longer token it could have missed extends at most (len(longer) - offset - + * len(shorter)) bytes past it, so that many bytes of lookahead disambiguate all of them; + * - a closing enclosure right before the EOL: its "doubled enclosure?" check looks + * ZSTR_LEN(enclosure) bytes past the enclosure, which can reach past the EOL. + * For the common single-byte dialects this returns 0, so a complete row buffered from a + * blocking stream (e.g. a FIFO) is delivered without waiting for further data. */ +static size_t csv_scanner_lookahead_needed( + const zend_string *delimiter, + const zend_string *enclosure, + const zend_string *eol_sequence +) { + const zend_string *tokens[3] = {delimiter, enclosure, eol_sequence}; + size_t needed = 0; + + for (int longer = 0; longer < 3; longer++) { + for (int shorter = 0; shorter < 3; shorter++) { + size_t longer_len = ZSTR_LEN(tokens[longer]); + size_t shorter_len = ZSTR_LEN(tokens[shorter]); + if (shorter_len > longer_len) { + continue; + } + for (size_t offset = 0; offset + shorter_len <= longer_len; offset++) { + if (longer == shorter && offset == 0) { + continue; + } + if (memcmp(ZSTR_VAL(tokens[longer]) + offset, ZSTR_VAL(tokens[shorter]), shorter_len) == 0) { + needed = MAX(needed, longer_len - offset - shorter_len); + } + } + } + } + + if (ZSTR_LEN(enclosure) > ZSTR_LEN(eol_sequence)) { + needed = MAX(needed, ZSTR_LEN(enclosure) - ZSTR_LEN(eol_sequence)); + } + + return needed; +} + +static void php_csv_lazy_collection_stream_next(php_csv_lazy_collection_object *lazy_collection, zval *current_row) +{ + smart_str *stream_buffer = &lazy_collection->stream_buffer; + const size_t lookahead_needed = csv_scanner_lookahead_needed( + lazy_collection->delimiter, lazy_collection->enclosure, lazy_collection->eol_sequence); + + for (;;) { + const char *start = NULL; + const char *end = NULL; + if (stream_buffer->s != NULL) { + start = ZSTR_VAL(stream_buffer->s) + lazy_collection->stream_buffer_position; + end = ZSTR_VAL(stream_buffer->s) + ZSTR_LEN(stream_buffer->s); + } + bool has_buffered_data = start != NULL && start < end; + + const char *end_of_row = NULL; + if (has_buffered_data) { + const char *resume_position = ZSTR_VAL(stream_buffer->s) + lazy_collection->stream_scan_position; + end_of_row = csv_find_end_of_row(&resume_position, &lazy_collection->stream_scan_in_escaped_field, + end, &lazy_collection->dialect); + lazy_collection->stream_scan_position = (size_t) (resume_position - ZSTR_VAL(stream_buffer->s)); + } + bool at_eof = php_stream_eof(lazy_collection->stream); + + /* Only trust a row boundary when the dialect-specific lookahead is available past + * it, or when no further bytes can arrive; re-reading and re-scanning resolves + * boundary ambiguities (see csv_scanner_lookahead_needed()). */ + if (end_of_row != NULL && (at_eof || (size_t) (end - end_of_row) >= lookahead_needed)) { + php_csv_lazy_collection_parse_buffered_row(lazy_collection, current_row, start, end_of_row); + return; + } + + if (at_eof) { + if (!has_buffered_data) { + /* End of iteration; current_row stays undef */ + return; + } + /* Final row without a terminating EOL sequence: hand the whole remainder to the + * parser, mirroring the end-of-buffer behaviour of createFromBuffer(). */ + php_csv_lazy_collection_parse_buffered_row(lazy_collection, current_row, start, end); + return; + } + + /* Read through the stream's read buffer the way php_stream_get_line() does: + * php_stream_fill_read_buffer() issues a single read against the underlying + * stream, so a partial read (e.g. from a FIFO whose writer keeps it open) is + * processed as soon as it arrives. php_stream_read() must not be used here, as + * it keeps reading plain-file streams until the full requested size is filled. + * A stream with read filters attached can still block across multiple underlying + * reads inside the fill; that matches what php_stream_get_line() based readers + * (e.g. fgetcsv()) do on such streams. */ + php_stream *stream = lazy_collection->stream; + if (stream->writepos <= stream->readpos + && UNEXPECTED(php_stream_fill_read_buffer(stream, PHP_CSV_STREAM_CHUNK_SIZE) != SUCCESS)) { + zend_throw_error(NULL, "Failed to read from the CSV file"); + return; + } + if (stream->writepos > stream->readpos) { + size_t available = (size_t) (stream->writepos - stream->readpos); + smart_str_appendl(stream_buffer, (const char *) stream->readbuf + stream->readpos, available); + stream->readpos += available; + stream->position += available; + continue; + } + /* The buffer could not be filled although the EOF flag was not set at the start + * of this iteration: no more data can be obtained, so treat the stream exactly + * like EOF instead of spinning on it. */ + if (end_of_row != NULL) { + php_csv_lazy_collection_parse_buffered_row(lazy_collection, current_row, start, end_of_row); + return; + } + if (!has_buffered_data) { + return; + } + php_csv_lazy_collection_parse_buffered_row(lazy_collection, current_row, start, end); + return; + } +} + +/** + * Csv\LazyLaxCollection internal iterator + * Copied from zend_test/iterator.c + * + * The iteration state lives in the iterator, so that nested loops over the same collection are + * independent. In stream mode the position in the stream is shared, so get_iterator() allows a + * single active iterator. + */ +typedef struct php_csv_lazy_collection_it { + zend_object_iterator intern; + /* Buffer mode: where the next row starts in the collection's buffer */ + const char *buffer_position; + /* Cannot use a HashTable as we need to be able to return a zval for current() */ + zval current_row; +} php_csv_lazy_collection_it; + +static php_csv_lazy_collection_it *php_csv_lazy_collection_it_fetch(zend_object_iterator *obj_iter) { + return (php_csv_lazy_collection_it *)obj_iter; +} + +static void php_csv_lazy_collection_it_dtor(zend_object_iterator *obj_iter) { + php_csv_lazy_collection_it *iterator = php_csv_lazy_collection_it_fetch(obj_iter); + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_fetch(&iterator->intern.data); + + zval_ptr_dtor(&iterator->current_row); + if (lazy_collection->is_stream_mode) { + lazy_collection->stream_has_active_iterator = false; + } + zval_ptr_dtor(&iterator->intern.data); +} + +static void php_csv_lazy_collection_it_next(zend_object_iterator *obj_iter) { + php_csv_lazy_collection_it *iterator = php_csv_lazy_collection_it_fetch(obj_iter); + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_fetch(&iterator->intern.data); + + zval_ptr_dtor(&iterator->current_row); + ZVAL_UNDEF(&iterator->current_row); + + if (lazy_collection->is_stream_mode) { + /* The stream is gone once the collection has been destroyed (e.g. at shutdown) */ + if (lazy_collection->stream != NULL) { + php_csv_lazy_collection_stream_next(lazy_collection, &iterator->current_row); + } + return; + } + + /* Do not move past eof */ + const char *end_of_buffer = ZSTR_VAL(lazy_collection->buffer) + ZSTR_LEN(lazy_collection->buffer); + if (iterator->buffer_position == end_of_buffer) { + return; + } + + HashTable *row_ht = rfc4180_string_to_hashtable(&iterator->buffer_position, end_of_buffer, &lazy_collection->dialect); + if (UNEXPECTED(row_ht == NULL)) { + return; + } + + ZVAL_ARR(&iterator->current_row, row_ht); +} + +static void php_csv_lazy_collection_it_rewind(zend_object_iterator *obj_iter) { + php_csv_lazy_collection_it *iterator = php_csv_lazy_collection_it_fetch(obj_iter); + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_fetch(&iterator->intern.data); + zval_ptr_dtor(&iterator->current_row); + ZVAL_UNDEF(&iterator->current_row); + + if (lazy_collection->is_stream_mode) { + if (lazy_collection->stream_iteration_started && lazy_collection->stream != NULL) { + if (UNEXPECTED(php_stream_rewind(lazy_collection->stream) != 0)) { + zend_throw_error(NULL, "Cannot rewind the CSV file stream"); + return; + } + smart_str_free(&lazy_collection->stream_buffer); + lazy_collection->stream_buffer_position = 0; + lazy_collection->stream_scan_position = 0; + lazy_collection->stream_scan_in_escaped_field = false; + } + lazy_collection->stream_iteration_started = true; + } else { + iterator->buffer_position = ZSTR_VAL(lazy_collection->buffer); + } + + /* Fetch first row as this is what is expected */ + php_csv_lazy_collection_it_next(obj_iter); +} + +static zend_result php_csv_lazy_collection_it_valid(zend_object_iterator *obj_iter) { + php_csv_lazy_collection_it *iterator = php_csv_lazy_collection_it_fetch(obj_iter); + + /* Upon reaching EOF the current row is freed and set to undef */ + return Z_ISUNDEF(iterator->current_row) ? FAILURE : SUCCESS; +} + +static zval *php_csv_lazy_collection_it_current(zend_object_iterator *obj_iter) { + php_csv_lazy_collection_it *iterator = php_csv_lazy_collection_it_fetch(obj_iter); + + if (UNEXPECTED(Z_ISUNDEF(iterator->current_row))) { + return NULL; + } + return &iterator->current_row; +} + +static const zend_object_iterator_funcs php_csv_lazy_collection_it_vtable = { + php_csv_lazy_collection_it_dtor, + php_csv_lazy_collection_it_valid, + php_csv_lazy_collection_it_current, + NULL, // get_current_key + php_csv_lazy_collection_it_next, + php_csv_lazy_collection_it_rewind, + NULL, // invalidate_current + NULL, // get_gc +}; + +static zend_object_iterator *php_csv_lazy_collection_get_iterator( + zend_class_entry *ce, + zval *object, + int by_ref +) { + if (by_ref) { + zend_throw_error(NULL, "An iterator cannot be used with foreach by reference"); + return NULL; + } + + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_fetch(object); + if (lazy_collection->is_stream_mode) { + if (UNEXPECTED(lazy_collection->stream_has_active_iterator)) { + zend_throw_error(NULL, "A Csv\\LazyLaxCollection created from a file cannot be iterated by more than one loop at a time"); + return NULL; + } + lazy_collection->stream_has_active_iterator = true; + } + + php_csv_lazy_collection_it *iterator = emalloc(sizeof(php_csv_lazy_collection_it)); + zend_iterator_init((zend_object_iterator*)iterator); + + ZVAL_OBJ_COPY(&iterator->intern.data, Z_OBJ_P(object)); + iterator->intern.funcs = &php_csv_lazy_collection_it_vtable; + iterator->buffer_position = NULL; + ZVAL_UNDEF(&iterator->current_row); + + return (zend_object_iterator*)iterator; +} + +PHP_METHOD(Csv_LazyLaxCollection, __construct) { + ZEND_PARSE_PARAMETERS_NONE(); + + zend_throw_error(NULL, "Cannot manually instantiate Csv\\LazyLaxCollection"); +} + +PHP_METHOD(Csv_LazyLaxCollection, getIterator) { + ZEND_PARSE_PARAMETERS_NONE(); + + zend_create_internal_iterator_zval(return_value, ZEND_THIS); +} + +/* See static void buffer_to_collection_generic */ +PHP_METHOD(Csv_LazyLaxCollection, createFromBuffer) { + zend_string *delimiter = NULL; + zend_string *enclosure = NULL; + zend_string *eol_sequence = NULL; + zend_string *buffer; + + if (zend_parse_parameters(ZEND_NUM_ARGS(), "S|SSS", &buffer, &delimiter, &enclosure, &eol_sequence) == FAILURE) { + RETURN_THROWS(); + } + + EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(2, 3, 4); + + zend_string_addref(buffer); + + object_init_ex(return_value, php_csv_lazy_collection_ce); + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_fetch(return_value); + + /* We have copies from the macro */ + lazy_collection->eol_sequence = eol_sequence; + lazy_collection->enclosure = enclosure; + lazy_collection->delimiter = delimiter; + csv_dialect_init(&lazy_collection->dialect, delimiter, enclosure, eol_sequence); + lazy_collection->buffer = buffer; +} + +PHP_METHOD(Csv_LazyLaxCollection, createFromFile) { + zend_string *file = NULL; + zend_string *delimiter = NULL; + zend_string *enclosure = NULL; + zend_string *eol_sequence = NULL; + + ZEND_PARSE_PARAMETERS_START(1, 4) + Z_PARAM_PATH_STR(file) + Z_PARAM_OPTIONAL + Z_PARAM_STR(delimiter) + Z_PARAM_STR(enclosure) + Z_PARAM_STR(eol_sequence) + ZEND_PARSE_PARAMETERS_END(); + + EOL_SEQUENCE_AND_DELIMITER_AND_ENCLOSURE_CHECKS(2, 3, 4); + + php_stream *stream = php_csv_stream_open(file, "rb", "reading"); + if (UNEXPECTED(stream == NULL)) { + zend_string_release(eol_sequence); + zend_string_release(delimiter); + zend_string_release(enclosure); + RETURN_THROWS(); + } + + object_init_ex(return_value, php_csv_lazy_collection_ce); + php_csv_lazy_collection_object *lazy_collection = php_csv_lazy_collection_object_fetch(return_value); + + /* We have copies from the macro */ + lazy_collection->eol_sequence = eol_sequence; + lazy_collection->enclosure = enclosure; + lazy_collection->delimiter = delimiter; + csv_dialect_init(&lazy_collection->dialect, delimiter, enclosure, eol_sequence); + lazy_collection->is_stream_mode = true; + lazy_collection->stream = stream; +} + +PHP_MINIT_FUNCTION(csv) +{ + php_csv_lazy_collection_ce = register_class_Csv_LazyLaxCollection(zend_ce_aggregate); + php_csv_lazy_collection_ce->create_object = php_csv_lazy_collection_object_new; + php_csv_lazy_collection_ce->get_iterator = php_csv_lazy_collection_get_iterator; + php_csv_lazy_collection_ce->default_object_handlers = &php_csv_lazy_collection_object_handlers; + + memcpy(&php_csv_lazy_collection_object_handlers, &std_object_handlers, sizeof(zend_object_handlers)); + php_csv_lazy_collection_object_handlers.offset = offsetof(php_csv_lazy_collection_object, std); + php_csv_lazy_collection_object_handlers.dtor_obj = php_csv_lazy_collection_object_dtor; + php_csv_lazy_collection_object_handlers.free_obj = php_csv_lazy_collection_object_free; + php_csv_lazy_collection_object_handlers.get_constructor = php_csv_lazy_collection_get_constructor; + php_csv_lazy_collection_object_handlers.compare = zend_objects_not_comparable; + php_csv_lazy_collection_object_handlers.clone_obj = NULL; + + return SUCCESS; +} + +zend_module_entry csv_module_entry = { + STANDARD_MODULE_HEADER, + "csv", /* Extension name */ + ext_functions, /* zend_function_entry */ + PHP_MINIT(csv), /* PHP_MINIT - Module initialization */ + NULL, /* PHP_MSHUTDOWN - Module shutdown */ + NULL, /* PHP_RINIT - Request initialization */ + NULL, /* PHP_RSHUTDOWN - Request shutdown */ + PHP_MINFO(csv), /* PHP_MINFO - Module info */ + PHP_VERSION, /* Version */ + STANDARD_MODULE_PROPERTIES +}; + +#ifdef COMPILE_DL_CSV +# ifdef ZTS +ZEND_TSRMLS_CACHE_DEFINE() +# endif +ZEND_GET_MODULE(csv) +#endif diff --git a/ext/csv/csv.stub.php b/ext/csv/csv.stub.php new file mode 100644 index 000000000000..849d76af8133 --- /dev/null +++ b/ext/csv/csv.stub.php @@ -0,0 +1,42 @@ +. | + | | + | SPDX-License-Identifier: BSD-3-Clause | + +----------------------------------------------------------------------+ + | Authors: Gina Peter Banyard | + | Damian Jóźwiak | + +----------------------------------------------------------------------+ +*/ +/* csv extension for PHP */ + +#ifndef PHP_CSV_H +# define PHP_CSV_H + +extern zend_module_entry csv_module_entry; +# define phpext_csv_ptr &csv_module_entry + +# if defined(ZTS) && defined(COMPILE_DL_CSV) +ZEND_TSRMLS_CACHE_EXTERN() +# endif + +#endif /* PHP_CSV_H */ diff --git a/ext/csv/tests/001_load-csv-extension-check.phpt b/ext/csv/tests/001_load-csv-extension-check.phpt new file mode 100644 index 000000000000..9ef43e75dde9 --- /dev/null +++ b/ext/csv/tests/001_load-csv-extension-check.phpt @@ -0,0 +1,14 @@ +--TEST-- +Check if csv is loaded +--SKIPIF-- + +--FILE-- + +--EXPECT-- +The extension "csv" is available diff --git a/ext/csv/tests/LazyCollection/deny_instantiation.phpt b/ext/csv/tests/LazyCollection/deny_instantiation.phpt new file mode 100644 index 000000000000..3e40824a341a --- /dev/null +++ b/ext/csv/tests/LazyCollection/deny_instantiation.phpt @@ -0,0 +1,20 @@ +--TEST-- +Csv\LazyLaxCollection must not be instantiable +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +Error: Cannot instantiate Csv\LazyLaxCollection class diff --git a/ext/csv/tests/LazyCollection/deny_instantiation_via_reflection.phpt b/ext/csv/tests/LazyCollection/deny_instantiation_via_reflection.phpt new file mode 100644 index 000000000000..d4984b6b10cc --- /dev/null +++ b/ext/csv/tests/LazyCollection/deny_instantiation_via_reflection.phpt @@ -0,0 +1,44 @@ +--TEST-- +Csv\LazyLaxCollection must not be instantiable +--SKIPIF-- + +--FILE-- +newInstance(); + var_dump($o); +} catch (\Throwable $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +try { + $o = $rc->newInstanceArgs(); + var_dump($o); +} catch (\Throwable $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +try { + $o = $rc->newInstanceWithoutConstructor(); + var_dump($o); +} catch (\Throwable $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +object(ReflectionClass)#1 (1) { + ["name"]=> + string(21) "Csv\LazyLaxCollection" +} +Error: Cannot instantiate Csv\LazyLaxCollection class +Error: Cannot instantiate Csv\LazyLaxCollection class +ReflectionException: Class Csv\LazyLaxCollection is an internal class marked as final that cannot be instantiated without invoking its constructor diff --git a/ext/csv/tests/LazyCollection/fromBuffer/basic.phpt b/ext/csv/tests/LazyCollection/fromBuffer/basic.phpt new file mode 100644 index 000000000000..9b86897d77ab --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromBuffer/basic.phpt @@ -0,0 +1,51 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromBuffer() with standard parameters +--SKIPIF-- + +--FILE-- + $row) { + var_dump($nb, $row); +} + +?> +--EXPECT-- +object(Csv\LazyLaxCollection)#1 (0) { +} +int(0) +array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" +} +int(1) +array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" +} +int(2) +array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" +} diff --git a/ext/csv/tests/LazyCollection/fromBuffer/lax_varying_field_count.phpt b/ext/csv/tests/LazyCollection/fromBuffer/lax_varying_field_count.phpt new file mode 100644 index 000000000000..7779b2139df2 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromBuffer/lax_varying_field_count.phpt @@ -0,0 +1,32 @@ +--TEST-- +Csv\LazyLaxCollection::createFromBuffer() is lax: rows may have a varying number of fields +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +bool(true) +ValueError: Buffer row 2 contains 1 fields compared to 3 fields on previous rows diff --git a/ext/csv/tests/LazyCollection/fromBuffer/nested_iteration.phpt b/ext/csv/tests/LazyCollection/fromBuffer/nested_iteration.phpt new file mode 100644 index 000000000000..5b6a92abd6ef --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromBuffer/nested_iteration.phpt @@ -0,0 +1,26 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromBuffer(): nested loops over the same collection are independent +--EXTENSIONS-- +csv +--FILE-- +getIterator(); +$it2 = $collection->getIterator(); +$it1->rewind(); +$it2->rewind(); +$it1->next(); +echo json_encode([$it1->current(), $it2->current()]), \PHP_EOL; +?> +--EXPECT-- +aa ab ac ba bb bc ca cb cc +[["b"],["a"]] diff --git a/ext/csv/tests/LazyCollection/fromBuffer/non_escaped_enclosure.phpt b/ext/csv/tests/LazyCollection/fromBuffer/non_escaped_enclosure.phpt new file mode 100644 index 000000000000..7907a74ab346 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromBuffer/non_escaped_enclosure.phpt @@ -0,0 +1,56 @@ +--TEST-- +Csv\LazyLaxCollection::createFromBuffer() non escaped enclosure in one row +--SKIPIF-- + +--FILE-- + $row) { + var_dump($nb, $row); + } +} catch (\Error $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +$string = "One,Two,Three\r\nFour,Deあux,Six\r\nSeven,Eight,Nine\r\n"; +$it = Csv\LazyLaxCollection::createFromBuffer($string, ',', 'あ'); + +try { + foreach ($it as $nb => $row) { + var_dump($nb, $row); + } +} catch (\Error $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +int(0) +array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" +} +ValueError: Enclosure sequence is used in a non escaped field +int(0) +array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" +} +ValueError: Enclosure sequence is used in a non escaped field diff --git a/ext/csv/tests/LazyCollection/fromBuffer/nul_byte_custom_eol_sequence.phpt b/ext/csv/tests/LazyCollection/fromBuffer/nul_byte_custom_eol_sequence.phpt new file mode 100644 index 000000000000..0517368246c4 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromBuffer/nul_byte_custom_eol_sequence.phpt @@ -0,0 +1,42 @@ +--TEST-- +Csv\LazyLaxCollection::createFromBuffer() with custom EOL sequence as a nul byte +--SKIPIF-- + +--FILE-- + $row) { + var_dump($row === $collection[$nb]); +} + +?> +--EXPECT-- +bool(true) +bool(true) +bool(true) diff --git a/ext/csv/tests/LazyCollection/fromBuffer/with_custom_eol_sequence.phpt b/ext/csv/tests/LazyCollection/fromBuffer/with_custom_eol_sequence.phpt new file mode 100644 index 000000000000..22918c645716 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromBuffer/with_custom_eol_sequence.phpt @@ -0,0 +1,42 @@ +--TEST-- +Csv\LazyLaxCollection::createFromBuffer() with standard parameters +--SKIPIF-- + +--FILE-- + $row) { + var_dump($row === $collection[$nb]); +} + +?> +--EXPECT-- +bool(true) +bool(true) +bool(true) diff --git a/ext/csv/tests/LazyCollection/fromBuffer/with_new_lines.phpt b/ext/csv/tests/LazyCollection/fromBuffer/with_new_lines.phpt new file mode 100644 index 000000000000..6de63bf89ae1 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromBuffer/with_new_lines.phpt @@ -0,0 +1,42 @@ +--TEST-- +Csv\LazyLaxCollection::createFromBuffer() with new lines in buffer +--SKIPIF-- + +--FILE-- + $row) { + var_dump($row === $collection[$nb]); +} + +?> +--EXPECT-- +bool(true) +bool(true) +bool(true) diff --git a/ext/csv/tests/LazyCollection/fromFile/basic.phpt b/ext/csv/tests/LazyCollection/fromFile/basic.phpt new file mode 100644 index 000000000000..c60b6b4d97b7 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/basic.phpt @@ -0,0 +1,24 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): basic behaviour +--EXTENSIONS-- +csv +--FILE-- + $row) { + echo $index, ': ', json_encode($row), \PHP_EOL; +} +?> +--CLEAN-- + +--EXPECT-- +bool(true) +0: ["Hello","World"] +1: ["with \"quotes\"","and\r\nan embedded EOL"] +2: ["last","row"] diff --git a/ext/csv/tests/LazyCollection/fromFile/chunk_boundary.phpt b/ext/csv/tests/LazyCollection/fromFile/chunk_boundary.phpt new file mode 100644 index 000000000000..fea9875ded74 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/chunk_boundary.phpt @@ -0,0 +1,39 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): rows spanning read-chunk boundaries match buffer parsing +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +int(10001) +bool(true) +bool(true) +["field9999","value,with,commas9999","text\r\nline9999"] diff --git a/ext/csv/tests/LazyCollection/fromFile/custom_delimiter_eol.phpt b/ext/csv/tests/LazyCollection/fromFile/custom_delimiter_eol.phpt new file mode 100644 index 000000000000..f356e8a2f442 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/custom_delimiter_eol.phpt @@ -0,0 +1,30 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): custom delimiter and multibyte EOL sequence +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +array(2) { + [0]=> + string(1) "a" + [1]=> + string(1) "b" +} +array(2) { + [0]=> + string(3) "c;d" + [1]=> + string(1) "e" +} diff --git a/ext/csv/tests/LazyCollection/fromFile/default_context.phpt b/ext/csv/tests/LazyCollection/fromFile/default_context.phpt new file mode 100644 index 000000000000..02bad6ebca94 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/default_context.phpt @@ -0,0 +1,36 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile() and Csv\collection_to_file(): the default stream context is used +--EXTENSIONS-- +csv +--FILE-- +context)['csvtest'] ?? null), \PHP_EOL; + return true; + } + public function stream_read($count) { + $chunk = substr(self::$data, $this->pos, $count); + $this->pos += strlen($chunk); + return $chunk; + } + public function stream_write($data) { self::$data .= $data; return strlen($data); } + public function stream_flush() { return true; } + public function stream_eof() { return $this->pos >= strlen(self::$data); } + public function stream_stat() { return []; } +} +stream_wrapper_register('csvtest', Wrapper::class); +stream_context_set_default(['csvtest' => ['option' => 'value']]); + +Csv\collection_to_file('csvtest://data', [['a', 'b']]); +foreach (Csv\LazyLaxCollection::createFromFile('csvtest://data') as $row) { + echo json_encode($row), \PHP_EOL; +} +?> +--EXPECT-- +wb: {"option":"value"} +rb: {"option":"value"} +["a","b"] diff --git a/ext/csv/tests/LazyCollection/fromFile/empty_file.phpt b/ext/csv/tests/LazyCollection/fromFile/empty_file.phpt new file mode 100644 index 000000000000..1ac6ac8d945b --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/empty_file.phpt @@ -0,0 +1,21 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): empty file yields no rows +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +int(0) diff --git a/ext/csv/tests/LazyCollection/fromFile/enclosure_prefix_chunk_boundary.phpt b/ext/csv/tests/LazyCollection/fromFile/enclosure_prefix_chunk_boundary.phpt new file mode 100644 index 000000000000..d6e84d11e752 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/enclosure_prefix_chunk_boundary.phpt @@ -0,0 +1,26 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): multibyte enclosure split across the read boundary +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +int(1) +bool(true) +string(1) "b" diff --git a/ext/csv/tests/LazyCollection/fromFile/eol_inside_delimiter_chunk_boundary.phpt b/ext/csv/tests/LazyCollection/fromFile/eol_inside_delimiter_chunk_boundary.phpt new file mode 100644 index 000000000000..6afd868df336 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/eol_inside_delimiter_chunk_boundary.phpt @@ -0,0 +1,28 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): EOL occurring inside a delimiter split at the read boundary +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +int(1) +int(2) +bool(true) +string(1) "b" diff --git a/ext/csv/tests/LazyCollection/fromFile/large_enclosed_field.phpt b/ext/csv/tests/LazyCollection/fromFile/large_enclosed_field.phpt new file mode 100644 index 000000000000..6cc4a511b1c3 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/large_enclosed_field.phpt @@ -0,0 +1,48 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): fields and errors spanning many reads +--DESCRIPTION-- +The row boundary scanner resumes where it stopped after each read instead of rescanning the +row from its start, which made a multi-megabyte enclosed field take seconds. +--EXTENSIONS-- +csv +--FILE-- +getMessage(), \PHP_EOL; +} +?> +--CLEAN-- + +--EXPECT-- +int(2) +bool(true) +string(1) "b" +array(2) { + [0]=> + string(1) "c" + [1]=> + string(1) "d" +} +bool(true) +ValueError: Enclosure sequence is not closed diff --git a/ext/csv/tests/LazyCollection/fromFile/nested_iteration.phpt b/ext/csv/tests/LazyCollection/fromFile/nested_iteration.phpt new file mode 100644 index 000000000000..d75d83cdcc48 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/nested_iteration.phpt @@ -0,0 +1,37 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): only one loop at a time can read the stream +--EXTENSIONS-- +csv +--FILE-- +getMessage(), \PHP_EOL; + } + echo json_encode($outer), \PHP_EOL; +} + +/* Once the first loop is over, the collection can be iterated again */ +foreach ($collection as $row) { + echo json_encode($row), \PHP_EOL; +} +?> +--CLEAN-- + +--EXPECT-- +Error: A Csv\LazyLaxCollection created from a file cannot be iterated by more than one loop at a time +["a"] +Error: A Csv\LazyLaxCollection created from a file cannot be iterated by more than one loop at a time +["b"] +["a"] +["b"] diff --git a/ext/csv/tests/LazyCollection/fromFile/no_trailing_eol.phpt b/ext/csv/tests/LazyCollection/fromFile/no_trailing_eol.phpt new file mode 100644 index 000000000000..d3a71b3ca453 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/no_trailing_eol.phpt @@ -0,0 +1,30 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): final row without a trailing EOL sequence +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +array(2) { + [0]=> + string(1) "a" + [1]=> + string(1) "b" +} +array(2) { + [0]=> + string(1) "c" + [1]=> + string(1) "d" +} diff --git a/ext/csv/tests/LazyCollection/fromFile/nonexistent_file.phpt b/ext/csv/tests/LazyCollection/fromFile/nonexistent_file.phpt new file mode 100644 index 000000000000..4d1c6811243e --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/nonexistent_file.phpt @@ -0,0 +1,14 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): nonexistent file throws +--EXTENSIONS-- +csv +--FILE-- +getMessage(), \PHP_EOL; +} +?> +--EXPECTF-- +Error: Csv\LazyLaxCollection::createFromFile(): Failed to open stream: No such file or directory diff --git a/ext/csv/tests/LazyCollection/fromFile/retained_resource_reference.phpt b/ext/csv/tests/LazyCollection/fromFile/retained_resource_reference.phpt new file mode 100644 index 000000000000..a9d12c7f47bb --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/retained_resource_reference.phpt @@ -0,0 +1,65 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): a resource reference retained during opening stays safe +--EXTENSIONS-- +csv +--FILE-- +datalen; + stream_bucket_append($out, $bucket); + } + return PSFS_PASS_ON; + } +} +stream_filter_register('grabbing', GrabbingFilter::class); + +$file = __DIR__ . '/retained_resource_reference.csv'; +file_put_contents($file, "a,b\r\nc,d\r\n"); + +$before = array_map('intval', get_resources('stream')); +$collection = Csv\LazyLaxCollection::createFromFile('php://filter/read=grabbing/resource=' . $file); +$internal = array_values(array_filter($GLOBALS['grabbed'], fn($res) => !in_array((int) $res, $before, true))); +unset($GLOBALS['grabbed']); +var_dump(count($internal)); +$res = $internal[0]; +unset($internal); + +/* The captured reference is the live internal stream, which userland cannot close */ +var_dump(gettype($res)); +var_dump(fclose($res)); +foreach ($collection as $row) { + echo json_encode($row), \PHP_EOL; +} +/* Destroying the collection closes the stream; the reference sees a closed resource */ +unset($collection); +var_dump(gettype($res)); +$id = (int) $res; +var_dump(in_array($id, array_map('intval', get_resources('Unknown')), true)); +/* Releasing the last reference frees the resource */ +unset($res); +var_dump(in_array($id, array_map('intval', get_resources()), true)); +echo "done", \PHP_EOL; +?> +--CLEAN-- + +--EXPECTF-- +int(1) +string(8) "resource" + +Warning: fclose(): cannot close the provided stream, as it must not be manually closed in %s on line %d +bool(false) +["a","b"] +["c","d"] +string(17) "resource (closed)" +bool(true) +bool(false) +done diff --git a/ext/csv/tests/LazyCollection/fromFile/retained_resource_reference_release_order.phpt b/ext/csv/tests/LazyCollection/fromFile/retained_resource_reference_release_order.phpt new file mode 100644 index 000000000000..1ebbf5828b68 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/retained_resource_reference_release_order.phpt @@ -0,0 +1,140 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): several retained references to the internal stream's resource, released in different orders +--DESCRIPTION-- +Userland can retain references to the resource of the internal stream (here captured while the +stream is being opened). The stream stays open while the collection is alive, becomes a closed +resource once the collection is destroyed, and the resource is freed when its last reference +is released, whatever the order of releases. +--EXTENSIONS-- +csv +--FILE-- +datalen; + stream_bucket_append($out, $bucket); + } + return PSFS_PASS_ON; + } +} +stream_filter_register('grabbing', GrabbingFilter::class); + +$file = __DIR__ . '/retained_resource_reference_release_order.csv'; +file_put_contents($file, "a,b\r\nc,d\r\n"); + +function open_collection(string $file): array { + $before = array_map('intval', get_resources('stream')); + $collection = Csv\LazyLaxCollection::createFromFile('php://filter/read=grabbing/resource=' . $file); + $internal = array_values(array_filter(GrabbingFilter::$grabbed, fn($res) => !in_array((int) $res, $before, true))); + GrabbingFilter::$grabbed = []; + var_dump(count($internal)); + return [$collection, $internal[0]]; +} + +/* "stream" while open, "closed" once closed but still referenced, "freed" afterwards */ +function state(int $id): string { + if (in_array($id, array_map('intval', get_resources('stream')), true)) { + return 'stream'; + } + if (in_array($id, array_map('intval', get_resources('Unknown')), true)) { + return 'closed'; + } + return in_array($id, array_map('intval', get_resources()), true) ? 'other' : 'freed'; +} + +function iterate(Csv\LazyLaxCollection $collection): void { + foreach ($collection as $row) { + echo json_encode($row), \PHP_EOL; + } +} + +echo "--- Release one reference, destroy the collection, then release the rest\n"; +[$collection, $a] = open_collection($file); +$b = $a; +$holder = new stdClass(); +$holder->res = $a; +$id = (int) $a; +echo state($id), \PHP_EOL; +/* get_resources() hands out further references; dropping them must not free it early */ +$again = get_resources('stream'); +unset($again); +echo state($id), \PHP_EOL; +iterate($collection); +unset($b); +echo state($id), \PHP_EOL; +unset($collection); +var_dump(gettype($a), gettype($holder->res)); +echo state($id), \PHP_EOL; +unset($holder); +echo state($id), \PHP_EOL; +unset($a); +echo state($id), \PHP_EOL; + +echo "--- Release every reference before the collection is used\n"; +[$collection, $a] = open_collection($file); +$b = $a; +$id = (int) $a; +unset($a, $b); +echo state($id), \PHP_EOL; +iterate($collection); +iterate($collection); +unset($collection); +echo state($id), \PHP_EOL; + +echo "--- Destroy the collection first, then release references in reverse order\n"; +[$collection, $a] = open_collection($file); +$refs = [$a, $a, $a]; +$id = (int) $a; +iterate($collection); +unset($collection, $a); +echo state($id), \PHP_EOL; +array_pop($refs); +array_pop($refs); +var_dump(gettype($refs[0])); +echo state($id), \PHP_EOL; +array_pop($refs); +echo state($id), \PHP_EOL; + +echo "done\n"; +?> +--CLEAN-- + +--EXPECT-- +--- Release one reference, destroy the collection, then release the rest +int(1) +stream +stream +["a","b"] +["c","d"] +stream +string(17) "resource (closed)" +string(17) "resource (closed)" +closed +closed +freed +--- Release every reference before the collection is used +int(1) +stream +["a","b"] +["c","d"] +["a","b"] +["c","d"] +freed +--- Destroy the collection first, then release references in reverse order +int(1) +["a","b"] +["c","d"] +closed +string(17) "resource (closed)" +closed +freed +done diff --git a/ext/csv/tests/LazyCollection/fromFile/rewind_iterates_twice.phpt b/ext/csv/tests/LazyCollection/fromFile/rewind_iterates_twice.phpt new file mode 100644 index 000000000000..c4c99c18e563 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/rewind_iterates_twice.phpt @@ -0,0 +1,52 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): iterating twice, including after an early break +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +array(2) { + [0]=> + string(1) "a" + [1]=> + string(1) "b" +} +--- second run --- +array(2) { + [0]=> + string(1) "a" + [1]=> + string(1) "b" +} +array(2) { + [0]=> + string(1) "c" + [1]=> + string(1) "d" +} +array(2) { + [0]=> + string(1) "e" + [1]=> + string(1) "f" +} diff --git a/ext/csv/tests/LazyCollection/fromFile/scanner_delimiter_eol_overlap.phpt b/ext/csv/tests/LazyCollection/fromFile/scanner_delimiter_eol_overlap.phpt new file mode 100644 index 000000000000..7d57b13d0ee2 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/scanner_delimiter_eol_overlap.phpt @@ -0,0 +1,24 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): EOL being a prefix of the delimiter must not split rows +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +[["a","b"]] +bool(true) diff --git a/ext/csv/tests/LazyCollection/fromFile/stream_cannot_be_closed.phpt b/ext/csv/tests/LazyCollection/fromFile/stream_cannot_be_closed.phpt new file mode 100644 index 000000000000..339801705370 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/stream_cannot_be_closed.phpt @@ -0,0 +1,40 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): userland cannot close the internal stream +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECTF-- +Warning: fclose(): cannot close the provided stream, as it must not be manually closed in %s on line %d +bool(false) +["a","b"] +["c","d"] +string(17) "resource (closed)" +done diff --git a/ext/csv/tests/LazyCollection/fromFile/stream_closed_at_shutdown.phpt b/ext/csv/tests/LazyCollection/fromFile/stream_closed_at_shutdown.phpt new file mode 100644 index 000000000000..6360cc396317 --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/stream_closed_at_shutdown.phpt @@ -0,0 +1,37 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): the stream of a collection alive at shutdown is closed while userland can still run +--EXTENSIONS-- +csv +--FILE-- += strlen(self::$data); } + public function stream_close() { echo "stream_close() called\n"; } + public function stream_stat() { return []; } +} +stream_wrapper_register('csvtest', Wrapper::class); + +/* A cycle keeps the collection alive until shutdown */ +class Holder { public $collection; public $self; } +$holder = new Holder; +$holder->self = $holder; +$holder->collection = Csv\LazyLaxCollection::createFromFile('csvtest://data'); +foreach ($holder->collection as $row) { + echo json_encode($row), \PHP_EOL; +} +echo "end of script\n"; +?> +--EXPECT-- +["a","b"] +["c","d"] +end of script +stream_close() called diff --git a/ext/csv/tests/LazyCollection/fromFile/zero_lookahead_single_read.phpt b/ext/csv/tests/LazyCollection/fromFile/zero_lookahead_single_read.phpt new file mode 100644 index 000000000000..3b537cd499db --- /dev/null +++ b/ext/csv/tests/LazyCollection/fromFile/zero_lookahead_single_read.phpt @@ -0,0 +1,35 @@ +--TEST-- +Test Csv\LazyLaxCollection::createFromFile(): default dialect delivers a row after a single partial read +--EXTENSIONS-- +csv +--FILE-- +chunks) ?? ''; + } + public function stream_eof(): bool { return $this->chunks === []; } + public function stream_close(): void {} + public function stream_stat(): array|false { return false; } +} +stream_wrapper_register('csvcount', CountingStream::class); + +$rows = []; +foreach (Csv\LazyLaxCollection::createFromFile('csvcount://input') as $row) { + $rows[] = [CountingStream::$reads, $row]; +} +foreach ($rows as [$readsSoFar, $row]) { + echo $readsSoFar, ': ', json_encode($row), \PHP_EOL; +} +?> +--EXPECT-- +1: ["a","b"] +2: ["c","d"] diff --git a/ext/csv/tests/LazyCollection/getIterator/basic.phpt b/ext/csv/tests/LazyCollection/getIterator/basic.phpt new file mode 100644 index 000000000000..85c8ef854759 --- /dev/null +++ b/ext/csv/tests/LazyCollection/getIterator/basic.phpt @@ -0,0 +1,54 @@ +--TEST-- +Test Csv\LazyLaxCollection::getIterator() +--SKIPIF-- + +--FILE-- +getIterator(); +var_dump($it); + +while ($it->valid()) { + var_dump($it->key()); + var_dump($it->current()); + $it->next(); +} + +?> +--EXPECT-- +object(InternalIterator)#3 (0) { +} +int(0) +array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" +} +int(1) +array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" +} +int(2) +array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" +} diff --git a/ext/csv/tests/LazyCollection/lazy_collection_to_buffer.phpt b/ext/csv/tests/LazyCollection/lazy_collection_to_buffer.phpt new file mode 100644 index 000000000000..1724f0a15a74 --- /dev/null +++ b/ext/csv/tests/LazyCollection/lazy_collection_to_buffer.phpt @@ -0,0 +1,20 @@ +--TEST-- +Check that a LazyLaxCollection created from a buffer generates the same output +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) diff --git a/ext/csv/tests/array_to_row/basic.phpt b/ext/csv/tests/array_to_row/basic.phpt new file mode 100644 index 000000000000..16c0de1c1b4d --- /dev/null +++ b/ext/csv/tests/array_to_row/basic.phpt @@ -0,0 +1,38 @@ +--TEST-- +Test Csv\array_to_row() with standard parameters +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +string(104) "Hello,Field has spaces,"Has delimiter, inside field","Has field escape sequence "" inside field",Basic +" +string(104) "Hello,Field has spaces,"Has delimiter, inside field","Has field escape sequence "" inside field",Basic +" +string(104) "Hello,Field has spaces,"Has delimiter, inside field","Has field escape sequence "" inside field",Basic +" +string(104) "Hello,Field has spaces,"Has delimiter, inside field","Has field escape sequence "" inside field",Basic +" diff --git a/ext/csv/tests/array_to_row/error.phpt b/ext/csv/tests/array_to_row/error.phpt new file mode 100644 index 000000000000..b107ed87a156 --- /dev/null +++ b/ext/csv/tests/array_to_row/error.phpt @@ -0,0 +1,69 @@ +--TEST-- +\ValueError conditions for Csv\array_to_row() +--SKIPIF-- + +--FILE-- +getMessage() . \PHP_EOL; +} + +try { + var_dump(Csv\array_to_row($fields, ',', '')); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +try { + var_dump(Csv\array_to_row($fields, ',', '"', "")); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Same delimiter and enclosure +try { + var_dump(Csv\array_to_row($fields, ',', ',')); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Default value for the enclosure is " +try { + var_dump(Csv\array_to_row($fields, '"')); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Delimiter or enclosure using the \r\n sequence +try { + var_dump(Csv\array_to_row($fields, "\r\n")); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Default value for the enclosure is " +try { + var_dump(Csv\array_to_row($fields, ',', "\r\n")); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +?> +--EXPECT-- +Csv\array_to_row(): Argument #2 ($delimiter) must not be empty +Csv\array_to_row(): Argument #3 ($enclosure) must not be empty +Csv\array_to_row(): Argument #4 ($eolSequence) must not be empty +Csv\array_to_row(): Argument #3 ($enclosure) must not be identical to argument #2 ($delimiter) +Csv\array_to_row(): Argument #3 ($enclosure) must not be identical to argument #2 ($delimiter) +Csv\array_to_row(): Argument #4 ($eolSequence) must not be identical to argument #2 ($delimiter) +Csv\array_to_row(): Argument #4 ($eolSequence) must not be identical to argument #3 ($enclosure) diff --git a/ext/csv/tests/array_to_row/multi_byte_delimiter.phpt b/ext/csv/tests/array_to_row/multi_byte_delimiter.phpt new file mode 100644 index 000000000000..3a79b0c32d20 --- /dev/null +++ b/ext/csv/tests/array_to_row/multi_byte_delimiter.phpt @@ -0,0 +1,28 @@ +--TEST-- +Test Csv\array_to_row() with multi byte delimiter +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +string(114) "Hello。Field has spaces。"Has delimiter。 inside field"。"Has field escape sequence "" inside field"。Basic +" diff --git a/ext/csv/tests/array_to_row/multi_byte_enclosure.phpt b/ext/csv/tests/array_to_row/multi_byte_enclosure.phpt new file mode 100644 index 000000000000..70b67ef5337e --- /dev/null +++ b/ext/csv/tests/array_to_row/multi_byte_enclosure.phpt @@ -0,0 +1,28 @@ +--TEST-- +Test Csv\array_to_row() with multi byte enclosure +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +string(126) "Hello。Field has spaces。あHas delimiter。 inside fieldあ。あHas field escape sequence ああ inside fieldあ。Basic +" diff --git a/ext/csv/tests/array_to_row/non_packed.phpt b/ext/csv/tests/array_to_row/non_packed.phpt new file mode 100644 index 000000000000..0b7409043d4a --- /dev/null +++ b/ext/csv/tests/array_to_row/non_packed.phpt @@ -0,0 +1,29 @@ +--TEST-- +Test Csv\array_to_row() with non-packed array +--SKIPIF-- + +--FILE-- + 'Hello', + 'b' => 'Field has spaces', + 'c' => 'Has delimiter, inside field', + 'd' => 'Has field escape sequence " inside field', + 'e' => 'Basic', +]; + +$output = "Hello,Field has spaces,\"Has delimiter, inside field\",\"Has field escape sequence \"\" inside field\",Basic\r\n"; + +var_dump($output === Csv\array_to_row($fields)); +var_dump(Csv\array_to_row($fields)); + +?> +--EXPECT-- +bool(true) +string(104) "Hello,Field has spaces,"Has delimiter, inside field","Has field escape sequence "" inside field",Basic +" diff --git a/ext/csv/tests/array_to_row/non_stringable_elements.phpt b/ext/csv/tests/array_to_row/non_stringable_elements.phpt new file mode 100644 index 000000000000..c353b1c4822b --- /dev/null +++ b/ext/csv/tests/array_to_row/non_stringable_elements.phpt @@ -0,0 +1,45 @@ +--TEST-- +Test Csv\array_to_row() with an array not containing string elements +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +$fields = [ + 'Hello', + 'Field has spaces', + new stdClass(), + 'Has field escape sequence " inside field', + 'Basic', +]; + +try { + var_dump(Csv\array_to_row($fields)); +} catch (\Throwable $e) { + echo $e::class, $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECTF-- +Warning: Array to string conversion in %s on line %d +string(80) "Hello,Field has spaces,Array,"Has field escape sequence "" inside field",Basic +" +ErrorObject of class stdClass could not be converted to string diff --git a/ext/csv/tests/array_to_row/nul_byte_custom_eol_sequence.phpt b/ext/csv/tests/array_to_row/nul_byte_custom_eol_sequence.phpt new file mode 100644 index 000000000000..02fd4a8080c6 --- /dev/null +++ b/ext/csv/tests/array_to_row/nul_byte_custom_eol_sequence.phpt @@ -0,0 +1,26 @@ +--TEST-- +Test Csv\array_to_row() with custom EOL sequence as a nul byte +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) diff --git a/ext/csv/tests/array_to_row/nul_byte_delimiter.phpt b/ext/csv/tests/array_to_row/nul_byte_delimiter.phpt new file mode 100644 index 000000000000..5073c6692c46 --- /dev/null +++ b/ext/csv/tests/array_to_row/nul_byte_delimiter.phpt @@ -0,0 +1,28 @@ +--TEST-- +Test Csv\array_to_row() with delimiter as a nul byte +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +bool(true) diff --git a/ext/csv/tests/array_to_row/nul_byte_enclosure.phpt b/ext/csv/tests/array_to_row/nul_byte_enclosure.phpt new file mode 100644 index 000000000000..723a33e79095 --- /dev/null +++ b/ext/csv/tests/array_to_row/nul_byte_enclosure.phpt @@ -0,0 +1,26 @@ +--TEST-- +Test Csv\array_to_row() with enclosure as a nul byte +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) diff --git a/ext/csv/tests/array_to_row/with_custom_eol_sequence.phpt b/ext/csv/tests/array_to_row/with_custom_eol_sequence.phpt new file mode 100644 index 000000000000..a257cfab5145 --- /dev/null +++ b/ext/csv/tests/array_to_row/with_custom_eol_sequence.phpt @@ -0,0 +1,27 @@ +--TEST-- +Test Csv\array_to_row() with multi byte enclosure +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +string(104) "Hello,Field has spaces,"Has delimiter, inside field","Has custom EOL sequence あ inside field",Basicあ" diff --git a/ext/csv/tests/array_to_row/with_new_lines.phpt b/ext/csv/tests/array_to_row/with_new_lines.phpt new file mode 100644 index 000000000000..b600c7d0c904 --- /dev/null +++ b/ext/csv/tests/array_to_row/with_new_lines.phpt @@ -0,0 +1,29 @@ +--TEST-- +Test Csv\array_to_row() with new lines +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +bool(true) +bool(true) diff --git a/ext/csv/tests/array_to_row/with_nul_bytes.phpt b/ext/csv/tests/array_to_row/with_nul_bytes.phpt new file mode 100644 index 000000000000..525e56837268 --- /dev/null +++ b/ext/csv/tests/array_to_row/with_nul_bytes.phpt @@ -0,0 +1,30 @@ +--TEST-- +Test Csv\array_to_row() with nul bytes +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +bool(true) +bool(true) diff --git a/ext/csv/tests/buffer_to_collection/basic.phpt b/ext/csv/tests/buffer_to_collection/basic.phpt new file mode 100644 index 000000000000..11076fbe7e43 --- /dev/null +++ b/ext/csv/tests/buffer_to_collection/basic.phpt @@ -0,0 +1,156 @@ +--TEST-- +Test Csv\buffer_to_collection() with standard parameters +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} diff --git a/ext/csv/tests/buffer_to_collection/errors.phpt b/ext/csv/tests/buffer_to_collection/errors.phpt new file mode 100644 index 000000000000..84e8d8a4cb55 --- /dev/null +++ b/ext/csv/tests/buffer_to_collection/errors.phpt @@ -0,0 +1,22 @@ +--TEST-- +Error conditions for Csv\buffer_to_collection() +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +ValueError: Buffer row 2 contains 2 fields compared to 3 fields on previous rows diff --git a/ext/csv/tests/buffer_to_collection/non_escaped_enclosure.phpt b/ext/csv/tests/buffer_to_collection/non_escaped_enclosure.phpt new file mode 100644 index 000000000000..f6e024442926 --- /dev/null +++ b/ext/csv/tests/buffer_to_collection/non_escaped_enclosure.phpt @@ -0,0 +1,29 @@ +--TEST-- +Csv\buffer_to_collection() non escaped enclosure in one row +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +$string = "One,Two,Three\r\nFour,Deあux,Six\r\nSeven,Eight,Nine\r\n"; +try { + var_dump(Csv\buffer_to_collection($string, ',', 'あ')); +} catch (\Error $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +ValueError: Enclosure sequence is used in a non escaped field +ValueError: Enclosure sequence is used in a non escaped field diff --git a/ext/csv/tests/buffer_to_collection/nul_byte_custom_eol_sequence.phpt b/ext/csv/tests/buffer_to_collection/nul_byte_custom_eol_sequence.phpt new file mode 100644 index 000000000000..060403e1c091 --- /dev/null +++ b/ext/csv/tests/buffer_to_collection/nul_byte_custom_eol_sequence.phpt @@ -0,0 +1,66 @@ +--TEST-- +Test Csv\buffer_to_collection() with custom EOL sequence as a nul byte +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} diff --git a/ext/csv/tests/buffer_to_collection/strict_rfc4180.phpt b/ext/csv/tests/buffer_to_collection/strict_rfc4180.phpt new file mode 100644 index 000000000000..43dc41b2e3f1 --- /dev/null +++ b/ext/csv/tests/buffer_to_collection/strict_rfc4180.phpt @@ -0,0 +1,43 @@ +--TEST-- +Test Csv\buffer_to_collection(): input that does not follow RFC 4180 section 2 is rejected consistently +--EXTENSIONS-- +csv +--FILE-- + "a\"b,c\r\n", + 'data after the closing quote' => "\"a\"b,c\r\n", + 'space after the closing quote' => "\"a\" ,b\r\n", + 'space before the opening quote' => " \"a\",b\r\n", + 'unterminated quote' => "\"a,b\r\nc,d\r\n", + 'unterminated quote, last row' => "a,b\r\n\"c,d", + 'LF only line endings' => "a,b\nc,d\n", + 'CR inside a non escaped field' => "a\rb,c\r\n", + 'BOM before a quoted field' => "\u{FEFF}\"a\",b\r\n", + /* Valid input */ + 'quoted fields' => "\"a\",\"b\"\r\n\"c\"\"d\",\"e\r\nf\"\r\n", + 'empty quoted field, no final EOL' => "\"\",b\r\nc,\"\"", + 'non-ASCII bytes' => "\u{e9},\u{fc}\r\n", +]; +foreach ($cases as $name => $buffer) { + echo $name, ': '; + try { + echo json_encode(Csv\buffer_to_collection($buffer)), \PHP_EOL; + } catch (\ValueError $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; + } +} +?> +--EXPECT-- +quote in a non escaped field: ValueError: Enclosure sequence is used in a non escaped field +data after the closing quote: ValueError: Closing enclosure sequence must be followed by the delimiter or the EOL sequence +space after the closing quote: ValueError: Closing enclosure sequence must be followed by the delimiter or the EOL sequence +space before the opening quote: ValueError: Enclosure sequence is used in a non escaped field +unterminated quote: ValueError: Enclosure sequence is not closed +unterminated quote, last row: ValueError: Enclosure sequence is not closed +LF only line endings: ValueError: A non escaped field must not contain CR or LF characters that are not part of the EOL sequence +CR inside a non escaped field: ValueError: A non escaped field must not contain CR or LF characters that are not part of the EOL sequence +BOM before a quoted field: ValueError: Enclosure sequence is used in a non escaped field +quoted fields: [["a","b"],["c\"d","e\r\nf"]] +empty quoted field, no final EOL: [["","b"],["c",""]] +non-ASCII bytes: [["\u00e9","\u00fc"]] diff --git a/ext/csv/tests/buffer_to_collection/with_custom_eol_sequence.phpt b/ext/csv/tests/buffer_to_collection/with_custom_eol_sequence.phpt new file mode 100644 index 000000000000..923551ad43ad --- /dev/null +++ b/ext/csv/tests/buffer_to_collection/with_custom_eol_sequence.phpt @@ -0,0 +1,66 @@ +--TEST-- +Test Csv\buffer_to_collection() with standard parameters +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} diff --git a/ext/csv/tests/buffer_to_collection/with_new_lines.phpt b/ext/csv/tests/buffer_to_collection/with_new_lines.phpt new file mode 100644 index 000000000000..fcb7bdb551a4 --- /dev/null +++ b/ext/csv/tests/buffer_to_collection/with_new_lines.phpt @@ -0,0 +1,36 @@ +--TEST-- +Test Csv\buffer_to_collection() with standard parameters +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) diff --git a/ext/csv/tests/buffer_to_collection_lax/non_escaped_enclosure.phpt b/ext/csv/tests/buffer_to_collection_lax/non_escaped_enclosure.phpt new file mode 100644 index 000000000000..3aec34ce3a8f --- /dev/null +++ b/ext/csv/tests/buffer_to_collection_lax/non_escaped_enclosure.phpt @@ -0,0 +1,29 @@ +--TEST-- +Csv\buffer_to_collection_lax() non escaped enclosure in one row +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +$string = "One,Two,Three\r\nFour,Deあux,Six\r\nSeven,Eight,Nine\r\n"; +try { + var_dump(Csv\buffer_to_collection_lax($string, ',', 'あ')); +} catch (\Error $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +ValueError: Enclosure sequence is used in a non escaped field +ValueError: Enclosure sequence is used in a non escaped field diff --git a/ext/csv/tests/buffer_to_collection_lax/with_uneven_rows.phpt b/ext/csv/tests/buffer_to_collection_lax/with_uneven_rows.phpt new file mode 100644 index 000000000000..2a8ba3420a96 --- /dev/null +++ b/ext/csv/tests/buffer_to_collection_lax/with_uneven_rows.phpt @@ -0,0 +1,63 @@ +--TEST-- +Error conditions for Csv\buffer_to_collection_lax() +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(2) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} diff --git a/ext/csv/tests/bug-gl14-uaf-parameters.phpt b/ext/csv/tests/bug-gl14-uaf-parameters.phpt new file mode 100644 index 000000000000..dda059eba018 --- /dev/null +++ b/ext/csv/tests/bug-gl14-uaf-parameters.phpt @@ -0,0 +1,28 @@ +--TEST-- +GitLab bug #14 (Use-after-free in the CSV PECL extension: all methods over-release the borrowed $delimiter / $enclosure / $eolSequence string arguments) +--SKIPIF-- + +--FILE-- + +--EXPECT-- +string(2) "AA" +string(2) "55" +string(8) "oooooooo" diff --git a/ext/csv/tests/bug-gl15-null-deref-row_to_array.phpt b/ext/csv/tests/bug-gl15-null-deref-row_to_array.phpt new file mode 100644 index 000000000000..73718638404d --- /dev/null +++ b/ext/csv/tests/bug-gl15-null-deref-row_to_array.phpt @@ -0,0 +1,21 @@ +--TEST-- +GitLab bug #15 (NULL dereference when rowToArray() emits an empty field) +--SKIPIF-- + +--FILE-- + +--EXPECT-- +array(2) { + [0]=> + string(0) "" + [1]=> + string(0) "" +} diff --git a/ext/csv/tests/bug-gl16-unbound-enclosure-lookahead.phpt b/ext/csv/tests/bug-gl16-unbound-enclosure-lookahead.phpt new file mode 100644 index 000000000000..e22bfe7431d6 --- /dev/null +++ b/ext/csv/tests/bug-gl16-unbound-enclosure-lookahead.phpt @@ -0,0 +1,19 @@ +--TEST-- +GitLab bug #16 (Unbounded enclosure lookahead remains after the 2022 buffer-overflow fix) +--SKIPIF-- + +--FILE-- + +--EXPECT-- +array(1) { + [0]=> + string(7) "aaaaaaa" +} diff --git a/ext/csv/tests/collection_to_buffer/basic.phpt b/ext/csv/tests/collection_to_buffer/basic.phpt new file mode 100644 index 000000000000..fa1711065faa --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/basic.phpt @@ -0,0 +1,56 @@ +--TEST-- +Test Csv\collection_to_buffer() with standard parameters +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" diff --git a/ext/csv/tests/collection_to_buffer/basic_with_generator.phpt b/ext/csv/tests/collection_to_buffer/basic_with_generator.phpt new file mode 100644 index 000000000000..03bdf1e9f7da --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/basic_with_generator.phpt @@ -0,0 +1,60 @@ +--TEST-- +Test Csv\collection_to_buffer() with a Generator as a collection +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" diff --git a/ext/csv/tests/collection_to_buffer/basic_with_iterator.phpt b/ext/csv/tests/collection_to_buffer/basic_with_iterator.phpt new file mode 100644 index 000000000000..dd694a677cc2 --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/basic_with_iterator.phpt @@ -0,0 +1,84 @@ +--TEST-- +Test Csv\collection_to_buffer() with an Iterator as a collection +--SKIPIF-- + +--FILE-- +i = 0; + } + + public function valid(): bool { + return $this->i < 3; + } + + public function current(): array { + return self::ROWS[$this->i]; + } + + public function key(): int { + return $this->i; + } + + public function next(): void { + $this->i++; + } + + public function __destruct() {} +} + +$collection = new Collection(); + +$output = "One,Two,Three\r\nFour,Five,Six\r\nSeven,Eight,Nine\r\n"; + +var_dump($output === Csv\collection_to_buffer($collection)); +var_dump(Csv\collection_to_buffer($collection)); +var_dump(Csv\collection_to_buffer($collection, ',')); +var_dump(Csv\collection_to_buffer($collection, ',', '"')); +var_dump(Csv\collection_to_buffer($collection, ',', '"', "\r\n")); + +?> +--EXPECT-- +bool(true) +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" diff --git a/ext/csv/tests/collection_to_buffer/basic_with_iterator_aggregate.phpt b/ext/csv/tests/collection_to_buffer/basic_with_iterator_aggregate.phpt new file mode 100644 index 000000000000..2774026aca14 --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/basic_with_iterator_aggregate.phpt @@ -0,0 +1,65 @@ +--TEST-- +Test Csv\collection_to_buffer() with a IteratorAggregate as a collection +--SKIPIF-- + +--FILE-- +property4 = ['Ten', 'Eleven', 'Twelve']; + } + + public function getIterator(): Traversable { + $blah = $this; + return (function () use ($blah) { + yield $blah->property1; + yield $blah->property2; + yield $blah->property3; + yield $blah->property4; + })(); + } +} + +$collection = new Collection(); + +$output = "One,Two,Three\r\nFour,Five,Six\r\nSeven,Eight,Nine\r\nTen,Eleven,Twelve\r\n"; + +var_dump($output === Csv\collection_to_buffer($collection)); +var_dump(Csv\collection_to_buffer($collection)); +var_dump(Csv\collection_to_buffer($collection, ',')); +var_dump(Csv\collection_to_buffer($collection, ',', '"')); +var_dump(Csv\collection_to_buffer($collection, ',', '"', "\r\n")); + +?> +--EXPECT-- +bool(true) +string(67) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +Ten,Eleven,Twelve +" +string(67) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +Ten,Eleven,Twelve +" +string(67) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +Ten,Eleven,Twelve +" +string(67) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +Ten,Eleven,Twelve +" diff --git a/ext/csv/tests/collection_to_buffer/errors.phpt b/ext/csv/tests/collection_to_buffer/errors.phpt new file mode 100644 index 000000000000..2b8cbab4682a --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/errors.phpt @@ -0,0 +1,63 @@ +--TEST-- +Error conditions for Csv\collection_to_buffer() +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +$collection = [ + [ + 'One', + 'Two', + 'Three', + ], + [ + 'Four', + 'Five', + 'Six' + ], + [ + 'Seven', + 'Eight', + 'Nine', + ], + [ + 'Ten', + 'Eleven', + ], +]; + +try { + var_dump(Csv\collection_to_buffer($collection)); +} catch (\ValueError $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +TypeError: Element 1 of the collection must be an array +ValueError: Element 3 of the collection contains 2 fields compared to 3 fields on previous rows diff --git a/ext/csv/tests/collection_to_buffer/iterator_cleanup_on_error.phpt b/ext/csv/tests/collection_to_buffer/iterator_cleanup_on_error.phpt new file mode 100644 index 000000000000..bb7609211f82 --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/iterator_cleanup_on_error.phpt @@ -0,0 +1,29 @@ +--TEST-- +Test Csv\collection_to_buffer(): a Generator is destroyed immediately when conversion fails +--EXTENSIONS-- +csv +--INI-- +; the exception backtrace would otherwise keep a reference to the generator argument +zend.exception_ignore_args=1 +--FILE-- + +--EXPECT-- +generator destroyed +ValueError +after catch diff --git a/ext/csv/tests/collection_to_buffer/non_packed.phpt b/ext/csv/tests/collection_to_buffer/non_packed.phpt new file mode 100644 index 000000000000..063a68754b97 --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/non_packed.phpt @@ -0,0 +1,41 @@ +--TEST-- +Test Csv\collection_to_buffer() with non-packed array +--SKIPIF-- + +--FILE-- + [ + 'One', + 'Two', + 'Three', + ], + 'fr' => [ + 'Four', + 'Five', + 'Six', + ], + 'ja' => [ + 'Seven', + 'Eight', + 'Nine', + ], +]; + +$output = "One,Two,Three\r\nFour,Five,Six\r\nSeven,Eight,Nine\r\n"; + +var_dump($output === Csv\collection_to_buffer($collection)); +var_dump(Csv\collection_to_buffer($collection)); + +?> +--EXPECT-- +bool(true) +string(48) "One,Two,Three +Four,Five,Six +Seven,Eight,Nine +" diff --git a/ext/csv/tests/collection_to_buffer/non_stringable_elements.phpt b/ext/csv/tests/collection_to_buffer/non_stringable_elements.phpt new file mode 100644 index 000000000000..c30f5d9dc8b1 --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/non_stringable_elements.phpt @@ -0,0 +1,67 @@ +--TEST-- +Test Csv\collection_to_buffer() with a row not containing string elements +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +$collection = [ + [ + 'One', + 'Two', + 'Three', + ], + [ + 'Four', + new stdClass(), + 'Six', + ], + [ + 'Seven', + 'Eight', + 'Nine', + ], +]; + +try { + var_dump(Csv\collection_to_buffer($collection)); +} catch (\Throwable $e) { + echo $e::class, $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECTF-- +Warning: Array to string conversion in %s on line %d +string(49) "One,Two,Three +Four,Array,Six +Seven,Eight,Nine +" +ErrorObject of class stdClass could not be converted to string diff --git a/ext/csv/tests/collection_to_buffer/reference_rows.phpt b/ext/csv/tests/collection_to_buffer/reference_rows.phpt new file mode 100644 index 000000000000..472e43878a65 --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/reference_rows.phpt @@ -0,0 +1,36 @@ +--TEST-- +Test Csv\collection_to_buffer() and Csv\collection_to_file() with rows that are references +--EXTENSIONS-- +csv +--FILE-- +getMessage(), \PHP_EOL; +} +?> +--CLEAN-- + +--EXPECT-- +"a,b\r\nc,d\r\n" +"a,b\r\nc,d\r\n" +"x,y\r\n" +TypeError: Element 0 of the collection must be an array diff --git a/ext/csv/tests/collection_to_buffer/row_freed_by_tostring.phpt b/ext/csv/tests/collection_to_buffer/row_freed_by_tostring.phpt new file mode 100644 index 000000000000..0d378f75a634 --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/row_freed_by_tostring.phpt @@ -0,0 +1,49 @@ +--TEST-- +Test Csv\collection_to_buffer() and Csv\collection_to_file(): a row freed by a field's __toString() is not used after free +--DESCRIPTION-- +Converting a field to string can run userland code that frees the row being formatted: resuming +the generator that produced it, or overwriting the element of an ArrayIterator. The row must +stay alive until it has been formatted. +--EXTENSIONS-- +csv +--FILE-- +next(); + return str_repeat('x', 8); + } +} +function rows(): Generator { + yield [new ResumesGenerator, str_repeat('a', 64), str_repeat('b', 64)]; + yield ['c', 'd', 'e']; +} + +$gen = rows(); +var_dump(Csv\collection_to_buffer($gen)); + +$file = __DIR__ . '/row_freed_by_tostring.csv'; +$gen = rows(); +Csv\collection_to_file($file, $gen); +var_dump(file_get_contents($file)); + +class OverwritesRow { + public function __toString(): string { + $GLOBALS['it'][0] = ['z', 'z', 'z']; + return 'y'; + } +} +$it = new ArrayIterator([[new OverwritesRow, str_repeat('a', 64), str_repeat('b', 64)]]); +var_dump(Csv\collection_to_buffer($it)); +?> +--CLEAN-- + +--EXPECT-- +string(140) "xxxxxxxx,aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa,bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb +" +string(140) "xxxxxxxx,aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa,bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb +" +string(133) "y,aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa,bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb +" diff --git a/ext/csv/tests/collection_to_buffer/sparse_collection.phpt b/ext/csv/tests/collection_to_buffer/sparse_collection.phpt new file mode 100644 index 000000000000..7867255ef02f --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/sparse_collection.phpt @@ -0,0 +1,29 @@ +--TEST-- +Test Csv\collection_to_buffer() and Csv\collection_to_file() with a sparse (holey) array +--EXTENSIONS-- +csv +--FILE-- +\n", Csv\collection_to_buffer($rows)); + +$file = __DIR__ . '/sparse_collection.csv'; +Csv\collection_to_file($file, $rows); +echo str_replace("\r\n", "\n", file_get_contents($file)); + +/* An array that is nothing but holes behaves like an empty collection */ +$holes = [['a']]; +unset($holes[0]); +var_dump(Csv\collection_to_buffer($holes)); +?> +--CLEAN-- + +--EXPECT-- +a,b +c,d +a,b +c,d +string(0) "" diff --git a/ext/csv/tests/collection_to_buffer/with_custom_eol_sequence.phpt b/ext/csv/tests/collection_to_buffer/with_custom_eol_sequence.phpt new file mode 100644 index 000000000000..63ba18c6c9ae --- /dev/null +++ b/ext/csv/tests/collection_to_buffer/with_custom_eol_sequence.phpt @@ -0,0 +1,38 @@ +--TEST-- +Test Csv\collection_to_buffer() with standard parameters +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +string(51) "One,Two,ThreeあFour,Five,SixあSeven,Eight,Nineあ" diff --git a/ext/csv/tests/collection_to_file/basic.phpt b/ext/csv/tests/collection_to_file/basic.phpt new file mode 100644 index 000000000000..9ec5c089df4d --- /dev/null +++ b/ext/csv/tests/collection_to_file/basic.phpt @@ -0,0 +1,27 @@ +--TEST-- +Test Csv\collection_to_file(): basic behaviour +--EXTENSIONS-- +csv +--FILE-- +\n", file_get_contents($file)); +?> +--CLEAN-- + +--EXPECT-- +Hello,World,this +is,a,CSV file +"with ""enclosures""","and,delimiters","and +new lines" diff --git a/ext/csv/tests/collection_to_file/basic_with_generator.phpt b/ext/csv/tests/collection_to_file/basic_with_generator.phpt new file mode 100644 index 000000000000..cda7a32c126d --- /dev/null +++ b/ext/csv/tests/collection_to_file/basic_with_generator.phpt @@ -0,0 +1,24 @@ +--TEST-- +Test Csv\collection_to_file(): with a Generator collection +--EXTENSIONS-- +csv +--FILE-- +\n", file_get_contents($file)); +?> +--CLEAN-- + +--EXPECT-- +a,b +c,d diff --git a/ext/csv/tests/collection_to_file/close_failure.phpt b/ext/csv/tests/collection_to_file/close_failure.phpt new file mode 100644 index 000000000000..4e7dca937efe --- /dev/null +++ b/ext/csv/tests/collection_to_file/close_failure.phpt @@ -0,0 +1,22 @@ +--TEST-- +Test Csv\collection_to_file(): a failure while flushing buffered data on close is reported +--EXTENSIONS-- +csv +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} +?> +--EXPECT-- +Error: Failed to finish writing to "compress.zlib:///dev/full" diff --git a/ext/csv/tests/collection_to_file/errors.phpt b/ext/csv/tests/collection_to_file/errors.phpt new file mode 100644 index 000000000000..5869694f1614 --- /dev/null +++ b/ext/csv/tests/collection_to_file/errors.phpt @@ -0,0 +1,48 @@ +--TEST-- +Test Csv\collection_to_file(): error conditions +--EXTENSIONS-- +csv +--FILE-- +getMessage(), \PHP_EOL; +} +var_dump(file_exists($file)); + +try { + Csv\collection_to_file(__DIR__ . '/no_such_dir/out.csv', [['a', 'b']]); +} catch (Error $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} + +try { + Csv\collection_to_file($file, [['a', 'b'], 'not an array']); +} catch (TypeError $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} +/* Rows written before the failure remain in the file */ +echo str_replace("\r\n", "\n", file_get_contents($file)); + +try { + Csv\collection_to_file($file, [['a', 'b'], ['c']]); +} catch (ValueError $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} +echo str_replace("\r\n", "\n", file_get_contents($file)); +?> +--CLEAN-- + +--EXPECTF-- +ValueError: Csv\collection_to_file(): Argument #3 ($delimiter) must not be empty +bool(false) +Error: Csv\collection_to_file(): Failed to open stream: No such file or directory +TypeError: Element 1 of the collection must be an array +a,b +ValueError: Element 1 of the collection contains 1 fields compared to 2 fields on previous rows +a,b diff --git a/ext/csv/tests/collection_to_file/retained_resource_reference.phpt b/ext/csv/tests/collection_to_file/retained_resource_reference.phpt new file mode 100644 index 000000000000..08079dd67c5a --- /dev/null +++ b/ext/csv/tests/collection_to_file/retained_resource_reference.phpt @@ -0,0 +1,66 @@ +--TEST-- +Test Csv\collection_to_file(): retained references to the internal stream's resource are released safely +--DESCRIPTION-- +Regression test for a leak caught by the debug builds of the GitHub CI on the ext/csv pull +request (see LazyCollection/fromFile/retained_resource_reference_release_order.phpt): +collection_to_file() makes its stream private the same way, so a resource reference retained +while the stream is being opened must remain a closed resource until it is released. +--EXTENSIONS-- +csv +--FILE-- +datalen; + stream_bucket_append($out, $bucket); + } + return PSFS_PASS_ON; + } +} +stream_filter_register('grabbing', GrabbingFilter::class); + +function closed_ids(): array { + return array_map('intval', array_values(get_resources('Unknown'))); +} + +$file = __DIR__ . '/retained_resource_reference_to_file.csv'; +$baseline = closed_ids(); + +Csv\collection_to_file('php://filter/write=grabbing/resource=' . $file, [['a', 'b'], ['c', 'd']]); +$closed = array_values(array_filter(GrabbingFilter::$grabbed, fn($res) => !is_resource($res))); +GrabbingFilter::$grabbed = []; +var_dump(count($closed)); +$a = $closed[0]; +$b = $a; +unset($closed); +$id = (int) $a; +var_dump(gettype($a)); +var_dump(in_array($id, closed_ids(), true)); +var_dump(file_get_contents($file)); +unset($a); +var_dump(in_array($id, closed_ids(), true)); +unset($b); +var_dump(closed_ids() === $baseline); +echo "done\n"; +?> +--CLEAN-- + +--EXPECT-- +int(1) +string(17) "resource (closed)" +bool(true) +string(10) "a,b +c,d +" +bool(true) +bool(true) +done diff --git a/ext/csv/tests/collection_to_file/stream_cannot_be_closed.phpt b/ext/csv/tests/collection_to_file/stream_cannot_be_closed.phpt new file mode 100644 index 000000000000..cfa2a17542fc --- /dev/null +++ b/ext/csv/tests/collection_to_file/stream_cannot_be_closed.phpt @@ -0,0 +1,35 @@ +--TEST-- +Test Csv\collection_to_file(): userland cannot close the internal stream during iteration +--EXTENSIONS-- +csv +--FILE-- +\n", file_get_contents($file)); +?> +--CLEAN-- + +--EXPECTF-- +Warning: fclose(): cannot close the provided stream, as it must not be manually closed in %s on line %d +bool(false) +string(17) "resource (closed)" +a,b +c,d diff --git a/ext/csv/tests/collection_to_file/with_custom_eol_sequence.phpt b/ext/csv/tests/collection_to_file/with_custom_eol_sequence.phpt new file mode 100644 index 000000000000..1a323e3ef27c --- /dev/null +++ b/ext/csv/tests/collection_to_file/with_custom_eol_sequence.phpt @@ -0,0 +1,23 @@ +--TEST-- +Test Csv\collection_to_file(): with a custom EOL sequence +--EXTENSIONS-- +csv +--FILE-- + +--CLEAN-- + +--EXPECT-- +string(16) "a,b|EOL|c,d|EOL|" diff --git a/ext/csv/tests/csv_functions_inverses.phpt b/ext/csv/tests/csv_functions_inverses.phpt new file mode 100644 index 000000000000..5ccb1bf95e50 --- /dev/null +++ b/ext/csv/tests/csv_functions_inverses.phpt @@ -0,0 +1,23 @@ +--TEST-- +Test cvs functions to see if they can invert them self +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) diff --git a/ext/csv/tests/enclosure_buffer_overflow.phpt b/ext/csv/tests/enclosure_buffer_overflow.phpt new file mode 100644 index 000000000000..3ebdd0cf045e --- /dev/null +++ b/ext/csv/tests/enclosure_buffer_overflow.phpt @@ -0,0 +1,148 @@ +--TEST-- +Csv\row_to_array() buffer overflow bug +--SKIPIF-- + +--FILE-- + +--EXPECT-- +array(3) { + [0]=> + string(4) "key1" + [1]=> + string(4) "key2" + [2]=> + string(4) "key3" +} +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} +array(3) { + [0]=> + array(3) { + [0]=> + string(3) "One" + [1]=> + string(3) "Two" + [2]=> + string(5) "Three" + } + [1]=> + array(3) { + [0]=> + string(4) "Four" + [1]=> + string(4) "Five" + [2]=> + string(3) "Six" + } + [2]=> + array(3) { + [0]=> + string(5) "Seven" + [1]=> + string(5) "Eight" + [2]=> + string(4) "Nine" + } +} diff --git a/ext/csv/tests/row_to_array/basic.phpt b/ext/csv/tests/row_to_array/basic.phpt new file mode 100644 index 000000000000..015c20aecfef --- /dev/null +++ b/ext/csv/tests/row_to_array/basic.phpt @@ -0,0 +1,78 @@ +--TEST-- +Test Csv\row_to_array() with standard parameters +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(5) { + [0]=> + string(5) "Hello" + [1]=> + string(16) "Field has spaces" + [2]=> + string(27) "Has delimiter, inside field" + [3]=> + string(40) "Has field escape sequence " inside field" + [4]=> + string(5) "Basic" +} +array(5) { + [0]=> + string(5) "Hello" + [1]=> + string(16) "Field has spaces" + [2]=> + string(27) "Has delimiter, inside field" + [3]=> + string(40) "Has field escape sequence " inside field" + [4]=> + string(5) "Basic" +} +array(5) { + [0]=> + string(5) "Hello" + [1]=> + string(16) "Field has spaces" + [2]=> + string(27) "Has delimiter, inside field" + [3]=> + string(40) "Has field escape sequence " inside field" + [4]=> + string(5) "Basic" +} +array(5) { + [0]=> + string(5) "Hello" + [1]=> + string(16) "Field has spaces" + [2]=> + string(27) "Has delimiter, inside field" + [3]=> + string(40) "Has field escape sequence " inside field" + [4]=> + string(5) "Basic" +} diff --git a/ext/csv/tests/row_to_array/empty_row.phpt b/ext/csv/tests/row_to_array/empty_row.phpt new file mode 100644 index 000000000000..b927d49bd63f --- /dev/null +++ b/ext/csv/tests/row_to_array/empty_row.phpt @@ -0,0 +1,32 @@ +--TEST-- +Test Csv\row_to_array() with an empty row is a single empty field +--SKIPIF-- + +--FILE-- + +--EXPECT-- +array(1) { + [0]=> + string(0) "" +} +array(1) { + [0]=> + string(0) "" +} +array(1) { + [0]=> + string(0) "" +} +string(2) " +" diff --git a/ext/csv/tests/row_to_array/error.phpt b/ext/csv/tests/row_to_array/error.phpt new file mode 100644 index 000000000000..ce6135d3cd3e --- /dev/null +++ b/ext/csv/tests/row_to_array/error.phpt @@ -0,0 +1,69 @@ +--TEST-- +\ValueError conditions for Csv\row_to_array() +--SKIPIF-- + +--FILE-- +getMessage() . \PHP_EOL; +} + +try { + var_dump(Csv\row_to_array($string, ',', '')); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +try { + var_dump(Csv\row_to_array($string, ',', '"', "")); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Same delimiter and enclosure +try { + var_dump(Csv\row_to_array($string, ',', ',')); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Default value for the enclosure is " +try { + var_dump(Csv\row_to_array($string, '"')); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Delimiter or enclosure using the \r\n sequence +try { + var_dump(Csv\row_to_array($string, "\r\n")); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +// Default value for the enclosure is " +try { + var_dump(Csv\row_to_array($string, ',', "\r\n")); +} catch (\ValueError $e) { + echo $e->getMessage() . \PHP_EOL; +} + +?> +--EXPECT-- +Csv\row_to_array(): Argument #2 ($delimiter) must not be empty +Csv\row_to_array(): Argument #3 ($enclosure) must not be empty +Csv\row_to_array(): Argument #4 ($eolSequence) must not be empty +Csv\row_to_array(): Argument #3 ($enclosure) must not be identical to argument #2 ($delimiter) +Csv\row_to_array(): Argument #3 ($enclosure) must not be identical to argument #2 ($delimiter) +Csv\row_to_array(): Argument #4 ($eolSequence) must not be identical to argument #2 ($delimiter) +Csv\row_to_array(): Argument #4 ($eolSequence) must not be identical to argument #3 ($enclosure) diff --git a/ext/csv/tests/row_to_array/more_than_one_row.phpt b/ext/csv/tests/row_to_array/more_than_one_row.phpt new file mode 100644 index 000000000000..5c6af6dba4d6 --- /dev/null +++ b/ext/csv/tests/row_to_array/more_than_one_row.phpt @@ -0,0 +1,32 @@ +--TEST-- +Test Csv\row_to_array(): data after the end of the row is an error +--EXTENSIONS-- +csv +--FILE-- +getMessage(), \PHP_EOL; + } +} +?> +--EXPECT-- +array(2) { + [0]=> + string(1) "a" + [1]=> + string(1) "b" +} +array(2) { + [0]=> + string(1) "a" + [1]=> + string(1) "b" +} +ValueError: Csv\row_to_array(): Argument #1 ($row) must contain a single row +ValueError: Csv\row_to_array(): Argument #1 ($row) must contain a single row +ValueError: Csv\row_to_array(): Argument #1 ($row) must contain a single row diff --git a/ext/csv/tests/row_to_array/multi_byte_delimiter.phpt b/ext/csv/tests/row_to_array/multi_byte_delimiter.phpt new file mode 100644 index 000000000000..ad404b39e6ee --- /dev/null +++ b/ext/csv/tests/row_to_array/multi_byte_delimiter.phpt @@ -0,0 +1,38 @@ +--TEST-- +Test Csv\row_to_array() with multi byte delimiter +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(5) { + [0]=> + string(5) "Hello" + [1]=> + string(16) "Field has spaces" + [2]=> + string(29) "Has delimiter。 inside field" + [3]=> + string(40) "Has field escape sequence " inside field" + [4]=> + string(5) "Basic" +} diff --git a/ext/csv/tests/row_to_array/multi_byte_enclosure.phpt b/ext/csv/tests/row_to_array/multi_byte_enclosure.phpt new file mode 100644 index 000000000000..1f04ffe2f55c --- /dev/null +++ b/ext/csv/tests/row_to_array/multi_byte_enclosure.phpt @@ -0,0 +1,38 @@ +--TEST-- +Test Csv\row_to_array() with multi byte enclosure +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(5) { + [0]=> + string(5) "Hello" + [1]=> + string(16) "Field has spaces" + [2]=> + string(29) "Has delimiter。 inside field" + [3]=> + string(42) "Has field escape sequence あ inside field" + [4]=> + string(5) "Basic" +} diff --git a/ext/csv/tests/row_to_array/multi_byte_enclosure_doubled_roundtrip.phpt b/ext/csv/tests/row_to_array/multi_byte_enclosure_doubled_roundtrip.phpt new file mode 100644 index 000000000000..b678a2662d54 --- /dev/null +++ b/ext/csv/tests/row_to_array/multi_byte_enclosure_doubled_roundtrip.phpt @@ -0,0 +1,19 @@ +--TEST-- +Test Csv\row_to_array(): field containing a multibyte enclosure sequence round-trips +--EXTENSIONS-- +csv +--FILE-- + +--EXPECT-- +"aaaaaaaaa\n" +["aaa"] +"x,--y----z--,w\n" +["x","y--z","w"] diff --git a/ext/csv/tests/row_to_array/non_escaped_enclosure.phpt b/ext/csv/tests/row_to_array/non_escaped_enclosure.phpt new file mode 100644 index 000000000000..1d7e3ab62d87 --- /dev/null +++ b/ext/csv/tests/row_to_array/non_escaped_enclosure.phpt @@ -0,0 +1,30 @@ +--TEST-- +Test Csv\row_to_array() with enclosure in middle of field +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +$string = "key1,keあy2,key3\r\n"; +try { + var_dump(Csv\row_to_array($string, ',', 'あ')); +} catch (\ValueError $e) { + echo $e->getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +Enclosure sequence is used in a non escaped field +Enclosure sequence is used in a non escaped field diff --git a/ext/csv/tests/row_to_array/nul_byte_custom_eol_sequence.phpt b/ext/csv/tests/row_to_array/nul_byte_custom_eol_sequence.phpt new file mode 100644 index 000000000000..2aae88d96dd7 --- /dev/null +++ b/ext/csv/tests/row_to_array/nul_byte_custom_eol_sequence.phpt @@ -0,0 +1,26 @@ +--TEST-- +Test Csv\row_to_array() with delimiter as a nul byte +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) diff --git a/ext/csv/tests/row_to_array/nul_byte_delimiter.phpt b/ext/csv/tests/row_to_array/nul_byte_delimiter.phpt new file mode 100644 index 000000000000..41479972ea85 --- /dev/null +++ b/ext/csv/tests/row_to_array/nul_byte_delimiter.phpt @@ -0,0 +1,28 @@ +--TEST-- +Test Csv\row_to_array() with delimiter as a nul byte +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +bool(true) diff --git a/ext/csv/tests/row_to_array/nul_byte_enclosure.phpt b/ext/csv/tests/row_to_array/nul_byte_enclosure.phpt new file mode 100644 index 000000000000..e6fe38cae0be --- /dev/null +++ b/ext/csv/tests/row_to_array/nul_byte_enclosure.phpt @@ -0,0 +1,26 @@ +--TEST-- +Test Csv\row_to_array() with enclosure as a nul byte +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) diff --git a/ext/csv/tests/row_to_array/self_overlapping_enclosure.phpt b/ext/csv/tests/row_to_array/self_overlapping_enclosure.phpt new file mode 100644 index 000000000000..394d606d212f --- /dev/null +++ b/ext/csv/tests/row_to_array/self_overlapping_enclosure.phpt @@ -0,0 +1,26 @@ +--TEST-- +Test Csv\row_to_array(): self-overlapping multibyte enclosures round-trip and stay strict +--EXTENSIONS-- +csv +--FILE-- + ['a', 'aa', 'aaa', 'xa', 'ax', 'a,a', "a\na", '', 'aaaa'], '--' => ['y-', '-y', '-', '---', 'y--z']] as $enclosure => $fields) { + foreach ($fields as $field) { + $row = Csv\array_to_row([$field, 'z'], ',', $enclosure, "\n"); + $parsed = Csv\row_to_array($row, ',', $enclosure, "\n"); + if ($parsed !== [$field, 'z']) { + echo "Mismatch for ", json_encode($field), " with ", $enclosure, ": ", json_encode($row), " -> ", json_encode($parsed), \PHP_EOL; + } + } +} +/* Data after the closing enclosure is still an error */ +try { + var_dump(Csv\row_to_array("--a--b,c\n", ',', '--', "\n")); +} catch (\ValueError $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} +echo "done", \PHP_EOL; +?> +--EXPECT-- +ValueError: Enclosure sequence is not closed +done diff --git a/ext/csv/tests/row_to_array/unicode_bom.phpt b/ext/csv/tests/row_to_array/unicode_bom.phpt new file mode 100644 index 000000000000..1edf9788d830 --- /dev/null +++ b/ext/csv/tests/row_to_array/unicode_bom.phpt @@ -0,0 +1,32 @@ +--TEST-- +Test Csv\row_to_array() with BOM at start and enclosures +--SKIPIF-- + +--FILE-- +getMessage(), \PHP_EOL; +} + +?> +--EXPECT-- +Enclosure sequence is used in a non escaped field diff --git a/ext/csv/tests/row_to_array/with_custom_eol_sequence.phpt b/ext/csv/tests/row_to_array/with_custom_eol_sequence.phpt new file mode 100644 index 000000000000..90ba4fccd208 --- /dev/null +++ b/ext/csv/tests/row_to_array/with_custom_eol_sequence.phpt @@ -0,0 +1,39 @@ +--TEST-- +Test Csv\row_to_array() with new lines +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +array(5) { + [0]=> + string(5) "Hello" + [1]=> + string(16) "Field has spaces" + [2]=> + string(27) "Has delimiter, inside field" + [3]=> + string(40) "Has custom EOL sequence あ inside field" + [4]=> + string(5) "Basic" +} diff --git a/ext/csv/tests/row_to_array/with_new_lines.phpt b/ext/csv/tests/row_to_array/with_new_lines.phpt new file mode 100644 index 000000000000..30f14ed55b72 --- /dev/null +++ b/ext/csv/tests/row_to_array/with_new_lines.phpt @@ -0,0 +1,29 @@ +--TEST-- +Test Csv\row_to_array() with new lines +--SKIPIF-- + +--FILE-- + +--EXPECT-- +bool(true) +bool(true) +bool(true) diff --git a/ext/csv/tests/row_to_array/with_new_lines_strict_compliance.phpt b/ext/csv/tests/row_to_array/with_new_lines_strict_compliance.phpt new file mode 100644 index 000000000000..f793a7757251 --- /dev/null +++ b/ext/csv/tests/row_to_array/with_new_lines_strict_compliance.phpt @@ -0,0 +1,54 @@ +--TEST-- +Test Csv\row_to_array() with new lines (strict compliance version) +--DESCRIPTION-- +RFC 4180 only allows CR and LF inside enclosed fields (TEXTDATA excludes them), which is also +how Csv\array_to_row() writes them. A CR or LF in a non escaped field that is not part of the +EOL sequence is an error, which also catches input using a different EOL sequence. +--EXTENSIONS-- +csv +--FILE-- +getMessage(), \PHP_EOL; + } +} + +/* With "\n" as the EOL sequence a CRLF terminated row leaves a stray CR */ +try { + var_dump(Csv\row_to_array("a,b\r\n", ',', '"', "\n")); +} catch (\ValueError $e) { + echo $e::class, ': ', $e->getMessage(), \PHP_EOL; +} +?> +--EXPECT-- +bool(true) +bool(true) +bool(true) +bool(true) +ValueError: A non escaped field must not contain CR or LF characters that are not part of the EOL sequence +ValueError: A non escaped field must not contain CR or LF characters that are not part of the EOL sequence +ValueError: A non escaped field must not contain CR or LF characters that are not part of the EOL sequence +ValueError: A non escaped field must not contain CR or LF characters that are not part of the EOL sequence +ValueError: A non escaped field must not contain CR or LF characters that are not part of the EOL sequence