Re: resend: I have written a new php function but I need help

From: Date: Wed, 09 Jun 1999 15:50:08 +0000
Subject: Re: resend: I have written a new php function but I need help
References: 1  Groups: php.dev 
Request: Send a blank email to php-dev+get-6776@lists.php.net to get a copy of this message
Hello Eric, On 09-Jun-99 11:36:40, you wrote: >problem, right? Well I am getting many "document contains no data" when >using the function but I added some debug code and established that the crash >is occurring after all of my code. So I am very confused about what the >problem is. Perhaps malloc related? I am not getting any core files that I >can use gdb on but that may be due to some other configuration on the system. I couldn't detect anything wrong, but it seems you are hitting somewhere out of the allocated memory space. If I were you I would try your code in PHP as standalone CGI program. That way you can single step with gdb and figure what the problem is. You should also enable PHP debugging mode and use PHP memory allocation functions instead (maybe this is your problem). These functions let you trace many allocated memory misusages. > I know I could just write a userspace function to accomplish the same >purpose of this function but I thought that it would be useful to include in >PHP itself and I wanted to contribute something. So please help me find the >problem or give me some ideas to help me track it down myself. TIA I also have written a PHP function to do precisely what you do in C (see source below). Anyway, I believe your function is very welcome in PHP because it is very requested despite it only supports ISOLatin1 encoding. But I think you could write it more efficiently using a sorted entities array and bsearch() to look up the entities in the array. Since I have already written something similar in C in the past, here it follows. Notice that it comes with a more complete array of entities than you were using. Regards, Manuel Lemos static const struct lISOLatinName { const unsigned char *Name; const unsigned char Code; } ISOLatin1Decoding[]= { {(const unsigned char *)"Aacute", 193}, {(const unsigned char *)"Acirc", 194}, {(const unsigned char *)"AElig", 198}, {(const unsigned char *)"Agrave", 192}, {(const unsigned char *)"Aring", 197}, {(const unsigned char *)"Atilde", 195}, {(const unsigned char *)"Auml", 196}, {(const unsigned char *)"Ccedil", 199}, {(const unsigned char *)"Eacute", 201}, {(const unsigned char *)"Ecirc", 202}, {(const unsigned char *)"Egrave", 200}, {(const unsigned char *)"ETH", 208}, {(const unsigned char *)"Euml", 203}, {(const unsigned char *)"Iacute", 205}, {(const unsigned char *)"Icirc", 206}, {(const unsigned char *)"Igrave", 204}, {(const unsigned char *)"Iuml", 207}, {(const unsigned char *)"Ntilde", 209}, {(const unsigned char *)"Oacute", 211}, {(const unsigned char *)"Ocirc", 212}, {(const unsigned char *)"Ograve", 210}, {(const unsigned char *)"Oslash", 216}, {(const unsigned char *)"Otilde", 213}, {(const unsigned char *)"Ouml", 214}, {(const unsigned char *)"THORN", 222}, {(const unsigned char *)"Uacute", 218}, {(const unsigned char *)"Ucirc", 219}, {(const unsigned char *)"Ugrave", 217}, {(const unsigned char *)"Uuml", 220}, {(const unsigned char *)"Yacute", 221}, {(const unsigned char *)"aacute", 225}, {(const unsigned char *)"acirc", 226}, {(const unsigned char *)"aelig", 230}, {(const unsigned char *)"agrave", 224}, {(const unsigned char *)"amp", '&'}, {(const unsigned char *)"aring", 229}, {(const unsigned char *)"atilde", 227}, {(const unsigned char *)"auml", 228}, {(const unsigned char *)"ccedil", 231}, {(const unsigned char *)"eacute", 233}, {(const unsigned char *)"ecirc", 234}, {(const unsigned char *)"egrave", 232}, {(const unsigned char *)"eth", 240}, {(const unsigned char *)"euml", 235}, {(const unsigned char *)"gt", '>'}, {(const unsigned char *)"iacute", 237}, {(const unsigned char *)"icirc", 238}, {(const unsigned char *)"igrave", 236}, {(const unsigned char *)"iuml", 239}, {(const unsigned char *)"lt", '<'}, {(const unsigned char *)"nbsp", 160}, {(const unsigned char *)"ntilde", 241}, {(const unsigned char *)"oacute", 243}, {(const unsigned char *)"ocirc", 244}, {(const unsigned char *)"ograve", 242}, {(const unsigned char *)"oslash", 248}, {(const unsigned char *)"otilde", 245}, {(const unsigned char *)"ouml", 246}, {(const unsigned char *)"quot", '"'}, {(const unsigned char *)"szlig", 223}, {(const unsigned char *)"thorn", 254}, {(const unsigned char *)"uacute", 250}, {(const unsigned char *)"ucirc", 251}, {(const unsigned char *)"ugrave", 249}, {(const unsigned char *)"uuml", 252}, {(const unsigned char *)"yacute", 253}, {(const unsigned char *)"yuml", 255}, }; struct lLimitedName { unsigned char *Name; size_t Length; }; static int CompareNames(const void *limited_name,const void *array) { size_t length; const unsigned char *name,*array_name; length=((struct lLimitedName *)limited_name)->Length; name=((struct lLimitedName *)limited_name)->Name; array_name=((struct lISOLatinName *)array)->Name; for(;;length--,array_name++,name++) { if(length==0) return(0 - *array_name); if(*name!=*array_name) return(*name- *array_name); } } unsigned char lGetISOLatin1NamedCharacter(name,length) char *name; size_t length; { struct lISOLatinName *array_name; struct lLimitedName limited_name; limited_name.Name=(unsigned char *)name; limited_name.Length=length; return((unsigned char)((array_name=(struct lISOLatinName *)bsearch(&limited_name,ISOLatin1Decoding,sizeof(ISOLatin1Decoding)/sizeof(ISOLatin1Decoding[0]),sizeof(ISOLatin1Decoding[0]),CompareNames)) ? array_name->Code : '\0')); } $ISOLatin1Encodings=array( "AElig"=>chr(198), "Aacute"=>chr(193), "Acirc"=>chr(194), "Agrave"=>chr(192), "Aring"=>chr(197), "Atilde"=>chr(195), "Auml"=>chr(196), "Ccedil"=>chr(199), "Eacute"=>chr(201), "Ecirc"=>chr(202), "Egrave"=>chr(200), "ETH"=>chr(208), "Euml"=>chr(203), "Iacute"=>chr(205), "Icirc"=>chr(206), "Igrave"=>chr(204), "Iuml"=>chr(207), "Ntilde"=>chr(209), "Oacute"=>chr(211), "Ocirc"=>chr(212), "Ograve"=>chr(210), "Oslash"=>chr(216), "Otilde"=>chr(213), "Ouml"=>chr(214), "THORN"=>chr(222), "Uacute"=>chr(218), "Ucirc"=>chr(219), "Ugrave"=>chr(217), "Uuml"=>chr(220), "Yacute"=>chr(221), "aacute"=>chr(225), "acirc"=>chr(226), "aelig"=>chr(230), "agrave"=>chr(224), "amp"=>"&", "aring"=>chr(229), "atilde"=>chr(227), "auml"=>chr(228), "ccedil"=>chr(231), "eacute"=>chr(233), "ecirc"=>chr(234), "egrave"=>chr(232), "eth"=>chr(240), "euml"=>chr(235), "gt"=>">", "iacute"=>chr(237), "icirc"=>chr(238), "igrave"=>chr(236), "iuml"=>chr(239), "lt"=>"<", "nbsp"=>chr(160), "ntilde"=>chr(241), "oacute"=>chr(243), "ocirc"=>chr(244), "ograve"=>chr(242), "oslash"=>chr(248), "otilde"=>chr(245), "ouml"=>chr(246), "quot"=>"\"", "szlig"=>chr(223), "thorn"=>chr(254), "uacute"=>chr(250), "ucirc"=>chr(251), "ugrave"=>chr(249), "uuml"=>chr(252), "yacute"=>chr(253), "yuml"=>chr(255) ); class html_class { var $html_encodings=array(); Function Decode($string) { for($mapped="";;) { $token=strtok($string,"&"); $string=strtok(""); $mapped.=$token; if(($token=strtok($string,";"))=="") break; $string=strtok(""); if(IsSet($this->html_encodings[$token])) $mapped.=$this->html_encodings[$token]; else { if(ereg("#([[:digit:]])+",$token) && ($order=intval(substr($token,1,strlen($token)-1)))<256) $mapped.=chr($order); else $mapped.="&$token;"; } } return($mapped); } }; Function ISOLatin1HTMLDecode($string) { global $ISOLatin1Encodings; $html=new html_class; $html->html_encodings=$ISOLatin1Encodings; return($html->Decode($string)); } E-mail: mlemos@acm.org URL: http://www.e-na.net/the_author.html PGP key: finger://mlemos@zeus.ci.ua.pt --

« previous php.dev (#6776) next »