PHP字符串截取,支持中文和其他编码

2020

2020年4月23日2020年4月24日KAng

PHP字符串截取,支持中文和其他编码

/**
 * 字符串截取，支持中文和其他编码
 * @static
 * @access public
 * @param string $str 需要转换的字符串
 * @param string $start 开始位置
 * @param string $length 截取长度
 * @param string $charset 编码格式
 * @param string $suffix 截断显示字符
 * @return string
 */
function msubstr($str, $start=0, $length, $charset="utf-8", $suffix=true) {
    if(function_exists("mb_substr"))
        $slice = mb_substr($str, $start, $length, $charset);
    elseif(function_exists('iconv_substr')) {
        $slice = iconv_substr($str,$start,$length,$charset);
        if(false === $slice) {
            $slice = '';
        }
    }else{
        $re['utf-8']   = "/[\x01-\x7f]|[\xc2-\xdf][\x80-\xbf]|[\xe0-\xef][\x80-\xbf]{2}|[\xf0-\xff][\x80-\xbf]{3}/";
        $re['gb2312'] = "/[\x01-\x7f]|[\xb0-\xf7][\xa0-\xfe]/";
        $re['gbk']    = "/[\x01-\x7f]|[\x81-\xfe][\x40-\xfe]/";
        $re['big5']   = "/[\x01-\x7f]|[\x81-\xfe]([\x40-\x7e]|\xa1-\xfe])/";
        preg_match_all($re[$charset], $str, $match);
        $slice = join("",array_slice($match[0], $start, $length));
    }
    return $suffix ? $slice.'...' : $slice;
}

// 中文字符截取 显示...
function msubstr($str, $start = 0, $length, $charset = "utf-8", $suffix = true) {
    $str = preg_replace("/<([a-zA-Z]+)[^>]*>/", "", $str);
    if (function_exists("mb_substr")) {
        if ($suffix && ceil(strlen($str) / 2) > $length)
            return mb_substr($str, $start, $length, $charset) . "...";
        else
            return mb_substr($str, $start, $length, $charset);
    } elseif (function_exists('iconv_substr')) {
        if ($suffix && ceil(strlen($str) / 2) > $length)
            return iconv_substr($str, $start, $length, $charset) . "...";
        else
            return iconv_substr($str, $start, $length, $charset);
    }
    $re ['utf-8'] = "/[\x01-\x7f]|[\xc2-\xdf][\x80-\xbf]|[\xe0-\xef][\x80-\xbf]{2}|[\xf0-\xff][\x80-\xbf]{3}/";
    $re ['gb2312'] = "/[\x01-\x7f]|[\xb0-\xf7][\xa0-\xfe]/";
    $re ['gbk'] = "/[\x01-\x7f]|[\x81-\xfe][\x40-\xfe]/";
    $re ['big5'] = "/[\x01-\x7f]|[\x81-\xfe]([\x40-\x7e]|\xa1-\xfe])/";
    preg_match_all($re [$charset], $str, $match);
    $slice = join("", array_slice($match [0], $start, $length));
    if ($suffix) {
        return $slice . "…";
    } else {
        return $slice;
    }
}

// 中文字符截取 显示***
function msubstr_xxx($str, $start = 0, $length, $charset = "utf-8", $suffix = true) {
    $str = preg_replace("/<([a-zA-Z]+)[^>]*>/", "", $str);
    if (function_exists("mb_substr")) {
        if ($suffix && ceil(strlen($str) / 2) > $length)
            return mb_substr($str, $start, $length, $charset) . "***";
        else
            return mb_substr($str, $start, $length, $charset);
    } elseif (function_exists('iconv_substr')) {
        if ($suffix && ceil(strlen($str) / 2) > $length)
            return iconv_substr($str, $start, $length, $charset) . "***";
        else
            return iconv_substr($str, $start, $length, $charset);
    }
    $re ['utf-8'] = "/[\x01-\x7f]|[\xc2-\xdf][\x80-\xbf]|[\xe0-\xef][\x80-\xbf]{2}|[\xf0-\xff][\x80-\xbf]{3}/";
    $re ['gb2312'] = "/[\x01-\x7f]|[\xb0-\xf7][\xa0-\xfe]/";
    $re ['gbk'] = "/[\x01-\x7f]|[\x81-\xfe][\x40-\xfe]/";
    $re ['big5'] = "/[\x01-\x7f]|[\x81-\xfe]([\x40-\x7e]|\xa1-\xfe])/";
    preg_match_all($re [$charset], $str, $match);
    $slice = join("", array_slice($match [0], $start, $length));
    if ($suffix) {
        return $slice . "***";
    } else {
        return $slice;
    }
}

KAng

Website https://www.lemont.cn

发表回复取消回复

Back To Top

鄂ICP备17008157号-1