01571: provide a UTF8 PageListSortCmpFunction in xlpage-utf-8.php

Summary: provide a UTF8 PageListSortCmpFunction in xlpage-utf-8.php
Created: 2026-10-03 02:20
Status: Open
Category: Suggestion
From: simon
Assigned:
Priority: 1
Version: latest
OS:

Description: This is a suggestion for an addition of xlpage-utf-8.php. along the lines of

function pmwiki_strcmpUTF8(string $a, string $b): int {
# accent and case insensitive UTF8 pageline string comparison
# returns the same values as strcmp()
    static $collator = NULL;
    if ($collator === NULL && \class_exists('Collator')) {
        # read system locale (e.g., "en_NZ.UTF-8")
        $sysLocale = \setlocale(LC_COLLATE, "0");
        # strip encoding suffixes so ICU can understand it
        $icuLocale = "";
        if (\is_string($sysLocale)) {
            $icuLocale = \preg_replace('/\..*$/', '', $sysLocale);
        }
        # try to create a Collator with the detected locale
        if ($icuLocale !== "" && $icuLocale !== "C") {
            $test = @new \Collator($icuLocale);
            if ($test !== NULL) {
                $collator = $test;
            }
        }
        # fallback locale if system locale is unusable
        if ($collator === NULL) {
            $collator = new \Collator('en_UK');
        }
        # PRIMARY strength = ignores both case and accents/diacritics
        $collator->setAttribute(\Collator::STRENGTH, \Collator::PRIMARY);
    }
    # Collator comparison
    if ($collator !== NULL) {
        return $collator->compare($a, $b);
    }
    # Fallback: decompose UTF-8 diacritics + case fold
    $aFold = \pmwiki_stripAccentsAndLowercase($a);
    $bFold = \pmwiki_stripAccentsAndLowercase($b);
    return \strcmp($aFold, $bFold);
} # end function pmwiki_strcmpUTF8
#
function pmwiki_stripAccentsAndLowercase(string $str): string {
    if (\class_exists('Normalizer')) {
        # Decompose characters into base letters + combining marks (NFD)
        $decomposed = \Normalizer::normalize($str, \Normalizer::FORM_D);
        if ($decomposed !== FALSE) {
            # Strip combining diacritical marks (\p{M})
            $tmp = \preg_replace('/\p{M}/u', '', $decomposed);
            if ($tmp !== FALSE) {
                $str = $tmp;
            }
        }
    } elseif (\function_exists('iconv')) {
        # Fallback diacritic removal
        $converted = @\iconv('UTF-8', 'ASCII//TRANSLIT//IGNORE', $str);
        if ($converted !== FALSE) {
            $str = $converted;
        }
    }
    return \function_exists('mb_strtolower') ? \mb_strtolower($str, 'UTF-8') : \strtolower($str);
} # end function pmwiki_stripAccentsAndLowercase
#
$PageListSortCmpFunction = 'pmwiki_strcmpUTF8';