Re: [PATCH] libcpp: Implement C++29 P3658R1 - Adjust identifier following new Unicode recommendations

Jason Merrill <[email protected]> Wed, 5 Aug 2026 18:18:16 -0400
Newsgroups gmane.comp.gcc.patches
Message-ID <[email protected]>
On 6/11/26 7:29 AM, Jakub Jelinek wrote:
> Hi!
> 
> The following patch attempts to implement the C++29
> P3658R1 - Adjust identifier following new Unicode recommendations
> paper as a DR (for all C++ versions back to C++98, like P1949R7
> has been applied as well).
> Now, unfortunately C23/C2Y use xid_identifier as well and I'm not
> aware of a similar paper and even if there is one later on for C2Y,
> not sure if it would be voted in as a DR against C23.
> So, this patch adds a bit to differentiate XID_Start+XID_Continue
> or XID_Continue from ID_Compat_Math_Start+ID_Compat_Math_Continue
> or ID_Compat_Math_Continue and makes C++ 98-26 accept union of
> those (pedantic and non-pedantic), while for C23/C2Y it accepts
> the union of those only for non-pedantic and for pedantic it still
> rejects those.
> I've included a patch against the generated ucnid.h as attachment
> because otherwise the patch was too large for mailing list limits.
> 
> Ok for trunk if this is voted into C++29 as a DR?

OK.
> 2026-06-11  Jakub Jelinek  <[email protected]>
> 
> contrib/
> 	* unicode/README: Add another argument to makeucnid
> 	invocation.
> gcc/testsuite/
> 	* g++.dg/cpp23/ucnid-3-utf8.C: New test.
> 	* g++.dg/cpp23/ucnid-4-utf8.C: New test.
> 	* gcc.dg/cpp/ucnid-16-utf8.c: New test.
> libcpp/
> 	* makeucnid.cc: Implement C++29 P3658R1 - Adjust identifier
> 	following new Unicode recommendations as a DR.  Adjust invocation
> 	in the toplevel comment.
> 	(NONC): Add another enumerator to the unnamed enum, or it into
> 	all_languages, renumber the rest of enumerators.
> 	(read_proplist): New function.
> 	(write_table): Print also NONC bit.
> 	(main): Expect 6 arguments rather than 5.  Call read_proplist.
> 	* charset.cc (NONC): Add another enumerator to the unnamed enum.
> 	(cpp_check_xid_property): Return 0 if NONC bit is set.
> 	(ucn_valid_in_identifier): If pedantic xid_identifiers in C,
> 	return 0 if NONC bit is set.
> 	* ucnid.h: Regenerate.
> 
> --- contrib/unicode/README.jj	2026-03-27 10:17:13.240345328 +0100
> +++ contrib/unicode/README	2026-06-11 11:13:39.719124244 +0200
> @@ -66,7 +66,7 @@ The procedure to update GCC's Unicode su
>   6.  Generate ucnid.h as follows:
>         ../../libcpp/makeucnid ../../libcpp/ucnid.tab UnicodeData.txt \
>   	DerivedNormalizationProps.txt DerivedCoreProperties.txt \
> -	> ../../libcpp/ucnid.h
> +	PropList.txt > ../../libcpp/ucnid.h
>   
>   7.  Read the corresponding Unicode's standard and update correspondingly
>       generated_ranges table in libcpp/makeuname2c.cc (in Unicode 17 all
> --- gcc/testsuite/g++.dg/cpp23/ucnid-3-utf8.C.jj	2026-06-11 12:30:57.706481489 +0200
> +++ gcc/testsuite/g++.dg/cpp23/ucnid-3-utf8.C	2026-06-11 12:44:27.465653602 +0200
> @@ -0,0 +1,23 @@
> +// P3658R1
> +// { dg-do compile }
> +// { dg-options "-pedantic-errors" }
> +
> +int ∂²³¹⁰⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int ∇₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int ∞²³¹⁰⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝛁₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝛛²³¹⁰⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝛻₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝜕²³¹⁰⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝜵₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝝏²³¹⁰⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝝯₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝞉²³¹⁰⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝞩₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +int 𝟃²³¹⁰⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾∂∇∞𝛁𝛛𝛻𝜕𝜵𝝏𝝯𝞉𝞩𝟃 = 42;
> +
> +int
> +foo (int j, int k, int xₖ₊₁, int xₖ₋₍ⱼ₊₂₎)
> +{
> +  return xₖ₊₁+xₖ₋₍ⱼ₊₂₎+(j-k);
> +}
> --- gcc/testsuite/g++.dg/cpp23/ucnid-4-utf8.C.jj	2026-06-11 12:34:11.968868366 +0200
> +++ gcc/testsuite/g++.dg/cpp23/ucnid-4-utf8.C	2026-06-11 12:40:19.551954960 +0200
> @@ -0,0 +1,34 @@
> +// P3658R1
> +// { dg-do compile }
> +// { dg-options "-pedantic-errors" }
> +
> +int ²x = 42;	// { dg-error "extended character ² is not valid at the start of an identifier" }
> +int ³x = 42;	// { dg-error "extended character ³ is not valid at the start of an identifier" }
> +int ¹x = 42;	// { dg-error "extended character ¹ is not valid at the start of an identifier" }
> +int ⁰x = 42;	// { dg-error "extended character ⁰ is not valid at the start of an identifier" }
> +int ⁴x = 42;	// { dg-error "extended character ⁴ is not valid at the start of an identifier" }
> +int ⁵x = 42;	// { dg-error "extended character ⁵ is not valid at the start of an identifier" }
> +int ⁶x = 42;	// { dg-error "extended character ⁶ is not valid at the start of an identifier" }
> +int ⁷x = 42;	// { dg-error "extended character ⁷ is not valid at the start of an identifier" }
> +int ⁸x = 42;	// { dg-error "extended character ⁸ is not valid at the start of an identifier" }
> +int ⁹x = 42;	// { dg-error "extended character ⁹ is not valid at the start of an identifier" }
> +int ⁺x = 42;	// { dg-error "extended character ⁺ is not valid at the start of an identifier" }
> +int ⁻x = 42;	// { dg-error "extended character ⁻ is not valid at the start of an identifier" }
> +int ⁼x = 42;	// { dg-error "extended character ⁼ is not valid at the start of an identifier" }
> +int ⁽x = 42;	// { dg-error "extended character ⁽ is not valid at the start of an identifier" }
> +int ⁾x = 42;	// { dg-error "extended character ⁾ is not valid at the start of an identifier" }
> +int ₀x = 42;	// { dg-error "extended character ₀ is not valid at the start of an identifier" }
> +int ₁x = 42;	// { dg-error "extended character ₁ is not valid at the start of an identifier" }
> +int ₂x = 42;	// { dg-error "extended character ₂ is not valid at the start of an identifier" }
> +int ₃x = 42;	// { dg-error "extended character ₃ is not valid at the start of an identifier" }
> +int ₄x = 42;	// { dg-error "extended character ₄ is not valid at the start of an identifier" }
> +int ₅x = 42;	// { dg-error "extended character ₅ is not valid at the start of an identifier" }
> +int ₆x = 42;	// { dg-error "extended character ₆ is not valid at the start of an identifier" }
> +int ₇x = 42;	// { dg-error "extended character ₇ is not valid at the start of an identifier" }
> +int ₈x = 42;	// { dg-error "extended character ₈ is not valid at the start of an identifier" }
> +int ₉x = 42;	// { dg-error "extended character ₉ is not valid at the start of an identifier" }
> +int ₊x = 42;	// { dg-error "extended character ₊ is not valid at the start of an identifier" }
> +int ₋x = 42;	// { dg-error "extended character ₋ is not valid at the start of an identifier" }
> +int ₌x = 42;	// { dg-error "extended character ₌ is not valid at the start of an identifier" }
> +int ₍x = 42;	// { dg-error "extended character ₍ is not valid at the start of an identifier" }
> +int ₎x = 42;	// { dg-error "extended character ₎ is not valid at the start of an identifier" }
> --- gcc/testsuite/gcc.dg/cpp/ucnid-16-utf8.c.jj	2026-06-11 12:47:37.061128846 +0200
> +++ gcc/testsuite/gcc.dg/cpp/ucnid-16-utf8.c	2026-06-11 12:56:26.016085390 +0200
> @@ -0,0 +1,90 @@
> +/* C++29 P3658R1 */
> +/* { dg-do compile } */
> +/* { dg-options "-std=c23 -pedantic-errors" } */
> +
> +int ² = 42;	/* { dg-error "stray|expected" } */
> +int ³ = 42;	/* { dg-error "stray|expected" } */
> +int ¹ = 42;	/* { dg-error "stray|expected" } */
> +int ⁰ = 42;	/* { dg-error "stray|expected" } */
> +int ⁴ = 42;	/* { dg-error "stray|expected" } */
> +int ⁵ = 42;	/* { dg-error "stray|expected" } */
> +int ⁶ = 42;	/* { dg-error "stray|expected" } */
> +int ⁷ = 42;	/* { dg-error "stray|expected" } */
> +int ⁸ = 42;	/* { dg-error "stray|expected" } */
> +int ⁹ = 42;	/* { dg-error "stray|expected" } */
> +int ⁺ = 42;	/* { dg-error "stray|expected" } */
> +int ⁻ = 42;	/* { dg-error "stray|expected" } */
> +int ⁼ = 42;	/* { dg-error "stray|expected" } */
> +int ⁽ = 42;	/* { dg-error "stray|expected" } */
> +int ⁾ = 42;	/* { dg-error "stray|expected" } */
> +int ₀ = 42;	/* { dg-error "stray|expected" } */
> +int ₁ = 42;	/* { dg-error "stray|expected" } */
> +int ₂ = 42;	/* { dg-error "stray|expected" } */
> +int ₃ = 42;	/* { dg-error "stray|expected" } */
> +int ₄ = 42;	/* { dg-error "stray|expected" } */
> +int ₅ = 42;	/* { dg-error "stray|expected" } */
> +int ₆ = 42;	/* { dg-error "stray|expected" } */
> +int ₇ = 42;	/* { dg-error "stray|expected" } */
> +int ₈ = 42;	/* { dg-error "stray|expected" } */
> +int ₉ = 42;	/* { dg-error "stray|expected" } */
> +int ₊ = 42;	/* { dg-error "stray|expected" } */
> +int ₋ = 42;	/* { dg-error "stray|expected" } */
> +int ₌ = 42;	/* { dg-error "stray|expected" } */
> +int ₍ = 42;	/* { dg-error "stray|expected" } */
> +int ₎ = 42;	/* { dg-error "stray|expected" } */
> +int ∂ = 42;	/* { dg-error "stray|expected" } */
> +int ∇ = 42;	/* { dg-error "stray|expected" } */
> +int ∞ = 42;	/* { dg-error "stray|expected" } */
> +int 𝛁 = 42;	/* { dg-error "stray|expected" } */
> +int 𝛛 = 42;	/* { dg-error "stray|expected" } */
> +int 𝛻 = 42;	/* { dg-error "stray|expected" } */
> +int 𝜕 = 42;	/* { dg-error "stray|expected" } */
> +int 𝜵 = 42;	/* { dg-error "stray|expected" } */
> +int 𝝏 = 42;	/* { dg-error "stray|expected" } */
> +int 𝝯 = 42;	/* { dg-error "stray|expected" } */
> +int 𝞉 = 42;	/* { dg-error "stray|expected" } */
> +int 𝞩 = 42;	/* { dg-error "stray|expected" } */
> +int 𝟃 = 42;	/* { dg-error "stray|expected" } */
> +int a² = 42;	/* { dg-error "stray|expected" } */
> +int b³ = 42;	/* { dg-error "stray|expected" } */
> +int c¹ = 42;	/* { dg-error "stray|expected" } */
> +int d⁰ = 42;	/* { dg-error "stray|expected" } */
> +int e⁴ = 42;	/* { dg-error "stray|expected" } */
> +int f⁵ = 42;	/* { dg-error "stray|expected" } */
> +int g⁶ = 42;	/* { dg-error "stray|expected" } */
> +int h⁷ = 42;	/* { dg-error "stray|expected" } */
> +int i⁸ = 42;	/* { dg-error "stray|expected" } */
> +int j⁹ = 42;	/* { dg-error "stray|expected" } */
> +int k⁺ = 42;	/* { dg-error "stray|expected" } */
> +int l⁻ = 42;	/* { dg-error "stray|expected" } */
> +int m⁼ = 42;	/* { dg-error "stray|expected" } */
> +int n⁽ = 42;	/* { dg-error "stray|expected" } */
> +int o⁾ = 42;	/* { dg-error "stray|expected" } */
> +int p₀ = 42;	/* { dg-error "stray|expected" } */
> +int q₁ = 42;	/* { dg-error "stray|expected" } */
> +int r₂ = 42;	/* { dg-error "stray|expected" } */
> +int s₃ = 42;	/* { dg-error "stray|expected" } */
> +int t₄ = 42;	/* { dg-error "stray|expected" } */
> +int u₅ = 42;	/* { dg-error "stray|expected" } */
> +int v₆ = 42;	/* { dg-error "stray|expected" } */
> +int w₇ = 42;	/* { dg-error "stray|expected" } */
> +int x₈ = 42;	/* { dg-error "stray|expected" } */
> +int y₉ = 42;	/* { dg-error "stray|expected" } */
> +int z₊ = 42;	/* { dg-error "stray|expected" } */
> +int A₋ = 42;	/* { dg-error "stray|expected" } */
> +int B₌ = 42;	/* { dg-error "stray|expected" } */
> +int C₍ = 42;	/* { dg-error "stray|expected" } */
> +int D₎ = 42;	/* { dg-error "stray|expected" } */
> +int E∂ = 42;	/* { dg-error "stray|expected" } */
> +int F∇ = 42;	/* { dg-error "stray|expected" } */
> +int G∞ = 42;	/* { dg-error "stray|expected" } */
> +int H𝛁 = 42;	/* { dg-error "stray|expected" } */
> +int I𝛛 = 42;	/* { dg-error "stray|expected" } */
> +int J𝛻 = 42;	/* { dg-error "stray|expected" } */
> +int K𝜕 = 42;	/* { dg-error "stray|expected" } */
> +int L𝜵 = 42;	/* { dg-error "stray|expected" } */
> +int M𝝏 = 42;	/* { dg-error "stray|expected" } */
> +int N𝝯 = 42;	/* { dg-error "stray|expected" } */
> +int O𝞉 = 42;	/* { dg-error "stray|expected" } */
> +int P𝞩 = 42;	/* { dg-error "stray|expected" } */
> +int Q𝟃 = 42;	/* { dg-error "stray|expected" } */
> --- libcpp/makeucnid.cc.jj	2026-03-27 10:17:21.571209381 +0100
> +++ libcpp/makeucnid.cc	2026-06-11 12:07:03.341788737 +0200
> @@ -17,7 +17,7 @@ along with this program; see the file CO
>   
>   /* Run this program as
>      ./makeucnid ucnid.tab UnicodeData.txt DerivedNormalizationProps.txt \
> -      DerivedCoreProperties.txt > ucnid.h
> +      DerivedCoreProperties.txt PropList.txt > ucnid.h
>   */
>   
>   #include <stdio.h>
> @@ -34,10 +34,11 @@ enum {
>     N11 = 16,
>     CXX23 = 32,
>     NXX23 = 64,
> -  all_languages = C99 | CXX | C11 | CXX23 | NXX23,
> -  not_NFC = 128,
> -  not_NFKC = 256,
> -  maybe_not_NFC = 512
> +  NONC = 128,
> +  all_languages = C99 | CXX | C11 | CXX23 | NXX23 | NONC,
> +  not_NFC = 256,
> +  not_NFKC = 512,
> +  maybe_not_NFC = 1024
>   };
>   
>   #define NUM_CODE_POINTS 0x110000
> @@ -311,6 +312,75 @@ read_derivedcore (char *fname)
>     fclose (f);
>   }
>   
> +/* Read PropList.txt and fill in languages version in
> +   flags from the ID_Compat_Math_Start and ID_Compat_Math_Continue
> +   properties.  */
> +
> +static void
> +read_proplist (char *fname)
> +{
> +  FILE * f = fopen (fname, "r");
> +
> +  if (!f)
> +    fail ("opening PropList.txt");
> +  for (;;)
> +    {
> +      char line[256];
> +      unsigned long codepoint_start, codepoint_end;
> +      char *l;
> +      int i, j;
> +
> +      if (!fgets (line, sizeof (line), f))
> +	break;
> +      if (line[0] == '#' || line[0] == '\n' || line[0] == '\r')
> +	continue;
> +      codepoint_start = strtoul (line, &l, 16);
> +      if (l == line)
> +	fail ("parsing PropList.txt, reading code point");
> +      if (codepoint_start > MAX_CODE_POINT)
> +	fail ("parsing PropList.txt, code point too large");
> +
> +      if (*l == '.' && l[1] == '.')
> +	{
> +	  char *l2 = l + 2;
> +	  codepoint_end = strtoul (l + 2, &l, 16);
> +	  if (l == l2 || codepoint_end < codepoint_start)
> +	    fail ("parsing PropList.txt, reading code point");
> +	  if (codepoint_end > MAX_CODE_POINT)
> +	    fail ("parsing PropList.txt, code point too large");
> +	}
> +      else
> +	codepoint_end = codepoint_start;
> +
> +      while (*l == ' ')
> +	l++;
> +      if (*l++ != ';')
> +	fail ("parsing PropList.txt, reading code point");
> +
> +      while (*l == ' ')
> +	l++;
> +
> +      if (codepoint_end < 0x80)
> +	continue;
> +
> +      if (strncmp (l, "ID_Compat_Math_Start ", 21) == 0)
> +	{
> +	  for (; codepoint_start <= codepoint_end; codepoint_start++)
> +	    flags[codepoint_start]
> +	      = (flags[codepoint_start] | CXX23 | NONC) & ~NXX23;
> +	}
> +      else if (strncmp (l, "ID_Compat_Math_Continue ", 24) == 0)
> +	{
> +	  for (; codepoint_start <= codepoint_end; codepoint_start++)
> +	    if ((flags[codepoint_start] & CXX23) == 0)
> +	      flags[codepoint_start] |= CXX23 | NXX23 | NONC;
> +	}
> +    }
> +  if (ferror (f))
> +    fail ("reading PropList.txt");
> +  fclose (f);
> +}
> +
>   /* Write out the table.
>      The table consists of two words per entry.  The first word is the flags
>      for the unicode code points up to and including the second word.  */
> @@ -331,7 +401,7 @@ write_table (void)
>   	|| really_safe != (decomp[i][0] == 0)
>   	|| combining_value[i] != last_combine)
>         {
> -	printf ("{ %s|%s|%s|%s|%s|%s|%s|%s|%s|%s|%s, %3d, %#06x },\n",
> +	printf ("{ %s|%s|%s|%s|%s|%s|%s|%s|%s|%s|%s|%s, %3d, %#06x },\n",
>   		last_flag & C99 ? "C99" : "  0",
>   		last_flag & N99 ? "N99" : "  0",
>   		last_flag & CXX ? "CXX" : "  0",
> @@ -343,6 +413,7 @@ write_table (void)
>   		last_flag & not_NFC ? "  0" : "NFC",
>   		last_flag & not_NFKC ? "  0" : "NKC",
>   		last_flag & maybe_not_NFC ? "CTX" : "  0",
> +		last_flag & NONC ? "NONC" : "   0",
>   		combining_value[i - 1],
>   		i - 1);
>   	last_flag = flags[i];
> @@ -511,12 +582,13 @@ write_copyright (void)
>   int
>   main(int argc, char ** argv)
>   {
> -  if (argc != 5)
> +  if (argc != 6)
>       fail ("too few arguments to makeucn");
>     read_ucnid (argv[1]);
>     read_table (argv[2]);
>     read_derived (argv[3]);
>     read_derivedcore (argv[4]);
> +  read_proplist (argv[5]);
>   
>     write_copyright ();
>     write_table ();
> --- libcpp/charset.cc.jj	2026-04-21 18:24:44.122033404 +0200
> +++ libcpp/charset.cc	2026-06-11 11:50:29.814178866 +0200
> @@ -985,7 +985,9 @@ enum {
>     /* Might be valid NFKC form?  */
>     NKC = 512,
>     /* Certain preceding characters might make it not valid NFC/NKFC form?  */
> -  CTX = 1024
> +  CTX = 1024,
> +  /* Valid in C++23 but not in C23.  Set only when CXX23 is also set.  */
> +  NONC = 2048
>   };
>   
>   struct ucnrange {
> @@ -1366,6 +1368,8 @@ cpp_check_xid_property (cppchar_t c)
>   
>     unsigned short flags = ucnranges[mn].flags;
>   
> +  if (flags & NONC)
> +    return 0;
>     if (flags & CXX23)
>       return CPP_XID_START | CPP_XID_CONTINUE;
>     if (flags & NXX23)
> @@ -1408,14 +1412,19 @@ ucn_valid_in_identifier (cpp_reader *pfi
>     if (CPP_PEDANTIC (pfile))
>       {
>         if (CPP_OPTION (pfile, xid_identifiers))
> -	valid_flags = CXX23;
> +	{
> +	  valid_flags = CXX23;
> +	  if (!CPP_OPTION (pfile, cplusplus)
> +	      && (ucnranges[mn].flags & NONC))
> +	    return 0;
> +	}
>         else if (CPP_OPTION (pfile, c11_identifiers))
>   	valid_flags = C11;
>         else if (CPP_OPTION (pfile, c99))
>   	valid_flags = C99;
>       }
>     if (! (ucnranges[mn].flags & valid_flags))
> -      return 0;
> +    return 0;
>   
>     /* Update NST.  */
>     if (ucnranges[mn].combine != 0 && ucnranges[mn].combine < nst->prev_class)
> 
> 	Jakub