Skip to content

Commit 5c41a2d

Browse files
authored
Improve error diagnostics (#31)
* Improve error diagnostics * Don't test fragile debug output * Improve error diagnostics * Remove utf8 handling * Print string directly * Update iris * Redesign `default_error_handler`'s `::on_trace(...)` * Redesign `default_error_handler`'s `::on_expectation_failure(...)` * Refactor * Add more `get_x4_info()` * Add `x4::parse_debug(...)` * Split Unicode related headers * Remove shadowing template parameter * Update iris * Fix dangling reference * Avoid global variables * Add unicode.hpp to pch
1 parent ef0c4cb commit 5c41a2d

41 files changed

Lines changed: 1245 additions & 851 deletions

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎include/iris/x4/ast/position_tagged.hpp‎

Lines changed: 0 additions & 104 deletions
This file was deleted.

‎include/iris/x4/char/char_set.hpp‎

Lines changed: 16 additions & 18 deletions
Original file line numberDiff line numberDiff line change
@@ -13,9 +13,11 @@
1313
#include <iris/x4/char/char_parser.hpp>
1414
#include <iris/x4/char/detail/cast_char.hpp>
1515
#include <iris/x4/char/detail/basic_chset.hpp>
16-
#include <iris/x4/string/utf8.hpp>
1716
#include <iris/x4/string/case_compare.hpp>
1817

18+
#include <iris/unicode/string.hpp>
19+
20+
#include <format>
1921
#include <ranges>
2022
#include <type_traits>
2123

@@ -47,6 +49,17 @@ struct char_range : char_parser<Encoding, char_range<Encoding, Attr>>
4749
}
4850

4951
char_type from, to;
52+
53+
[[nodiscard]] std::string get_x4_info() const
54+
{
55+
// TODO: make more user-friendly && make the format consistent with above
56+
// TODO: escape
57+
return std::format(
58+
"char_range \"{}-{}\"",
59+
iris::unicode::transcode<char>(typename Encoding::string_type(1, this->from)),
60+
iris::unicode::transcode<char>(typename Encoding::string_type(1, this->to))
61+
);
62+
}
5063
};
5164

5265
// Parser for a character set
@@ -103,29 +116,14 @@ struct char_set : char_parser<Encoding, char_set<Encoding, Attr>>
103116
}
104117

105118
detail::basic_chset<char_type> chset;
106-
};
107119

108-
template<class Encoding, X4Attribute Attr>
109-
struct get_info<char_set<Encoding, Attr>>
110-
{
111-
using result_type = std::string;
112-
[[nodiscard]] constexpr std::string operator()(char_set<Encoding, Attr> const& /* p */) const
120+
[[nodiscard]] std::string get_x4_info() const
113121
{
122+
// TODO: escape
114123
return "char-set"; // TODO: make more user-friendly
115124
}
116125
};
117126

118-
template<class Encoding, X4Attribute Attr>
119-
struct get_info<char_range<Encoding, Attr>>
120-
{
121-
using result_type = std::string;
122-
[[nodiscard]] constexpr std::string operator()(char_range<Encoding, Attr> const& p) const
123-
{
124-
// TODO: make more user-friendly && make the format consistent with above
125-
return "char_range \"" + x4::to_utf8(Encoding::toucs4(p.from)) + '-' + x4::to_utf8(Encoding::toucs4(p.to))+ '"';
126-
}
127-
};
128-
129127
} // iris::x4
130128

131129
#endif

‎include/iris/x4/char/literal_char.hpp‎

Lines changed: 12 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -11,9 +11,11 @@
1111
==============================================================================*/
1212

1313
#include <iris/x4/char/char_parser.hpp>
14-
#include <iris/x4/string/utf8.hpp>
1514
#include <iris/x4/string/case_compare.hpp>
1615

16+
#include <iris/unicode/string.hpp>
17+
18+
#include <format>
1719
#include <type_traits>
1820
#include <concepts>
1921

@@ -51,18 +53,17 @@ struct literal_char : char_parser<Encoding, literal_char<Encoding, Attr>>
5153

5254
[[nodiscard]] constexpr classify_type classify_ch() const noexcept { return classify_ch_; }
5355

54-
private:
55-
classify_type classify_ch_{};
56-
};
57-
58-
template<class Encoding, X4Attribute Attr>
59-
struct get_info<literal_char<Encoding, Attr>>
60-
{
61-
using result_type = std::string;
62-
[[nodiscard]] std::string operator()(literal_char<Encoding, Attr> const& p) const
56+
[[nodiscard]] std::string get_x4_info() const
6357
{
64-
return '\'' + x4::to_utf8(Encoding::toucs4(p.classify_ch())) + '\'';
58+
// TODO: escape quote
59+
return std::format(
60+
"'{}'",
61+
iris::unicode::transcode<char>(typename Encoding::string_type(1, this->classify_ch_))
62+
);
6563
}
64+
65+
private:
66+
classify_type classify_ch_{};
6667
};
6768

6869
} // iris::x4

‎include/iris/x4/char_encoding/standard.hpp‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -11,6 +11,8 @@
1111
file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt)
1212
=============================================================================*/
1313

14+
#include <string>
15+
1416
#include <cassert>
1517
#include <cstdint>
1618
#include <cctype>
@@ -22,6 +24,7 @@ namespace iris::x4::char_encoding {
2224
struct standard
2325
{
2426
using char_type = char;
27+
using string_type = std::string;
2528
using classify_type = unsigned char;
2629

2730
[[nodiscard]] static constexpr bool

‎include/iris/x4/char_encoding/standard_wide.hpp‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -24,6 +24,7 @@ namespace iris::x4::char_encoding {
2424
struct standard_wide
2525
{
2626
using char_type = wchar_t;
27+
using string_type = std::wstring;
2728
using classify_type = wchar_t;
2829

2930
template<class Char>

‎include/iris/x4/char_encoding/unicode.hpp‎

Lines changed: 6 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -11,7 +11,11 @@
1111
file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt)
1212
=============================================================================*/
1313

14-
#include <iris/x4/char_encoding/unicode/classification.hpp>
14+
#include <iris/x4/char_encoding/unicode/classify_category.hpp>
15+
#include <iris/x4/char_encoding/unicode/classify_script.hpp>
16+
#include <iris/x4/char_encoding/unicode/classify_case.hpp>
17+
18+
#include <string>
1519

1620
#include <cstdint>
1721

@@ -20,6 +24,7 @@ namespace iris::x4::char_encoding {
2024
struct unicode
2125
{
2226
using char_type = char32_t;
27+
using string_type = std::u32string;
2328
using classify_type = x4::unicode::classify_type;
2429

2530
[[nodiscard]] static constexpr bool

include/iris/x4/char_encoding/unicode/classification.hpp renamed to include/iris/x4/char_encoding/unicode/category.hpp

Lines changed: 2 additions & 121 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,5 @@
1-
#ifndef IRIS_ZZ_X4_CHAR_ENCODING_UNICODE_CLASSIFICATION_HPP
2-
#define IRIS_ZZ_X4_CHAR_ENCODING_UNICODE_CLASSIFICATION_HPP
1+
#ifndef IRIS_ZZ_X4_CHAR_ENCODING_UNICODE_CATEGORY_HPP
2+
#define IRIS_ZZ_X4_CHAR_ENCODING_UNICODE_CATEGORY_HPP
33

44
/*=============================================================================
55
Copyright (c) 2001-2011 Joel de Guzman
@@ -13,11 +13,6 @@
1313
table builder) (c) Peter Kankowski, 2008
1414
==============================================================================*/
1515

16-
#include <iris/x4/char_encoding/unicode/detail/category_table.hpp>
17-
#include <iris/x4/char_encoding/unicode/detail/script_table.hpp>
18-
#include <iris/x4/char_encoding/unicode/detail/lowercase_table.hpp>
19-
#include <iris/x4/char_encoding/unicode/detail/uppercase_table.hpp>
20-
2116
#include <cstdint>
2217

2318
namespace iris::x4::unicode {
@@ -266,120 +261,6 @@ enum script
266261

267262
} // properties
268263

269-
[[nodiscard]] constexpr properties::category get_category(classify_type ch) noexcept
270-
{
271-
return static_cast<properties::category>(detail::category_lookup(ch) & 0x3F);
272-
}
273-
274-
[[nodiscard]] constexpr properties::major_category get_major_category(classify_type ch) noexcept
275-
{
276-
return static_cast<properties::major_category>(unicode::get_category(ch) >> 3);
277-
}
278-
279-
[[nodiscard]] constexpr bool is_punctuation(classify_type ch) noexcept
280-
{
281-
return unicode::get_major_category(ch) == properties::punctuation;
282-
}
283-
284-
[[nodiscard]] constexpr bool is_decimal_number(classify_type ch) noexcept
285-
{
286-
return unicode::get_category(ch) == properties::decimal_number;
287-
}
288-
289-
[[nodiscard]] constexpr bool is_hex_digit(classify_type ch) noexcept
290-
{
291-
return (detail::category_lookup(ch) & properties::hex_digit) != 0;
292-
}
293-
294-
[[nodiscard]] constexpr bool is_control(classify_type ch) noexcept
295-
{
296-
return unicode::get_category(ch) == properties::control;
297-
}
298-
299-
[[nodiscard]] constexpr bool is_alphabetic(classify_type ch) noexcept
300-
{
301-
return (detail::category_lookup(ch) & properties::alphabetic) != 0;
302-
}
303-
304-
[[nodiscard]] constexpr bool is_alphanumeric(classify_type ch) noexcept
305-
{
306-
return unicode::is_decimal_number(ch) || unicode::is_alphabetic(ch);
307-
}
308-
309-
[[nodiscard]] constexpr bool is_uppercase(classify_type ch) noexcept
310-
{
311-
return (detail::category_lookup(ch) & properties::uppercase) != 0;
312-
}
313-
314-
[[nodiscard]] constexpr bool is_lowercase(classify_type ch) noexcept
315-
{
316-
return (detail::category_lookup(ch) & properties::lowercase) != 0;
317-
}
318-
319-
[[nodiscard]] constexpr bool is_white_space(classify_type ch) noexcept
320-
{
321-
return (detail::category_lookup(ch) & properties::white_space) != 0;
322-
}
323-
324-
[[nodiscard]] constexpr bool is_blank(classify_type ch) noexcept
325-
{
326-
switch (ch)
327-
{
328-
case '\n': case '\v': case '\f': case '\r':
329-
return false;
330-
default:
331-
return unicode::is_white_space(ch) &&
332-
!(
333-
unicode::get_category(ch) == properties::line_separator ||
334-
unicode::get_category(ch) == properties::paragraph_separator
335-
);
336-
}
337-
}
338-
339-
[[nodiscard]] constexpr bool is_graph(classify_type ch) noexcept
340-
{
341-
return !(
342-
unicode::is_white_space(ch) ||
343-
unicode::get_category(ch) == properties::control ||
344-
unicode::get_category(ch) == properties::surrogate ||
345-
unicode::get_category(ch) == properties::unassigned
346-
);
347-
}
348-
349-
[[nodiscard]] constexpr bool is_print(classify_type ch) noexcept
350-
{
351-
return (unicode::is_graph(ch) || unicode::is_blank(ch)) && !unicode::is_control(ch);
352-
}
353-
354-
[[nodiscard]] constexpr bool is_noncharacter_code_point(classify_type ch) noexcept
355-
{
356-
return (detail::category_lookup(ch) & properties::noncharacter_code_point) != 0;
357-
}
358-
359-
[[nodiscard]] constexpr bool is_default_ignorable_code_point(classify_type ch) noexcept
360-
{
361-
return (detail::category_lookup(ch) & properties::default_ignorable_code_point) != 0;
362-
}
363-
364-
[[nodiscard]] constexpr properties::script get_script(classify_type ch) noexcept
365-
{
366-
return static_cast<properties::script>(detail::script_lookup(ch));
367-
}
368-
369-
[[nodiscard]] constexpr classify_type to_lowercase(classify_type ch) noexcept
370-
{
371-
// The table returns 0 to signal that this code maps to itself
372-
classify_type const r = detail::lowercase_lookup(ch);
373-
return r == 0 ? ch : r;
374-
}
375-
376-
[[nodiscard]] constexpr classify_type to_uppercase(classify_type ch) noexcept
377-
{
378-
// The table returns 0 to signal that this code maps to itself
379-
classify_type const r = detail::uppercase_lookup(ch);
380-
return r == 0 ? ch : r;
381-
}
382-
383264
} // iris::x4::unicode
384265

385266
#endif

0 commit comments

Comments
 (0)