diff --git a/Source/LibationFileManager/Templates/IListFormat[TList].cs b/Source/LibationFileManager/Templates/IListFormat[TList].cs index ff3c70d5..3aef7d6e 100644 --- a/Source/LibationFileManager/Templates/IListFormat[TList].cs +++ b/Source/LibationFileManager/Templates/IListFormat[TList].cs @@ -11,13 +11,31 @@ internal partial interface IListFormat where TList : IListFormat { static IEnumerable FilteredList(string formatString, IEnumerable items, CultureInfo? culture) where T : IFormattable { - return Max(formatString, Slice(formatString, Unique(formatString, items, culture))); + return Max(formatString, Slice(formatString, Unique(formatString, Filter(formatString, items, culture), culture))); static StringComparer GetStringComparer(CultureInfo? culture) { return StringComparer.Create(culture ?? CultureInfo.CurrentCulture, ignoreCase: true); } + static IEnumerable Filter(string formatString, IEnumerable items, CultureInfo? culture) + { + if (!FilterRegex().TryMatch(formatString, out var filterMatch)) + return items; + + // read the format to apply on each item + var format = filterMatch.ResolveValue("format"); + // use the operator to get a predicate function that compares the formatted item to the value specified in the filter + var predicate = CompareCondition.GetPredicate(filterMatch.Value, filterMatch.ResolveValue("op")); + // the value to compare the formatted item to. Might be a number or a quoted string. + CommonFormatters.TryGetLiteral(filterMatch.ResolveValue("value"), out var value); + + // return only the items that match the predicate + return items.Where(FilterPredicate); + + bool FilterPredicate(T n) => predicate(n.ToString(format, culture), value, culture); + } + static IEnumerable Unique(string formatString, IEnumerable items, CultureInfo? culture) { return UniqueRegex().TryMatch(formatString, out var uniqueMatch) @@ -120,11 +138,36 @@ internal partial interface IListFormat where TList : IListFormat [GeneratedRegex("""[Ss]eparator\((?(?:\\.|'[^']*'|"[^"]*"|[^\\'"])*?)\)""")] private static partial Regex SeparatorRegex(); - /// Count will substitute all list members with a single number equal to there count + /// Count will substitute all list members with a single number equal to their count [GeneratedRegex("""[Cc]ount\((?(?:\\.|'[^']*'|"[^"]*"|[^\\'"])*?)\)""")] private static partial Regex CountRegex(); /// Unique will shrink the list to unique members after applying format to them [GeneratedRegex("""[Uu]nique\((?(?:\\.|'[^']*'|"[^"]*"|[^\\'"])*?)\)""")] private static partial Regex UniqueRegex(); -} + + /// The filter will reduce the list, keeping only the items that match the specified criteria. + [GeneratedRegex(""" + (?x) # option x: ignore all unescaped whitespace in pattern and allow comments starting with # + [Ff]ilter # name of the command 'filter' or 'Filter' + \( # details are enclosed in brackets + (?(?: # the first part captured as specifies how to format items before comparison + \\. # - '\' escapes always the next character. + | '[^']*' # - allow 'string' to be included in the format, with '' being an escaped ' character + | "[^"]*" # - allow "string" to be included in the format, with "" being an escaped " character + | [^\\'"] # - match any other character. This will not catch the operator at first. Because ... + ) *? ) # With *? the pattern above tries not to consume the operator. + \s* # Separate the following operator with whitespace + (? # capture operator in + [\#!≡=≠~<>≤≥&∉∌∈∌⋂⊆⊇⊂⊃-]+ # allow a wide range of operators, all non alphanumeric so that no operator is confused as value + | :[a-z_]+: # allow :named: operators for readability, e.g. :contains: + ) \s* # ignore space between operator and second property + (? # the second operand is captured as and is a quoted string encapsulated in either single or double quotes + '(?:[^']|'')*' # - allow 'string' to be included in the format, with '' being an escaped ' character + | "(?:[^"]|"")*" # - allow "string" to be included in the format, with "" being an escaped " character + | \d+ # - allow a number + ) # + \s* \) # end the filter details with optional whitespace and a closing bracket + """)] + private static partial Regex FilterRegex(); +} \ No newline at end of file diff --git a/Source/_Tests/FileManager.Tests/CommonFormattersTests.cs b/Source/_Tests/FileManager.Tests/CommonFormattersTests.cs index bbfa9642..fb559da6 100644 --- a/Source/_Tests/FileManager.Tests/CommonFormattersTests.cs +++ b/Source/_Tests/FileManager.Tests/CommonFormattersTests.cs @@ -1,7 +1,6 @@ using System; using System.Collections.Generic; using System.Globalization; -using AssertionHelper; using FileManager.NamingTemplate; using Microsoft.VisualStudio.TestTools.UnitTesting; @@ -322,10 +321,33 @@ public class CommonFormattersTests Assert.AreEqual(expected, unescaped); } + [TestMethod] + [DataRow(null, false, null, "null")] + [DataRow("", false, null, "emptystring")] + [DataRow("42", true, 42, "number")] + [DataRow("\"only\" at start", false, null, "partly quoted")] + [DataRow("\"mismatched'", false, null, "wrong pair of quotes")] + [DataRow("\"simple string\"", true, "simple string", "simple quoted with double quotes")] + [DataRow("'simple string'", true, "simple string", "simple quoted with single quotes")] + [DataRow("\"string with \"\"escaped\"\" quotes\"", true, "string with \"escaped\" quotes", "quoted with embedded doubled quotes")] + [DataRow("'string with ''escaped'' quotes'", true, "string with 'escaped' quotes", "quoted with embedded doubled single quotes")] + [DataRow("\"string with 'single' quotes\"", true, "string with 'single' quotes", "quoted with embedded other quote type")] + [DataRow("\"string with ''doubled single'' and \\\"escaped double\\\" quotes\"", true, "string with ''doubled single'' and \\\"escaped double\\\" quotes", "quoted with embedded doubling")] + [DataRow(" \"string with whitespace\" ", true, "string with whitespace", "quoted with whitespace")] + [DataRow("\"\"", true, "", "empty quoted string")] + public void TryQuotedString_Various(string? value, bool expectedSuccess, object? expectedValue, string testDescription) + { + // WHEN + var result = CommonFormatters.TryGetLiteral(value, out var unQuotedValue); + + // THEN + Assert.AreEqual(expectedSuccess, result, $"Failed for: {testDescription}"); + Assert.AreEqual(expectedValue, unQuotedValue, $"Failed for: {testDescription}"); + } private class TestClass { - public string? Author { get; set; } - public string? Title { get; set; } + public string? Author { get; init; } + public string? Title { get; init; } } } \ No newline at end of file diff --git a/Source/_Tests/LibationFileManager.Tests/TemplatesTests.cs b/Source/_Tests/LibationFileManager.Tests/TemplatesTests.cs index dbcaa794..0b661dc6 100644 --- a/Source/_Tests/LibationFileManager.Tests/TemplatesTests.cs +++ b/Source/_Tests/LibationFileManager.Tests/TemplatesTests.cs @@ -400,6 +400,12 @@ namespace TemplatesTests [DataRow("", "Charles E. Gannon, Emma Gannon")] [DataRow("", "Emma Gannon, Charles E. Gannon")] [DataRow("", "Browne, Gannon, Fetherolf, Montgomery, Van Doren")] + [DataRow("", "2")] + [DataRow("", "Browne, Bon Jovi")] // match correct position of operator + [DataRow(@"", "Browne, Bon Jovi")] // allow quoted quotes + [DataRow("", "")] // strings with numerical operators are substituted by their length + [DataRow("", "6")] + [DataRow("", "Fetherolf, Montgomery")] [DataRow("", "7")] [DataRow("", "7")] [DataRow("", "2")] @@ -873,6 +879,7 @@ namespace TemplatesTests [DataRow("", "03")] [DataRow("", "Tag3")] [DataRow("", "1")] + [DataRow("", "Tag2, Tag3")] [DataRow("", "Tag1")] [DataRow("", "Tag2, Tag3")] [DataRow("", "Tag3, Tag2, Tag1")] diff --git a/docs/features/naming-templates.md b/docs/features/naming-templates.md index 726e0f35..2fdd53ab 100644 --- a/docs/features/naming-templates.md +++ b/docs/features/naming-templates.md @@ -131,19 +131,20 @@ Text formatting can change length and case of the text. Use \<#\>, \<#\>\ ### Text List Formatters -| Formatter | Description | Example Usage | Example Result | -|---------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|----------------------------------------------| -| separator() | Specify the text used to join
multiple entries.

Default is ", " | `` | Tag1_Tag2_Tag3_Tag4_Tag5 | -| format(\{S\}) **†** | Formats the entries by placing their values into the specified template.
Use \{S:[Text formatters](#text-formatters)\} to place the entry and optionally apply a format. | ``separator(;)]>` | Tag=tag1;Tag=tag2;Tag=tag3;Tag=tag4;Tag=tag5 | -| unique(FMT) **†** | Reduce list members to a unique set. Entries are compared to each other after applying the given format. Duplicate entries (after format is applied) are removed, keeping the first occurrence. | ``
``separator(;)]>` | Tag1, Tag2, Tag3
tag1 | -| sort(S) | Sorts the elements by their value.

*Sorting direction:*
uppercase = ascending
lowercase = descending

Default is unsorted | ``separator(;)]>` | Tag5;Tag4;Tag3;Tag2;Tag1 | -| max(#) | Only use the first # of entries | `` | Tag1 | -| slice(#) | Only use the nth entry of the list | `` | Tag2 | -| slice(#..) | Only use entries of the list starting from # | `` | Tag2, Tag3, Tag4, Tag5 | -| slice(..#) | Like max(#). Only use the first # of entries | `` | Tag1 | -| slice(#..#) | Only use entries of the list starting from # and ending at # (inclusive) | `` | Tag2, Tag3, Tag4 | -| slice(-#..-#) | Numbers may be specified negative. In that case positions ar counted from the end with -1 pointing at the last member | `` | Tag3, Tag4 | -| count(FMT) **‡** | Instead of returning some or all members of the list, print out the number of entries using the specified [format](#number-formatters). | ``
`` | 5
05 | +| Formatter | Description | Example Usage | Example Result | +|--------------------------------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|----------------------------------------------| +| separator() | Specify the text used to join
multiple entries.

Default is ", " | `` | Tag1_Tag2_Tag3_Tag4_Tag5 | +| format(\{S\}) **†** | Formats the entries by placing their values into the specified template.
Use \{S:[Text formatters](#text-formatters)\} to place the entry and optionally apply a format. | ``separator(;)]>` | Tag=tag1;Tag=tag2;Tag=tag3;Tag=tag4;Tag=tag5 | +| unique(FMT) **†** | Reduce list members to a unique set. Entries are compared to each other after applying the given format. Duplicate entries (after format is applied) are removed, keeping the first occurrence. | ``
``separator(;)]>` | Tag1, Tag2, Tag3
tag1 | +| sort(S) | Sorts the elements by their value.

*Sorting direction:*
uppercase = ascending
lowercase = descending

Default is unsorted | ``separator(;)]>` | Tag5;Tag4;Tag3;Tag2;Tag1 | +| max(#) | Only use the first # of entries | `` | Tag1 | +| slice(#) | Only use the nth entry of the list | `` | Tag2 | +| slice(#..) | Only use entries of the list starting from # | `` | Tag2, Tag3, Tag4, Tag5 | +| slice(..#) | Like max(#). Only use the first # of entries | `` | Tag1 | +| slice(#..#) | Only use entries of the list starting from # and ending at # (inclusive) | `` | Tag2, Tag3, Tag4 | +| slice(-#..-#) | Numbers may be specified negative. In that case positions ar counted from the end with -1 pointing at the last member | `` | Tag3, Tag4 | +| count(FMT) **‡** | Instead of returning some or all members of the list, print out the number of entries using the specified [format](#number-formatters). | ``
`` | 5
05 | +| filter(FMT [CHECK](#checks) VALUE) **†** | Filter list entries based on a condition. Each item is first formatted using the specified text format (or the default format if FMT is omitted), then compared against VALUE using the specified [CHECK](#checks). Only matching entries are included in the output.

**Syntax:** `filter(FORMAT CHECK VALUE)` or `filter(CHECK VALUE)`
- `FORMAT`: Optional text format to apply to each entry (e.g., `{S}`, `{S:L}`, `{S:3}`); defaults to `{S}`
- `CHECK`: Comparison operator (e.g., `=`, `!=`, `~`)
- `VALUE`: The value to compare against | ``
``
`Tag1, Tag2, Tag4, Tag5
TagA, TagB, TagC | **†** For further information on format templates, please refer to the [Format templates](#format-templates) section. @@ -159,15 +160,16 @@ Text formatting can change length and case of the text. Use \<#\>, \<#\>\ ### Series List Formatters -| Formatter | Description | Example Usage | Example Result | -|---------------------------------| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |-------------------------------------------------------------------------------------------| ------------------------------------------------------------------------------------------------------------------- | -| separator() | Specify the text used to join
multiple series names.

Default is ", " | `` | Sherlock Holmes; Some Other Series | -| format(\{N \| # \| ID\}) **†** | Formats the series properties
using the name series tags.
See [Series Formatter Usage](#series-formatters) above. | ``separator(; )]>`
`` | Sherlock Holmes, 1-6; Book Collection, 1
B08376S3R2-Sherlock Holmes, 01.0-06.0, B000000000-Book Collection, 01.0 | -| unique(FMT) **†** | Reduce list members to a unique set. Entries are compared to each other after applying the given format. Duplicate entries (after format is applied) are removed, keeping the first occurrence. | ``
``separator(; )]>` | Sherlock Holmes; Some Other Series
sherlock holmes; some other series | -| sort(N \| # \| ID) | Sorts the series by name, number or ID.

These terms define the primary, secondary, tertiary, … sorting order.
You may combine multiple terms in sequence to specify multi‑level sorting.

*Sorting direction:*
uppercase = ascending
lowercase = descending

Default is unsorted | ``separator(; )]>` | Book Collection, 1; Sherlock Holmes, 1-6 | -| max(#) | Only use the first # of series | `` | Sherlock Holmes | -| slice(#..#) | Only use entries of the series list starting from # and ending at # (inclusive)

See [Text List Formatter Usage](#Text-List-Formatters) above for details on all the variants of `slice()` | `` | Sherlock Holmes | -| count(FMT) **‡** | Instead of returning some or all members of the list, print out the number of series using the specified [format](#number-formatters). | ``
`` | 2
02 | +| Formatter | Description | Example Usage | Example Result | +|-------------------------------------------| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |-------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------| +| separator() | Specify the text used to join
multiple series names.

Default is ", " | `` | Sherlock Holmes; Some Other Series | +| format(\{N \| # \| ID\}) **†** | Formats the series properties
using the name series tags.
See [Series Formatter Usage](#series-formatters) above. | ``separator(; )]>`
`` | Sherlock Holmes, 1-6; Book Collection, 1
B08376S3R2-Sherlock Holmes, 01.0-06.0, B000000000-Book Collection, 01.0 | +| unique(FMT) **†** | Reduce list members to a unique set. Entries are compared to each other after applying the given format. Duplicate entries (after format is applied) are removed, keeping the first occurrence. | ``
``separator(; )]>` | Sherlock Holmes; Some Other Series
sherlock holmes; some other series | +| sort(N \| # \| ID) | Sorts the series by name, number or ID.

These terms define the primary, secondary, tertiary, … sorting order.
You may combine multiple terms in sequence to specify multi‑level sorting.

*Sorting direction:*
uppercase = ascending
lowercase = descending

Default is unsorted | ``separator(; )]>` | Book Collection, 1; Sherlock Holmes, 1-6 | +| max(#) | Only use the first # of series | `` | Sherlock Holmes | +| slice(#..#) | Only use entries of the series list starting from # and ending at # (inclusive)

See [Text List Formatter Usage](#Text-List-Formatters) above for details on all the variants of `slice()` | `` | Sherlock Holmes | +| count(FMT) **‡** | Instead of returning some or all members of the list, print out the number of series using the specified [format](#number-formatters). | ``
`` | 2
02 | +| filter(FMT [CHECK](#checks) VALUE) **†** | Filter list entries based on a condition. Each series is first formatted using the specified [Series Format](#series-formatters) (or the default format if FMT is omitted), then compared against VALUE using the specified [CHECK](#checks). Only matching entries are included in the output.

**Syntax:** `filter(FORMAT CHECK VALUE)` or `filter(CHECK VALUE)`
- `FORMAT`: Optional series format to apply to each entry (e.g., `{N}`, `{N:L}`, `{#}`); defaults to `{N}`
- `CHECK`: Comparison operator (e.g., `=`, `!=`, `~`)
- `VALUE`: The value to compare against | ``
`Sherlock Holmes | **†** For further information on format templates, please refer to the [Format templates](#format-templates) section. @@ -183,15 +185,16 @@ Text formatting can change length and case of the text. Use \<#\>, \<#\>\ ### Name List Formatters -| Formatter | Description | Example Usage | Example Result | -|------------------------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------| -| separator() | Specify the text used to join
multiple people's names.

Default is ", " | `` | Arthur Conan Doyle; Stephen Fry | -| format(\{T \| F \| M \| L \| S \| ID\}) **†** | Formats the human name using
the name part tags.
See [Name Formatter Usage](#name-formatters) above. | ``separator(; )]>`
``_{ID}_) separator(; )]>` | DOYLE, Arthur; FRY, Stephen
Doyle, A. \_B000AQ43GQ\_;
Fry, S. \_B000APAGVS\_ | -| unique(FMT) **†** | Reduce list members to a unique set. Entries are compared to each other after applying the given format. Duplicate entries (after format is applied) are removed, keeping the first occurrence. | ``
``separator(; )]>` | Arthur Conan Doyle, Stephen Fry
doyle; fry | -| sort(T \| F \| M \| L \| S \| ID) | Sorts the names by title,
first, middle, or last name,
suffix or Audible Contributor ID

These terms define the primary, secondary, tertiary, … sorting order.
You may combine multiple terms in sequence to specify multi‑level sorting.

*Sorting direction:*
uppercase = ascending
lowercase = descending

Default is unsorted | ``
``
`` | Stephen Fry, Arthur Conan Doyle
Stephen King, Stephen Fry
John P. Smith \_B000TTTBBB\_, John P. Smith \_B000TTTCCC\_, John S. Smith \_B000HHHVVV\_ | -| max(#) | Only use the first # of names

Default is all names | `` | Arthur Conan Doyle | -| slice(#..#) | Only use entries of the names list starting from # and ending at # (inclusive)

See [Text List Formatter Usage](#Text-List-Formatters) above for details on all the variants of `slice()` | `` | Arthur Conan Doyle | -| count(FMT) **‡** | Instead of returning some or all members of the list, print out the number of names using the specified [format](#number-formatters). | ``
`` | 2
02 | +| Formatter | Description | Example Usage | Example Result | +|-----------------------------------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------| +| separator() | Specify the text used to join
multiple people's names.

Default is ", " | `` | Arthur Conan Doyle; Stephen Fry | +| format(\{T \| F \| M \| L \| S \| ID\}) **†** | Formats the human name using
the name part tags.
See [Name Formatter Usage](#name-formatters) above. | ``separator(; )]>`
``_{ID}_) separator(; )]>` | DOYLE, Arthur; FRY, Stephen
Doyle, A. \_B000AQ43GQ\_;
Fry, S. \_B000APAGVS\_ | +| unique(FMT) **†** | Reduce list members to a unique set. Entries are compared to each other after applying the given format. Duplicate entries (after format is applied) are removed, keeping the first occurrence. | ``
``separator(; )]>` | Arthur Conan Doyle, Stephen Fry
doyle; fry | +| sort(T \| F \| M \| L \| S \| ID) | Sorts the names by title,
first, middle, or last name,
suffix or Audible Contributor ID

These terms define the primary, secondary, tertiary, … sorting order.
You may combine multiple terms in sequence to specify multi‑level sorting.

*Sorting direction:*
uppercase = ascending
lowercase = descending

Default is unsorted | ``
``
`` | Stephen Fry, Arthur Conan Doyle
Stephen King, Stephen Fry
John P. Smith \_B000TTTBBB\_, John P. Smith \_B000TTTCCC\_, John S. Smith \_B000HHHVVV\_ | +| max(#) | Only use the first # of names

Default is all names | `` | Arthur Conan Doyle | +| slice(#..#) | Only use entries of the names list starting from # and ending at # (inclusive)

See [Text List Formatter Usage](#Text-List-Formatters) above for details on all the variants of `slice()` | `` | Arthur Conan Doyle | +| count(FMT) **‡** | Instead of returning some or all members of the list, print out the number of names using the specified [format](#number-formatters). | ``
`` | 2
02 | +| filter(FMT [CHECK](#checks) VALUE) **†** | Filter list entries based on a condition. Each person is first formatted using the specified name format (or the default format if FMT is omitted), then compared against VALUE using the specified [CHECK](#checks). Only matching entries are included in the output.

**Syntax:** `filter(FORMAT CHECK VALUE)` or `filter(CHECK VALUE)`
- `FORMAT`: Optional name format to apply to each entry (e.g., `{L}`, `{M}`, `{L}, {F}`); defaults to `{T} {F} {M} {L} {S}`
- `CHECK`: Comparison operator (e.g., `=`, `!=`, `~`)
- `VALUE`: The value to compare against | ``
``
`Stephen Fry
Stephen Fry | **†** For further information on format templates, please refer to the [Format templates](#format-templates) section. @@ -314,9 +317,9 @@ string literal `O'Reilly`, you can use either `'O''Reilly'` or `"O'Reilly"`. | String Checks | Unicode Operator | Description | Examples | |---------------|------------------|--------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------| -| = | | Matches if values are equal (case-insensitive) | \ | -| != | | Matches if values are not equal (case-insensitive) | \
\ | -| ~ | | Matches if the first parameter matches the regular expression specified by the second parameter (case-insensitive) | \ | +| = | | Matches if values are equal (case-insensitive) | \
\ | +| != | | Matches if values are not equal (case-insensitive) | \
\
\ | +| ~ | | Matches if the first parameter matches the regular expression specified by the second parameter (case-insensitive) | \
\
\ | | Number Checks | Unicode Operator | Description | Examples | |---------------|------------------|----------------------------------------------------------------|-----------------------------------------------------------|