| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Diagnostics; |
| | | 5 | | using System.Globalization; |
| | | 6 | | using System.Runtime.CompilerServices; |
| | | 7 | | using System.Runtime.InteropServices; |
| | | 8 | | using static System.Buffers.StringSearchValuesHelper; |
| | | 9 | | |
| | | 10 | | namespace System.Buffers |
| | | 11 | | { |
| | | 12 | | /// <summary> |
| | | 13 | | /// An implementation of the Rabin-Karp algorithm we use as a fallback for |
| | | 14 | | /// short inputs that we can't handle with Teddy. |
| | | 15 | | /// https://en.wikipedia.org/wiki/Rabin%E2%80%93Karp_algorithm |
| | | 16 | | /// Has an O(i * m) worst-case, but we will only use it for very short inputs. |
| | | 17 | | /// </summary> |
| | | 18 | | internal readonly struct RabinKarp |
| | | 19 | | { |
| | | 20 | | // The number of values we'll accept before falling back to Aho-Corasick. |
| | | 21 | | // This also affects when Teddy may be used. |
| | | 22 | | public const int MaxValues = 80; |
| | | 23 | | |
| | | 24 | | // This is a tradeoff between memory consumption and the number of false positives |
| | | 25 | | // we have to rule out during the verification step. |
| | | 26 | | private const nuint BucketCount = 64; |
| | | 27 | | |
| | | 28 | | // 18 = Vector128<byte>.Count + 2 (MatchStartOffset for N=3) |
| | | 29 | | // The logic in this class is not safe from overflows, but we avoid any issues by |
| | | 30 | | // only calling into it for inputs that are too short for Teddy to handle. |
| | | 31 | | private const int MaxInputLength = 18 - 1; |
| | | 32 | | |
| | | 33 | | // We're using nuint as the rolling hash, so we can spread the hash over more bits on 64bit. |
| | | 34 | | private static int HashShiftPerElement => IntPtr.Size == 8 ? 2 : 1; |
| | | 35 | | |
| | | 36 | | private readonly string[]?[] _buckets; |
| | | 37 | | private readonly int _hashLength; |
| | | 38 | | private readonly nuint _hashUpdateMultiplier; |
| | | 39 | | |
| | | 40 | | public RabinKarp(ReadOnlySpan<string> values) |
| | | 41 | | { |
| | 0 | 42 | | Debug.Assert(values.Length <= MaxValues); |
| | | 43 | | |
| | 0 | 44 | | int minimumLength = int.MaxValue; |
| | 0 | 45 | | foreach (string value in values) |
| | | 46 | | { |
| | 0 | 47 | | minimumLength = Math.Min(minimumLength, value.Length); |
| | | 48 | | } |
| | | 49 | | |
| | 0 | 50 | | Debug.Assert(minimumLength > 1); |
| | | 51 | | |
| | 0 | 52 | | _hashLength = minimumLength; |
| | 0 | 53 | | _hashUpdateMultiplier = (nuint)1 << ((minimumLength - 1) * HashShiftPerElement); |
| | | 54 | | |
| | 0 | 55 | | if (minimumLength > MaxInputLength) |
| | | 56 | | { |
| | | 57 | | // All the values are long. They'll either be handled by Teddy or won't match at all. |
| | | 58 | | // There's no point in allocating the buckets as they will never be accessed. |
| | 0 | 59 | | _buckets = null!; |
| | 0 | 60 | | return; |
| | | 61 | | } |
| | | 62 | | |
| | 0 | 63 | | string[]?[] buckets = _buckets = new string[BucketCount][]; |
| | | 64 | | |
| | 0 | 65 | | foreach (string value in values) |
| | | 66 | | { |
| | 0 | 67 | | if (value.Length > MaxInputLength) |
| | | 68 | | { |
| | | 69 | | // This value can never match. There's no point in including it in the buckets. |
| | | 70 | | continue; |
| | | 71 | | } |
| | | 72 | | |
| | 0 | 73 | | nuint hash = 0; |
| | 0 | 74 | | for (int i = 0; i < minimumLength; i++) |
| | | 75 | | { |
| | 0 | 76 | | hash = (hash << HashShiftPerElement) + value[i]; |
| | | 77 | | } |
| | | 78 | | |
| | 0 | 79 | | nuint bucket = hash % BucketCount; |
| | | 80 | | string[] newBucket; |
| | | 81 | | |
| | | 82 | | // Start with a bucket containing 1 element and reallocate larger ones if needed. |
| | | 83 | | // As MaxValues is similar to BucketCount, we will have 1 value per bucket on average. |
| | 0 | 84 | | if (buckets[bucket] is string[] existingBucket) |
| | | 85 | | { |
| | 0 | 86 | | newBucket = new string[existingBucket.Length + 1]; |
| | 0 | 87 | | existingBucket.AsSpan().CopyTo(newBucket); |
| | | 88 | | } |
| | | 89 | | else |
| | | 90 | | { |
| | 0 | 91 | | newBucket = new string[1]; |
| | | 92 | | } |
| | | 93 | | |
| | 0 | 94 | | newBucket[^1] = value; |
| | 0 | 95 | | buckets[bucket] = newBucket; |
| | | 96 | | } |
| | 0 | 97 | | } |
| | | 98 | | |
| | | 99 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 100 | | public readonly int IndexOfAny<TCaseSensitivity>(ReadOnlySpan<char> span) |
| | | 101 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 102 | | { |
| | 0 | 103 | | return typeof(TCaseSensitivity) == typeof(CaseInsensitiveUnicode) |
| | 0 | 104 | | ? IndexOfAnyCaseInsensitiveUnicode(span) |
| | 0 | 105 | | : IndexOfAnyCore<TCaseSensitivity>(span); |
| | | 106 | | } |
| | | 107 | | |
| | | 108 | | private readonly int IndexOfAnyCore<TCaseSensitivity>(ReadOnlySpan<char> span) |
| | | 109 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 110 | | { |
| | 0 | 111 | | Debug.Assert(typeof(TCaseSensitivity) != typeof(CaseInsensitiveUnicode)); |
| | 0 | 112 | | Debug.Assert(span.Length <= MaxInputLength, "Teddy should have handled short inputs."); |
| | | 113 | | |
| | 0 | 114 | | ref char current = ref MemoryMarshal.GetReference(span); |
| | | 115 | | |
| | 0 | 116 | | int hashLength = _hashLength; |
| | | 117 | | |
| | 0 | 118 | | if (span.Length >= hashLength) |
| | | 119 | | { |
| | 0 | 120 | | ref char end = ref Unsafe.Add(ref MemoryMarshal.GetReference(span), (uint)(span.Length - hashLength)); |
| | | 121 | | |
| | 0 | 122 | | nuint hash = 0; |
| | 0 | 123 | | for (uint i = 0; i < hashLength; i++) |
| | | 124 | | { |
| | 0 | 125 | | hash = (hash << HashShiftPerElement) + TCaseSensitivity.TransformInput(Unsafe.Add(ref current, i)); |
| | | 126 | | } |
| | | 127 | | |
| | 0 | 128 | | Debug.Assert(_buckets is not null); |
| | 0 | 129 | | ref string[]? bucketsRef = ref MemoryMarshal.GetArrayDataReference(_buckets); |
| | | 130 | | |
| | 0 | 131 | | while (true) |
| | | 132 | | { |
| | 0 | 133 | | ValidateReadPosition(span, ref current); |
| | | 134 | | |
| | 0 | 135 | | if (Unsafe.Add(ref bucketsRef, hash % BucketCount) is string[] bucket) |
| | | 136 | | { |
| | 0 | 137 | | int startOffset = (int)((nuint)Unsafe.ByteOffset(ref MemoryMarshal.GetReference(span), ref curre |
| | | 138 | | |
| | 0 | 139 | | if (StartsWith<TCaseSensitivity>(ref current, span.Length - startOffset, bucket)) |
| | | 140 | | { |
| | 0 | 141 | | return startOffset; |
| | | 142 | | } |
| | | 143 | | } |
| | | 144 | | |
| | 0 | 145 | | if (Unsafe.IsAddressGreaterThanOrEqualTo(ref current, ref end)) |
| | | 146 | | { |
| | | 147 | | break; |
| | | 148 | | } |
| | | 149 | | |
| | 0 | 150 | | char previous = TCaseSensitivity.TransformInput(current); |
| | 0 | 151 | | char next = TCaseSensitivity.TransformInput(Unsafe.Add(ref current, (uint)hashLength)); |
| | | 152 | | |
| | | 153 | | // Update the hash by removing the previous character and adding the next one. |
| | 0 | 154 | | hash = ((hash - (previous * _hashUpdateMultiplier)) << HashShiftPerElement) + next; |
| | 0 | 155 | | current = ref Unsafe.Add(ref current, 1); |
| | | 156 | | } |
| | | 157 | | } |
| | | 158 | | |
| | 0 | 159 | | return -1; |
| | | 160 | | } |
| | | 161 | | |
| | | 162 | | private readonly unsafe int IndexOfAnyCaseInsensitiveUnicode(ReadOnlySpan<char> span) |
| | | 163 | | { |
| | 0 | 164 | | Debug.Assert(span.Length <= MaxInputLength, "Teddy should have handled long inputs."); |
| | | 165 | | |
| | 0 | 166 | | if (_hashLength > span.Length) |
| | | 167 | | { |
| | | 168 | | // Can't possibly match, all the values are longer than our input span. |
| | 0 | 169 | | return -1; |
| | | 170 | | } |
| | | 171 | | |
| | 0 | 172 | | Span<char> upperCase = stackalloc char[MaxInputLength].Slice(0, span.Length); |
| | | 173 | | |
| | 0 | 174 | | int charsWritten = Ordinal.ToUpperOrdinal(span, upperCase); |
| | 0 | 175 | | Debug.Assert(charsWritten == upperCase.Length); |
| | | 176 | | |
| | | 177 | | // CaseSensitive instead of CaseInsensitiveUnicode as we've already done the case conversion. |
| | 0 | 178 | | return IndexOfAnyCore<CaseSensitive>(upperCase); |
| | | 179 | | } |
| | | 180 | | } |
| | | 181 | | } |
| | | 182 | | |