packages feed

canontra-0.2.0.0: test/Canontra/SIMDScanSpec.hs

{-# LANGUAGE OverloadedStrings #-}
module Canontra.SIMDScanSpec (spec) where

import qualified Data.ByteString as BS
import qualified Data.Text.Encoding as TE
import Test.Hspec

import Canontra.Canonical.FastScan (ScanResult (..), scanAsciiAndLineEndings)
import Canontra.Canonical.SIMDScan
  ( SIMDScanResult (..)
  , detectByteMatch64
  , detectZeroBytes64
  , fastCanonicalizeSIMD
  , isPureAsciiUnixSIMD
  , scanSourceSIMD
  , scanSourceSIMDFull
  )

spec :: Spec
spec = do
  describe "Canontra.Canonical.SIMDScan: 256-Bit Hardware SIMD Scanning Kernel" $ do

    describe "Step 3.1: SWAR Primitives & Vector Lane Helpers" $ do
      it "detects zero bytes within 64-bit machine words" $ do
        detectZeroBytes64 0x0000000000000000 `shouldBe` 0x8080808080808080
        detectZeroBytes64 0x0102030405060708 `shouldBe` 0
        (detectZeroBytes64 0x0100030405060708 /= 0) `shouldBe` True

      it "detects byte matches within 64-bit machine words" $ do
        let patCR = 0x0D0D0D0D0D0D0D0D
        detectByteMatch64 patCR 0x0A0A0A0A0A0A0A0A `shouldBe` 0
        detectByteMatch64 patCR 0x0D0A0D0A0D0A0D0A `shouldBe` 0x8000800080008000

    describe "256-Bit SIMD Kernel Classification & Equivalence" $ do
      it "classifies pure ASCII Unix streams as PureAsciiUnix" $ do
        let ascii = "def add(x, y):\n    return x + y\n"
        scanSourceSIMD ascii `shouldBe` PureAsciiUnix
        isPureAsciiUnixSIMD ascii `shouldBe` True

      it "identifies Windows CRLF line endings as ContainsCRLF" $ do
        let crlf = "def add(x, y):\r\n    return x + y\r\n"
        scanSourceSIMD crlf `shouldBe` ContainsCRLF
        isPureAsciiUnixSIMD crlf `shouldBe` False

      it "identifies UTF-8 non-ASCII characters as RequiresUnicodeNFC" $ do
        let utf8 = TE.encodeUtf8 "def greet():\n    return 'Hello, 世界'\n"
        scanSourceSIMD utf8 `shouldBe` RequiresUnicodeNFC
        isPureAsciiUnixSIMD utf8 `shouldBe` False

      it "guarantees bit-for-bit equivalence with scanAsciiAndLineEndings" $ do
        let cases =
              [ ""
              , "a"
              , "def foo(): pass\n"
              , "def bar():\r\n  return 42\r\n"
              , "comment = '# ñ'\n"
              , BS.replicate 31 0x61 -- 31 bytes
              , BS.replicate 32 0x61 -- exactly 32 bytes (1 lane)
              , BS.replicate 33 0x61 -- 33 bytes (1 lane + 1 remainder)
              , BS.replicate 64 0x61 -- 64 bytes (2 lanes)
              , BS.replicate 100 0x61 <> "\r\n"
              , BS.replicate 128 0x61 <> "µ"
              ]
        mapM_ (\bs -> scanSourceSIMD bs `shouldBe` scanAsciiAndLineEndings bs) cases

    describe "Detailed Vector Metrics & Delimiter Counts" $ do
      it "accurately counts string quote delimiters across 256-bit boundaries" $ do
        let source = "x = \"hello\" + 'world' + \"test\"\n"
            metrics = scanSourceSIMDFull source
        ssrQuoteCount metrics `shouldBe` 6
        ssrClassification metrics `shouldBe` PureAsciiUnix

      it "accurately counts comment delimiters across 256-bit boundaries" $ do
        let source = "# line 1\n# line 2\n// C-style comment\n"
            metrics = scanSourceSIMDFull source
        -- 2 hashes + 2 slashes = 4 comment markers
        ssrCommentCount metrics `shouldBe` 4
        ssrClassification metrics `shouldBe` PureAsciiUnix

      it "canonicalizes text with zero unnecessary NFC allocations" $ do
        let clean = "def clean():\n    return True\n"
        fastCanonicalizeSIMD clean `shouldBe` "def clean():\n    return True\n"
        let withCR = "def cr():\r\n    return False\r\n"
        fastCanonicalizeSIMD withCR `shouldBe` "def cr():\n    return False\n"