anemll commited on Apr 10, 2025

Commit

01a207f

verified ·

1 Parent(s): edd7407

Upload folder using huggingface_hub

Browse files

Files changed (27) hide show

.DS_Store +0 -0
chat.py +862 -0
chat_full.py +960 -0
config.json +4 -0
llama_FFN_PF_chunk_01of02.mlmodelc/analytics/coremldata.bin +3 -0
llama_FFN_PF_chunk_01of02.mlmodelc/coremldata.bin +3 -0
llama_FFN_PF_chunk_01of02.mlmodelc/metadata.json +332 -0
llama_FFN_PF_chunk_01of02.mlmodelc/model.mil +0 -0
llama_FFN_PF_chunk_01of02.mlmodelc/weights/weight.bin +3 -0
llama_FFN_PF_chunk_02of02.mlmodelc/analytics/coremldata.bin +3 -0
llama_FFN_PF_chunk_02of02.mlmodelc/coremldata.bin +3 -0
llama_FFN_PF_chunk_02of02.mlmodelc/metadata.json +332 -0
llama_FFN_PF_chunk_02of02.mlmodelc/model.mil +0 -0
llama_FFN_PF_chunk_02of02.mlmodelc/weights/weight.bin +3 -0
llama_embeddings.mlmodelc/analytics/coremldata.bin +3 -0
llama_embeddings.mlmodelc/coremldata.bin +3 -0
llama_embeddings.mlmodelc/metadata.json +67 -0
llama_embeddings.mlmodelc/model.mil +11 -0
llama_embeddings.mlmodelc/weights/weight.bin +3 -0
llama_lm_head.mlmodelc/analytics/coremldata.bin +3 -0
llama_lm_head.mlmodelc/coremldata.bin +3 -0
llama_lm_head.mlmodelc/metadata.json +138 -0
llama_lm_head.mlmodelc/model.mil +98 -0
llama_lm_head.mlmodelc/weights/weight.bin +3 -0
meta.yaml +23 -0
tokenizer.json +0 -0
tokenizer_config.json +2062 -0

.DS_Store ADDED Viewed

Binary file (8.2 kB). View file

chat.py ADDED Viewed

	@@ -0,0 +1,862 @@

+# chat.py
+#!/usr/bin/env python3
+# chat.py
+# Copyright (c) 2025 Anemll
+# Licensed under the MIT License
+import argparse
+import os
+import re
+import glob
+from pathlib import Path
+import coremltools as ct
+from transformers import LlamaTokenizer, AutoTokenizer
+import torch
+import torch.nn.functional as F
+import numpy as np
+import queue
+import threading
+import time
+import yaml
+import sys
+# ANSI color codes
+LIGHT_BLUE = "\033[94m"
+DARK_BLUE = "\033[34m"
+LIGHT_GREEN = "\033[92m"
+RESET_COLOR = "\033[0m"
+# Add at top with other constants
+WARMUP_TOKEN_LIMIT = 10  # Maximum tokens to generate during warmup
+class TokenPrinter:
+    """Handles background printing of generated tokens."""
+    def __init__(self, tokenizer):
+        self.tokenizer = tokenizer
+        self.token_queue = queue.Queue()
+        self.stop_event = threading.Event()
+        self.thread = None
+        self.buffer = ""
+        self.lock = threading.Lock()
+        self.thinking = True  # Track if we're still in thinking mode
+        self.decoding_buffer = []  # Buffer for token IDs
+        # Add token counting and timing
+        self.start_time = time.time()
+        self.token_count = 0
+        self.start()
+    def start(self):
+        """Start the printer thread."""
+        if self.thread is None:
+            self.thread = threading.Thread(target=self._print_worker)
+            self.thread.daemon = True
+            self.thread.start()
+    def add_token(self, token_id):
+        """Add a token to the print queue."""
+        if not self.stop_event.is_set():
+            self.token_queue.put(token_id)
+            self.token_count += 1
+    def drain_buffer(self):
+        """Decode token IDs from decoding_buffer in the main thread."""
+        if not self.decoding_buffer:
+            return
+        # Decode all tokens at once in the main thread
+        token_str = self.tokenizer.decode(self.decoding_buffer)
+        self.decoding_buffer.clear()
+        # Color-handling logic
+        if self.thinking and "</think>" in token_str:
+            self.thinking = False
+            parts = token_str.split("</think>")
+            if len(parts) > 0:
+                print(parts[0] + "</think>", end='', flush=True)
+                if len(parts) > 1:
+                    print(LIGHT_BLUE + parts[1], end='', flush=True)
+        else:
+            if not self.thinking:
+                print(LIGHT_BLUE + token_str, end='', flush=True)
+            else:
+                print(token_str, end='', flush=True)
+    def _print_worker(self):
+        """Worker thread that takes token_ids from the queue."""
+        while not self.stop_event.is_set():
+            try:
+                token_id = self.token_queue.get(timeout=0.01)
+                with self.lock:
+                    self.decoding_buffer.append(token_id)
+                self.token_queue.task_done()
+            except queue.Empty:
+                continue
+            except Exception as e:
+                print(f"\nError: Token printer error: {str(e)}")
+                break
+    def stop(self):
+        """Stop the printer thread."""
+        if self.thread and self.thread.is_alive():
+            self.stop_event.set()
+            try:
+                self.thread.join(timeout=1.0)
+            except Exception:
+                pass
+            # Calculate and print tokens/s with shorter format in blue
+            elapsed = time.time() - self.start_time
+            if elapsed > 0 and self.token_count > 0:
+                tokens_per_sec = self.token_count / elapsed
+                print(f"\n{DARK_BLUE}{tokens_per_sec:.1f} t/s{RESET_COLOR}")
+            else:
+                print(RESET_COLOR)  # Reset color at the end
+        return self.buffer
+def parse_model_path(path):
+    """Parse model path and return full path with .mlmodelc or .mlpackage extension."""
+    path = Path(path)
+    # If path exists exactly as specified, return it
+    if path.exists():
+        return str(path)
+    # Try with both extensions
+    candidates = [
+        path,  # Original path
+        path.with_suffix('.mlmodelc'),  # With .mlmodelc
+        path.with_suffix('.mlpackage'),  # With .mlpackage
+        Path(str(path) + '.mlmodelc'),  # Handle case where extension is included
+        Path(str(path) + '.mlpackage')
+    ]
+    # Try all possible paths
+    for candidate in candidates:
+        if candidate.exists():
+            print(f"Found model at: {candidate}")
+            return str(candidate)
+    # If we get here, no valid path was found
+    print("\nError: Model not found. Tried following paths:")
+    for candidate in candidates:
+        print(f"  {candidate}")
+    raise FileNotFoundError(f"Model not found: {path}")
+def parse_ffn_filename(path):
+    """Parse FFN model filename to extract chunk information."""
+    path = Path(path)
+    pattern = r'FFN_PF.*_chunk_(\d+)of(\d+)'
+    match = re.search(pattern, path.name)
+    if match:
+        current_chunk = int(match.group(1))
+        total_chunks = int(match.group(2))
+        return current_chunk, total_chunks
+    return None, None
+def find_all_chunks(base_path):
+    """Find all chunk files matching the base FFN path pattern."""
+    path = Path(base_path)
+    pattern = re.sub(r'_chunk_\d+of\d+', '_chunk_*', str(path))
+    return sorted(glob.glob(pattern))
+def load_model(path, function_name=None):
+    """Load a CoreML model, handling both .mlmodelc and .mlpackage formats."""
+    path = Path(path)
+    compute_unit = ct.ComputeUnit.CPU_AND_NE
+    try:
+        if path.suffix == '.mlmodelc':
+            # For compiled models (.mlmodelc), use CompiledMLModel
+            if function_name:
+                return ct.models.CompiledMLModel(str(path), compute_unit, function_name=function_name)
+            else:
+                return ct.models.CompiledMLModel(str(path), compute_unit)
+        else:
+            # For packages (.mlpackage)
+            if function_name:
+                return ct.models.MLModel(str(path), function_name=function_name)
+            else:
+                return ct.models.MLModel(str(path))
+    except RuntimeError as e:
+        if "valid manifest does not exist" in str(e):
+            print(f"\nError: Could not load compiled model at {path}")
+            print("This might be because:")
+            print("1. The model is not properly compiled")
+            print("2. The model was compiled for a different OS version")
+            print("3. The model needs to be recompiled")
+            print("\nTry using the .mlpackage version instead, or recompile the model.")
+        raise
+def load_metadata(model,args):
+    # Extract metadata and config parameters
+    metadata = {}
+    if hasattr(model, 'user_defined_metadata'):
+        meta = model.user_defined_metadata
+        # Extract key parameters with defaults
+        metadata['context_length'] = int(meta.get('com.anemll.context_length', 512))
+        metadata['state_length'] = int(meta.get('com.anemll.state_length', metadata['context_length']))  # Added state_length
+        metadata['batch_size'] = int(meta.get('com.anemll.batch_size', 64))
+        metadata['lut_bits'] = int(meta.get('com.anemll.lut_bits', 0))
+        metadata['num_chunks'] = int(meta.get('com.anemll.num_chunks', 1))
+        print("\nExtracted Parameters:")
+        print(f"  Context Length: {metadata['context_length']}")
+        print(f"  State Length: {metadata['state_length']}")
+        print(f"  Prefill Batch Size: {metadata['batch_size']}")
+        print(f"  LUT Bits: {metadata['lut_bits']}")
+        print(f"  Number of Chunks: {metadata['num_chunks']}")
+        # Print model info
+        print("\nModel Info:")
+        if 'com.anemll.info' in meta:
+            print(f"  {meta['com.anemll.info']}")
+        if 'com.github.apple.coremltools.version' in meta:
+            print(f"  CoreML Tools: {meta['com.github.apple.coremltools.version']}")
+        # Print model input/output shapes
+        print("\nModel Shapes:")
+        if hasattr(model, 'input_description'):
+            print("  Inputs:")
+            for name, desc in model.input_description.items():
+                print(f"    {name}: {desc}")
+        if hasattr(model, 'output_description'):
+            print("  Outputs:")
+            for name, desc in model.output_description.items():
+                print(f"    {name}: {desc}")
+    else:
+        print("\nWarning: No metadata found in model")
+        # Check if model directory name contains context length pattern (ctxXXX)
+        ctx_len = 512
+        if args.context_length is  None:
+            import re
+            ctx_match = re.search(r'ctx(\d+)', str(args.d))
+            if ctx_match:
+                ctx_len0 = int(ctx_match.group(1))
+                if 512 <= ctx_len0 <= 8096:
+                    ctx_len = ctx_len0
+                    print(f"\nDetected context length {ctx_len} from directory name")
+            else:
+                print(f"\nWarning: No context length found in directory  {ctx_len} from directory name {args.d}")
+        else:
+            ctx_len = args.context_length
+        # Use defaults or values from args
+        metadata['context_length'] = ctx_len
+        metadata['state_length'] = ctx_len
+        # Get batch size from args or use default
+        metadata['batch_size'] = getattr(args, 'batch_size', 64)
+        metadata['lut_bits'] = 4
+        metadata['num_chunks'] = getattr(args, 'num_chunks', 4)
+        print("\nUsing parameters:")
+        print(f"  Context Length: {metadata['context_length']}")
+        print(f"  State Length: {metadata['state_length']}")
+        print(f"  Prefill Batch Size: {metadata['batch_size']}")
+        print(f"  LUT Bits: {metadata['lut_bits']}")
+        print(f"  Number of Chunks: {metadata['num_chunks']}")
+    # Override with values from args if they exist
+    if hasattr(args, 'batch_size') and args.batch_size is not None:
+        metadata['batch_size'] = args.batch_size
+        print(f"\nOverriding batch size from args: {args.batch_size}")
+    if hasattr(args, 'num_chunks') and args.num_chunks is not None:
+        metadata['num_chunks'] = args.num_chunks
+        print(f"\nOverriding num chunks from args: {args.num_chunks}")
+    return metadata
+def load_models(args,metadata):
+    """Load all required models and extract metadata."""
+    print("\nLoading models...")
+    try:
+        # Load embeddings model
+        print("\nLoading embeddings model...")
+        embed_path = parse_model_path(args.embed)
+        print(f"Loading from: {embed_path}")
+        embed_model = load_model(embed_path)
+        print("Embeddings model loaded successfully")
+        metadata = load_metadata(embed_model,args)
+        # Load LM head model
+        print("\nLoading LM head model...")
+        lmhead_path = parse_model_path(args.lmhead)
+        print(f"Loading from: {lmhead_path}")
+        lmhead_model = load_model(lmhead_path)
+        print("LM head model loaded successfully")
+        # Parse FFN path and find chunks if needed
+        print("\nLoading FFN+PREFILL model(s)...")
+        ffn_path = parse_model_path(args.ffn)
+        chunk_no, total_chunks = parse_ffn_filename(ffn_path)
+        ffn_models = []
+        if chunk_no and total_chunks:
+            print(f"\nDetected chunked FFN+PREFILL model ({total_chunks} chunks)")
+            # Find and load all chunks
+            chunk_paths = find_all_chunks(ffn_path)
+            if len(chunk_paths) != total_chunks:
+                raise ValueError(f"Found {len(chunk_paths)} chunks but filename indicates {total_chunks} chunks")
+            for chunk_path in chunk_paths:
+                print(f"\nLoading FFN+PREFILL chunk: {Path(chunk_path).name}")
+                try:
+                    # For chunked models, we need both infer and prefill functions
+                    ffn_models.append({
+                        'infer': load_model(chunk_path, function_name='infer'),
+                        'prefill': load_model(chunk_path, function_name='prefill')
+                    })
+                    print("Chunk loaded successfully")
+                except Exception as e:
+                    print(f"Error loading chunk {chunk_path}: {str(e)}")
+                    raise
+            metadata = load_metadata(ffn_models[0],args)
+        else:
+            print("\nLoading single FFN model...")
+            ffn_models.append(load_model(ffn_path))
+            print("FFN model loaded successfully")
+        return embed_model, ffn_models, lmhead_model, metadata
+    except Exception as e:
+        print(f"\nError loading models: {str(e)}")
+        print("\nPlease ensure all model files exist and are accessible.")
+        print("Expected files:")
+        print(f"  Embeddings: {args.embed}")
+        print(f"  LM Head: {args.lmhead}")
+        print(f"  FFN: {args.ffn}")
+        raise
+# At the top of the file, make this a default path
+def initialize_tokenizer(model_path=None):
+    """Initialize and configure the tokenizer."""
+    try:
+        tokenizer = AutoTokenizer.from_pretrained(
+            str(model_path),
+            use_fast=False,
+            trust_remote_code=True
+        )
+        print("\nTokenizer Configuration:")
+        print(f"Tokenizer type: {type(tokenizer)}")
+        print(f"Tokenizer name: {tokenizer.__class__.__name__}")
+        print(f"Vocabulary size: {len(tokenizer)}")
+        print(f"Model max length: {tokenizer.model_max_length}")
+        if tokenizer.pad_token is None:
+            tokenizer.pad_token = tokenizer.eos_token
+            tokenizer.pad_token_id = tokenizer.eos_token_id
+            print("Set PAD token to EOS token")
+        tokenizer.padding_side = "left"
+        print(f"\nSpecial Tokens:")
+        print(f"PAD token: '{tokenizer.pad_token}' (ID: {tokenizer.pad_token_id})")
+        print(f"EOS token: '{tokenizer.eos_token}' (ID: {tokenizer.eos_token_id})")
+        print(f"BOS token: '{tokenizer.bos_token}' (ID: {tokenizer.bos_token_id})")
+        print(f"UNK token: '{tokenizer.unk_token}' (ID: {tokenizer.unk_token_id})")
+        return tokenizer
+    except Exception as e:
+        print(f"\nError: Failed to load tokenizer from {model_path}")
+        print(f"Error details: {str(e)}")
+        print(f"Error type: {type(e)}")
+        print("\nThis code requires a Llama 3.2 model for chat template functionality.")
+        print("Please provide the path to a Llama 3.2 model directory.")
+        import traceback
+        traceback.print_exc()
+        raise
+def make_causal_mask(length, start):
+    """Create causal attention mask."""
+    mask = np.full((1, 1, length, length), -np.inf, dtype=np.float16)
+    row_indices = np.arange(length).reshape(length, 1)
+    col_indices = np.arange(length).reshape(1, length)
+    mask[:, :, col_indices <= (row_indices + start)] = 0
+    return mask
+def initialize_causal_mask(context_length):
+    """Initialize causal mask for transformer attention."""
+    causal_mask = make_causal_mask(context_length, 0)
+    causal_mask = torch.tensor(causal_mask, dtype=torch.float16)
+    print(f"\nInitialized causal mask for context length {context_length}")
+    return causal_mask
+def run_prefill(embed_model, ffn_models, input_ids, context_pos, context_length, batch_size=64, state=None, causal_mask=None):
+    """Run prefill on the input sequence."""
+    # Use provided causal mask or create one if not provided
+    if causal_mask is None:
+        causal_mask = make_causal_mask(context_length, 0)
+        causal_mask = torch.tensor(causal_mask, dtype=torch.float16)
+    # Process in batches
+    batch_pos = 0
+    while batch_pos < context_pos:
+        batch_end = min(batch_pos + batch_size, context_pos)
+        current_batch_size = batch_end - batch_pos
+        # Get current batch
+        batch_input = input_ids[:, batch_pos:batch_end]
+        # Always pad to full batch size for prefill
+        batch_input = F.pad(
+            batch_input,
+            (0, batch_size - current_batch_size),
+            value=0
+        )
+        # Generate position IDs for full batch size
+        position_ids = torch.arange(batch_size, dtype=torch.int32)  # Changed: Always use full batch size
+        batch_causal_mask = causal_mask[:, :, :batch_size, :]  # Changed: Use full batch size
+        # Run embeddings with proper batch size
+        hidden_states = torch.from_numpy(
+            embed_model.predict({
+                'input_ids': batch_input.numpy(),
+                'batch_size': np.array([batch_size], dtype=np.int32)  # Add batch_size parameter
+            })['hidden_states']
+        )
+        # Run through FFN chunks with state
+        for ffn_model in ffn_models:
+            if isinstance(ffn_model, dict):
+                inputs = {
+                    'hidden_states': hidden_states.numpy(),  # [1, 64, hidden_size]
+                    'position_ids': position_ids.numpy(),    # [64]
+                    'causal_mask': batch_causal_mask.numpy(), # [1, 1, 64, context_length]
+                    'current_pos': np.array([batch_pos], dtype=np.int32)  # [1]
+                }
+                output = ffn_model['prefill'].predict(inputs, state)
+                hidden_states = torch.from_numpy(output['output_hidden_states'])
+        batch_pos = batch_end
+    return torch.tensor([context_pos], dtype=torch.int32)
+def generate_next_token(embed_model, ffn_models, lmhead_model, input_ids, pos, context_length, state=None, causal_mask=None, temperature=0.0):
+    """Generate the next token."""
+    # Get current token
+    current_token = input_ids[:, pos-1:pos]  # [1, 1]
+    # Run embeddings
+    hidden_states = torch.from_numpy(
+        embed_model.predict({'input_ids': current_token.numpy()})['hidden_states']
+    )  # [1, 1, hidden_size]
+    # Create masks
+    update_mask = torch.zeros((1, 1, context_length, 1), dtype=torch.float16)
+    update_mask[0, 0, pos-1, 0] = 1.0
+    position_ids = torch.tensor([pos-1], dtype=torch.int32)  # [1]
+    # Use provided causal mask or create one if not provided
+    if causal_mask is None:
+        causal_mask_data = make_causal_mask(context_length, 0)
+        single_causal_mask = torch.tensor(causal_mask_data[:, :, pos-1:pos, :], dtype=torch.float16)  # [1, 1, 1, context_length]
+    else:
+        single_causal_mask = causal_mask[:, :, pos-1:pos, :]
+    # Run through FFN chunks with state
+    for ffn_model in ffn_models:
+        if isinstance(ffn_model, dict):
+            inputs = {
+                'hidden_states': hidden_states.numpy(),
+                'update_mask': update_mask.numpy(),
+                'position_ids': position_ids.numpy(),
+                'causal_mask': single_causal_mask.numpy(),
+                'current_pos': position_ids.numpy()
+            }
+            output = ffn_model['infer'].predict(inputs, state)
+            hidden_states = torch.from_numpy(output['output_hidden_states'])
+    # Run LM head
+    lm_output = lmhead_model.predict({'hidden_states': hidden_states.numpy()})
+    # Debug print
+    #print("\nLM Head output keys:", list(lm_output.keys()))
+    # Combine logits1-8 if they exist
+    if 'logits1' in lm_output:
+        # Concatenate all logits parts
+        logits_parts = []
+        for i in range(1, 9):
+            key = f'logits{i}'
+            if key in lm_output:
+                logits_parts.append(torch.from_numpy(lm_output[key]))
+        logits = torch.cat(logits_parts, dim=-1)  # Concatenate along vocab dimension
+    else:
+        # Try output_logits as fallback
+        logits = torch.from_numpy(lm_output['output_logits'])
+    # Apply temperature and sample
+    if temperature > 0:
+        logits = logits / temperature
+        probs = F.softmax(logits[0, -1, :], dim=-1)
+        next_token = torch.multinomial(probs, num_samples=1).item()
+    else:
+        next_token = torch.argmax(logits[0, -1, :]).item()
+    return next_token
+def create_unified_state(ffn_models, context_length):
+    """Create unified KV cache state for transformer."""
+    if isinstance(ffn_models[0], dict):
+        # Use first FFN model's prefill function to create state
+        state = ffn_models[0]['prefill'].make_state()
+        print(f"\nCreated unified transformer state for {len(ffn_models)} chunks")
+        return state
+    else:
+        state = ffn_models[0].make_state()
+        print("\nCreated unified transformer state")
+        return state
+def chat_loop(embed_model, ffn_models, lmhead_model, tokenizer, metadata, state, causal_mask=None, auto_prompt=None, warmup=False):
+    """Interactive chat loop."""
+    context_length = metadata.get('context_length')
+    batch_size = metadata.get('batch_size', 64)
+    if not warmup:
+        print(f"\nUsing context length: {context_length}")
+        print("\nStarting chat session. Press Ctrl+D to exit.")
+        print("Type your message and press Enter to chat.")
+    # Check if tokenizer has chat template and if it works
+    has_chat_template = False
+    try:
+        # Test if chat template works
+        test_messages = [{"role": "user", "content": "test"}]
+        tokenizer.apply_chat_template(test_messages, return_tensors="pt")
+        has_chat_template = True
+        if not warmup:
+            print("\nUsing chat template for prompts")
+    except:
+        if not warmup:
+            print("\nUsing manual formatting for prompts")
+    conversation = []
+    try:
+        while True:
+            try:
+                if not warmup:
+                    print(f"\n{LIGHT_GREEN}You:{RESET_COLOR}", end=' ', flush=True)
+                if auto_prompt is not None:
+                    user_input = auto_prompt
+                    if not warmup:
+                        print(user_input)
+                else:
+                    user_input = input().strip()
+            except EOFError:
+                if not warmup:
+                    print("\nExiting chat...")
+                break
+            if not user_input:
+                continue
+            # Format prompt based on tokenizer capabilities
+            if has_chat_template:
+                messages = [{"role": "user", "content": user_input}]
+                input_ids = tokenizer.apply_chat_template(
+                    messages,
+                    return_tensors="pt",
+                    add_generation_prompt=True
+                ).to(torch.int32)
+            else:
+                # Manual formatting for Llama models without chat template
+                formatted_prompt = f"[INST] {user_input} [/INST]"
+                input_ids = tokenizer(
+                    formatted_prompt,
+                    return_tensors="pt",
+                    add_special_tokens=True
+                ).input_ids.to(torch.int32)
+            context_pos = input_ids.size(1)
+            if not warmup:
+                print(f"\n{LIGHT_BLUE}Assistant:{RESET_COLOR}", end=' ', flush=True)
+            # Initialize token printer
+            token_printer = TokenPrinter(tokenizer)
+            tokens_generated = 0  # Track number of tokens
+            try:
+                # Start prefill timing
+                prefill_start = time.time()
+                # Run prefill with state and causal mask
+                current_pos = run_prefill(
+                    embed_model,
+                    ffn_models,
+                    input_ids,
+                    context_pos,
+                    context_length,
+                    batch_size,
+                    state,
+                    causal_mask
+                )
+                # Calculate prefill timing
+                prefill_time = time.time() - prefill_start
+                prefill_tokens = context_pos  # Number of tokens in input
+                prefill_tokens_per_sec = prefill_tokens / prefill_time if prefill_time > 0 else 0
+                # Generation loop with state
+                input_ids = input_ids
+                pos = context_pos
+                inference_start = time.time()
+                inference_tokens = 0
+                while pos < context_length - 1:
+                    # Generate next token with causal mask
+                    next_token = generate_next_token(
+                        embed_model,
+                        ffn_models,
+                        lmhead_model,
+                        input_ids,
+                        pos,
+                        context_length,
+                        state,
+                        causal_mask
+                    )
+                    # Add token to sequence
+                    if pos < input_ids.size(1):
+                        input_ids[0, pos] = next_token
+                    else:
+                        input_ids = torch.cat([
+                            input_ids,
+                            torch.tensor([[next_token]], dtype=torch.int32)
+                        ], dim=1)
+                    # Add to printer only if not in warmup
+                    if not warmup:
+                        token_printer.add_token(next_token)
+                        token_printer.drain_buffer()
+                    pos += 1
+                    tokens_generated += 1
+                    inference_tokens += 1
+                    # Check limits
+                    if warmup and tokens_generated >= WARMUP_TOKEN_LIMIT:
+                        break
+                    if next_token == tokenizer.eos_token_id:
+                        break
+                # Calculate inference timing
+                inference_time = time.time() - inference_start
+                inference_tokens_per_sec = inference_tokens / inference_time if inference_time > 0 else 0
+                # Get final response and add to conversation
+                if not warmup:
+                    response = token_printer.stop()
+                    # Print timing stats
+                    prefill_ms = prefill_time * 1000  # Convert to milliseconds
+                    print(f"\nPrefill: {prefill_ms:.1f}ms ({prefill_tokens_per_sec:.1f} t/s)")
+                    print(f"Inference: {inference_tokens_per_sec:.1f} t/s")
+                    print(f"Total: Generated {tokens_generated} tokens in {prefill_time + inference_time:.2f}s")
+                    conversation.append({"role": "assistant", "content": response})
+                else:
+                    token_printer.stop()  # Clean up without printing stats
+                # Exit after one response in auto_prompt mode
+                if auto_prompt is not None:
+                    break
+            except KeyboardInterrupt:
+                print("\nGeneration interrupted")
+                token_printer.stop()
+                continue
+    except Exception as e:
+        print(f"\nError in chat loop: {str(e)}")
+        import traceback
+        traceback.print_exc()
+def parse_args():
+    parser = argparse.ArgumentParser(description='Chat with CoreML LLaMA, gil resolved  (c) 2025 Anemll')
+    # Add meta.yaml option
+    parser.add_argument('--meta', type=str, help='Path to meta.yaml to load all parameters')
+    # Model paths
+    parser.add_argument('--d', '--dir', type=str, default='.',
+                       help='Directory containing model files (default: current directory)')
+    parser.add_argument('--embed', type=str, required=False,
+                       help='Path to embeddings model (relative to --dir)')
+    parser.add_argument('--ffn', type=str, required=False,
+                       help='Path to FFN model (can be chunked, relative to --dir)')
+    parser.add_argument('--lmhead', type=str, required=False,
+                       help='Path to LM head model (relative to --dir)')
+    parser.add_argument('--tokenizer', type=str, required=False,
+                       help='Path to tokenizer')
+    # Add new argument for auto-generation
+    parser.add_argument('--prompt', type=str,
+                       help='If specified, run once with this prompt and exit')
+    # Add no-warmup flag
+    parser.add_argument('--nw', action='store_true',
+                       help='Skip warmup phase')
+    # Model configuration
+    parser.add_argument('--context-length', type=int,
+                       help='Context length for the model (default: 512), if not provided, it will be detected from the model directory name ctxNUMBER')
+    parser.add_argument('--batch-size', type=int,
+                       help='Batch size for prefill (default: 64)')
+    args = parser.parse_args()
+    # If meta.yaml is provided, load parameters from it
+    if args.meta:
+        try:
+            with open(args.meta, 'r') as f:
+                meta = yaml.safe_load(f)
+            params = meta['model_info']['parameters']
+            # Set model directory to meta.yaml directory if not specified
+            if not args.d or args.d == '.':
+                args.d = str(Path(args.meta).parent)
+            # Build model paths based on parameters
+            prefix = params.get('model_prefix', 'llama')  # Default to 'llama' if not specified
+            lut_ffn = f"_lut{params['lut_ffn']}" if params['lut_ffn'] != 'none' else ''
+            lut_lmhead = f"_lut{params['lut_lmhead']}" if params['lut_lmhead'] != 'none' else ''
+            lut_embeddings = f"_lut{params['lut_embeddings']}" if params['lut_embeddings'] != 'none' else ''
+            num_chunks = int(params['num_chunks'])
+            # Set model paths if not specified
+            if not args.lmhead:
+                args.lmhead = f'{prefix}_lm_head{lut_lmhead}'
+            if not args.embed:
+                args.embed = f'{prefix}_embeddings{lut_embeddings}'  # Changed from lm_head to embeddings
+            if not args.ffn:
+                args.ffn = f'{prefix}_FFN_PF{lut_ffn}_chunk_01of{num_chunks:02d}'
+            if not args.tokenizer:
+                args.tokenizer = args.d
+            # Set other parameters if not overridden by command line
+            if args.context_length is None:
+                args.context_length = int(params['context_length'])
+            if args.batch_size is None:
+                args.batch_size = int(params['batch_size'])
+            args.num_chunks = num_chunks
+            print(f"\nLoaded parameters from {args.meta}:")
+            print(f"  Context Length: {args.context_length}")
+            print(f"  Batch Size: {args.batch_size}")
+            print(f"  Num Chunks: {args.num_chunks}")
+            print(f"  Models Directory: {args.d}")
+            print(f"  Embeddings: {args.embed}")
+            print(f"  LM Head: {args.lmhead}")
+            print(f"  FFN: {args.ffn}")
+        except Exception as e:
+            print(f"\nError loading meta.yaml: {str(e)}")
+            sys.exit(1)
+    return args
+def main():
+    args = parse_args()
+    # Convert directory to absolute path
+    model_dir = Path(args.d).resolve()
+    if not model_dir.exists():
+        print(f"\nError: Model directory not found: {model_dir}")
+        return 1
+    print(f"\nUsing model directory: {model_dir}")
+    print(f"Context length: {args.context_length}")
+    try:
+        # Update paths to be relative to model directory
+        args.embed = str(model_dir / args.embed)
+        args.ffn = str(model_dir / args.ffn)
+        args.lmhead = str(model_dir / args.lmhead)
+        # Handle tokenizer path separately since it's not relative to model_dir
+        if args.tokenizer is None:
+            args.tokenizer = str(model_dir)
+        if not Path(args.tokenizer).exists():
+            print(f"\nError: Tokenizer directory not found: {args.tokenizer}")
+            return 1
+        args.tokenizer = str(Path(args.tokenizer).resolve())  # Convert to absolute path
+        print(f"Using tokenizer path: {args.tokenizer}")
+        metadata = {}
+        # Load models and extract metadata
+        embed_model, ffn_models, lmhead_model, metadata = load_models(args,metadata)
+        print(f"\nMetadata befor args.context_length: {metadata}")
+        # Override context length from command line if provided
+        if args.context_length is not None:
+            metadata['context_length'] = args.context_length
+            metadata['state_length'] = args.context_length  # Also update state_length
+            print(f"\nOverriding context length from command line: {args.context_length}")
+        print(f"\nMetadata after load_models: {metadata}")
+        # Load tokenizer with resolved path
+        tokenizer = initialize_tokenizer(args.tokenizer)
+        if tokenizer is None:
+            raise RuntimeError("Failed to initialize tokenizer")
+        # Create unified state once
+        state = create_unified_state(ffn_models, metadata['context_length'])
+        # Initialize causal mask once
+        causal_mask = initialize_causal_mask(metadata['context_length'])
+        # Warmup runs to prevent Python GIL issues with CoreML !
+        if not args.nw:
+            for i in range(2):
+                chat_loop(
+                    embed_model=embed_model,
+                    ffn_models=ffn_models,
+                    lmhead_model=lmhead_model,
+                    tokenizer=tokenizer,
+                    metadata=metadata,
+                    state=state,
+                    causal_mask=causal_mask,  # Pass the causal mask
+                    warmup=True,
+                    auto_prompt="who are you?"
+                )
+        # Main run
+        chat_loop(
+            embed_model=embed_model,
+            ffn_models=ffn_models,
+            lmhead_model=lmhead_model,
+            tokenizer=tokenizer,
+            metadata=metadata,
+            state=state,
+            causal_mask=causal_mask,  # Pass the causal mask
+            warmup=False,
+            auto_prompt=args.prompt
+        )
+    except Exception as e:
+        print(f"\nError: {str(e)}")
+        import traceback
+        traceback.print_exc()
+        return 1
+    return 0
+if __name__ == "__main__":
+    exit(main())

chat_full.py ADDED Viewed

	@@ -0,0 +1,960 @@

+# chat.py
+#!/usr/bin/env python3
+# chat.py
+# Copyright (c) 2025 Anemll
+# Licensed under the MIT License
+import argparse
+import os
+import re
+import glob
+from pathlib import Path
+import coremltools as ct
+from transformers import LlamaTokenizer, AutoTokenizer
+import torch
+import torch.nn.functional as F
+import numpy as np
+import queue
+import threading
+import time
+import yaml
+import sys
+# ANSI color codes
+LIGHT_BLUE = "\033[94m"
+DARK_BLUE = "\033[34m"
+LIGHT_GREEN = "\033[92m"
+RESET_COLOR = "\033[0m"
+# Add at the top with other constants
+WARMUP_TOKEN_LIMIT = 10  # Maximum tokens to generate during warmup
+THINKING_MODE = False
+THINKING_PROMPT = """You are a deep thinking AI, you may use extremely long chains of thought to deeply consider the problem and deliberate with yourself via systematic reasoning processes to help come to a correct solution prior to answering. You should enclose your thoughts and internal monologue inside <think> </think> tags, and then provide your solution or response to the problem."""
+class TokenPrinter:
+    """Handles background printing of generated tokens."""
+    def __init__(self, tokenizer):
+        self.tokenizer = tokenizer
+        self.token_queue = queue.Queue()
+        self.stop_event = threading.Event()
+        self.thread = None
+        self.buffer = ""
+        self.lock = threading.Lock()
+        self.thinking = True  # Track if we're still in thinking mode
+        self.decoding_buffer = []  # Buffer for token IDs
+        # Timing and stats tracking
+        self.start_time = time.time()
+        self.token_count = 0
+        self.prefill_time = 0
+        self.inference_time = 0
+        self.context_pos = 0
+        self.start()
+    def start(self):
+        """Start the printer thread."""
+        if self.thread is None:
+            self.thread = threading.Thread(target=self._print_worker)
+            self.thread.daemon = True
+            self.thread.start()
+    def add_token(self, token_id):
+        """Add a token to the print queue."""
+        if not self.stop_event.is_set():
+            self.token_queue.put(token_id)
+            self.token_count += 1
+    def drain_buffer(self):
+        """Decode token IDs from decoding_buffer in the main thread."""
+        if not self.decoding_buffer:
+            return
+        # Decode all tokens at once in the main thread
+        token_str = self.tokenizer.decode(self.decoding_buffer)
+        self.decoding_buffer.clear()
+        # Color-handling logic
+        if self.thinking and "</think>" in token_str:
+            self.thinking = False
+            parts = token_str.split("</think>")
+            if len(parts) > 0:
+                print(parts[0] + "</think>", end='', flush=True)
+                if len(parts) > 1:
+                    print(LIGHT_BLUE + parts[1], end='', flush=True)
+        else:
+            if not self.thinking:
+                print(LIGHT_BLUE + token_str, end='', flush=True)
+            else:
+                print(token_str, end='', flush=True)
+    def _print_worker(self):
+        """Worker thread that takes token_ids from the queue."""
+        while not self.stop_event.is_set():
+            try:
+                token_id = self.token_queue.get(timeout=0.01)
+                with self.lock:
+                    self.decoding_buffer.append(token_id)
+                self.token_queue.task_done()
+            except queue.Empty:
+                continue
+            except Exception as e:
+                print(f"\nError: Token printer error: {str(e)}")
+                break
+    def stop(self):
+        """Stop the printer thread."""
+        if self.thread and self.thread.is_alive():
+            self.stop_event.set()
+            try:
+                self.thread.join(timeout=1.0)
+            except Exception:
+                pass
+            print(RESET_COLOR)  # Reset color at the end
+        return self.buffer
+    def set_timing(self, prefill_time, inference_time, context_pos):
+        """Set timing information."""
+        self.prefill_time = prefill_time
+        self.inference_time = inference_time
+        self.context_pos = context_pos
+def parse_model_path(path):
+    """Parse model path and return full path with .mlmodelc or .mlpackage extension."""
+    path = Path(path)
+    # If path exists exactly as specified, return it
+    if path.exists():
+        return str(path)
+    # Try with both extensions
+    candidates = [
+        path,  # Original path
+        path.with_suffix('.mlmodelc'),  # With .mlmodelc
+        path.with_suffix('.mlpackage'),  # With .mlpackage
+        Path(str(path) + '.mlmodelc'),  # Handle case where extension is included
+        Path(str(path) + '.mlpackage')
+    ]
+    # Try all possible paths
+    for candidate in candidates:
+        if candidate.exists():
+            print(f"Found model at: {candidate}")
+            return str(candidate)
+    # If we get here, no valid path was found
+    print("\nError: Model not found. Tried following paths:")
+    for candidate in candidates:
+        print(f"  {candidate}")
+    raise FileNotFoundError(f"Model not found: {path}")
+def parse_ffn_filename(path):
+    """Parse FFN model filename to extract chunk information."""
+    path = Path(path)
+    pattern = r'FFN_PF.*_chunk_(\d+)of(\d+)'
+    match = re.search(pattern, path.name)
+    if match:
+        current_chunk = int(match.group(1))
+        total_chunks = int(match.group(2))
+        return current_chunk, total_chunks
+    return None, None
+def find_all_chunks(base_path):
+    """Find all chunk files matching the base FFN path pattern."""
+    path = Path(base_path)
+    pattern = re.sub(r'_chunk_\d+of\d+', '_chunk_*', str(path))
+    return sorted(glob.glob(pattern))
+def load_model(path, function_name=None):
+    """Load a CoreML model, handling both .mlmodelc and .mlpackage formats."""
+    path = Path(path)
+    compute_unit = ct.ComputeUnit.CPU_AND_NE
+    try:
+        if path.suffix == '.mlmodelc':
+            # For compiled models (.mlmodelc), use CompiledMLModel
+            if function_name:
+                return ct.models.CompiledMLModel(str(path), compute_unit, function_name=function_name)
+            else:
+                return ct.models.CompiledMLModel(str(path), compute_unit)
+        else:
+            # For packages (.mlpackage)
+            if function_name:
+                return ct.models.MLModel(str(path), function_name=function_name)
+            else:
+                return ct.models.MLModel(str(path))
+    except RuntimeError as e:
+        if "valid manifest does not exist" in str(e):
+            print(f"\nError: Could not load compiled model at {path}")
+            print("This might be because:")
+            print("1. The model is not properly compiled")
+            print("2. The model was compiled for a different OS version")
+            print("3. The model needs to be recompiled")
+            print("\nTry using the .mlpackage version instead, or recompile the model.")
+        raise
+def parse_args():
+    parser = argparse.ArgumentParser(description='Full Chat with CoreML LLaMA with context window shifting, gil resolved (c) 2025 Anemll')
+    # Add meta.yaml option
+    parser.add_argument('--meta', type=str, help='Path to meta.yaml to load all parameters')
+    # Add existing arguments
+    parser.add_argument('--d', '--dir', type=str, default='.',
+                       help='Directory containing model files (default: current directory)')
+    parser.add_argument('--embed', type=str, required=False,
+                       help='Path to embeddings model (relative to --dir)')
+    parser.add_argument('--ffn', type=str, required=False,
+                       help='Path to FFN model (can be chunked, relative to --dir)')
+    parser.add_argument('--lmhead', type=str, required=False,
+                       help='Path to LM head model (relative to --dir)')
+    parser.add_argument('--tokenizer', type=str, required=False,
+                       help='Path to tokenizer')
+    # Add new argument for auto-generation
+    parser.add_argument('--prompt', type=str,
+                       help='If specified, run once with this prompt and exit')
+    # Add no-warmup flag
+    parser.add_argument('--nw', action='store_true',
+                       help='Skip warmup phase')
+    # Model configuration
+    parser.add_argument('--context-length', type=int,
+                       help='Context length for the model (default: 512), if not provided, it will be detected from the model directory name ctxNUMBER')
+    parser.add_argument('--batch-size', type=int,
+                       help='Batch size for prefill (default: 64)')
+    args = parser.parse_args()
+    # If meta.yaml is provided, load parameters from it
+    if args.meta:
+        try:
+            with open(args.meta, 'r') as f:
+                meta = yaml.safe_load(f)
+            params = meta['model_info']['parameters']
+            # Set model directory to meta.yaml directory if not specified
+            if not args.d or args.d == '.':
+                args.d = str(Path(args.meta).parent)
+            # Build model paths based on parameters
+            prefix = params.get('model_prefix', 'llama')  # Default to 'llama' if not specified
+            lut_ffn = f"_lut{params['lut_ffn']}" if params['lut_ffn'] != 'none' else ''
+            lut_lmhead = f"_lut{params['lut_lmhead']}" if params['lut_lmhead'] != 'none' else ''
+            lut_embeddings = f"_lut{params['lut_embeddings']}" if params['lut_embeddings'] != 'none' else ''
+            num_chunks = int(params['num_chunks'])
+            # Set model paths if not specified
+            if not args.lmhead:
+                args.lmhead = f'{prefix}_lm_head{lut_lmhead}'
+            if not args.embed:
+                args.embed = f'{prefix}_embeddings{lut_embeddings}'  # Changed from lm_head to embeddings
+            if not args.ffn:
+                args.ffn = f'{prefix}_FFN_PF{lut_ffn}_chunk_01of{num_chunks:02d}'
+            if not args.tokenizer:
+                args.tokenizer = args.d
+            # Set other parameters if not overridden by command line
+            if args.context_length is None:
+                args.context_length = int(params['context_length'])
+            if args.batch_size is None:
+                args.batch_size = int(params['batch_size'])
+            args.num_chunks = num_chunks
+            print(f"\nLoaded parameters from {args.meta}:")
+            print(f"  Context Length: {args.context_length}")
+            print(f"  Batch Size: {args.batch_size}")
+            print(f"  Num Chunks: {args.num_chunks}")
+            print(f"  Models Directory: {args.d}")
+            print(f"  Embeddings: {args.embed}")
+            print(f"  LM Head: {args.lmhead}")
+            print(f"  FFN: {args.ffn}")
+        except Exception as e:
+            print(f"\nError loading meta.yaml: {str(e)}")
+            sys.exit(1)
+    return args
+def load_metadata(model,args):
+    # Extract metadata and config parameters
+    metadata = {}
+    if hasattr(model, 'user_defined_metadata'):
+        meta = model.user_defined_metadata
+        # Extract key parameters with defaults
+        metadata['context_length'] = int(meta.get('com.anemll.context_length', 512))
+        metadata['state_length'] = int(meta.get('com.anemll.state_length', metadata['context_length']))  # Added state_length
+        metadata['batch_size'] = int(meta.get('com.anemll.batch_size', 64))
+        metadata['lut_bits'] = int(meta.get('com.anemll.lut_bits', 0))
+        metadata['num_chunks'] = int(meta.get('com.anemll.num_chunks', 1))
+        print("\nExtracted Parameters:")
+        print(f"  Context Length: {metadata['context_length']}")
+        print(f"  State Length: {metadata['state_length']}")
+        print(f"  Prefill Batch Size: {metadata['batch_size']}")
+        print(f"  LUT Bits: {metadata['lut_bits']}")
+        print(f"  Number of Chunks: {metadata['num_chunks']}")
+        # Print model info
+        print("\nModel Info:")
+        if 'com.anemll.info' in meta:
+            print(f"  {meta['com.anemll.info']}")
+        if 'com.github.apple.coremltools.version' in meta:
+            print(f"  CoreML Tools: {meta['com.github.apple.coremltools.version']}")
+        # Print model input/output shapes
+        print("\nModel Shapes:")
+        if hasattr(model, 'input_description'):
+            print("  Inputs:")
+            for name, desc in model.input_description.items():
+                print(f"    {name}: {desc}")
+        if hasattr(model, 'output_description'):
+            print("  Outputs:")
+            for name, desc in model.output_description.items():
+                print(f"    {name}: {desc}")
+    else:
+        print("\nWarning: No metadata found in model")
+        # Check if model directory name contains context length pattern (ctxXXX)
+        ctx_len = 512
+        if args.context_length is  None:
+            import re
+            ctx_match = re.search(r'ctx(\d+)', str(args.d))
+            if ctx_match:
+                ctx_len0 = int(ctx_match.group(1))
+                if 512 <= ctx_len0 <= 8096:
+                    ctx_len = ctx_len0
+                    print(f"\nDetected context length {ctx_len} from directory name")
+            else:
+                print(f"\nWarning: No context length found in directory  {ctx_len} from directory name {args.d}")
+        else:
+            ctx_len = args.context_length
+        # Use defaults or values from args
+        metadata['context_length'] = ctx_len
+        metadata['state_length'] = ctx_len
+        # Get batch size from args or use default
+        metadata['batch_size'] = getattr(args, 'batch_size', 64)
+        metadata['lut_bits'] = 4
+        metadata['num_chunks'] = getattr(args, 'num_chunks', 4)
+        print("\nUsing parameters:")
+        print(f"  Context Length: {metadata['context_length']}")
+        print(f"  State Length: {metadata['state_length']}")
+        print(f"  Prefill Batch Size: {metadata['batch_size']}")
+        print(f"  LUT Bits: {metadata['lut_bits']}")
+        print(f"  Number of Chunks: {metadata['num_chunks']}")
+    # Override with values from args if they exist
+    if hasattr(args, 'batch_size') and args.batch_size is not None:
+        metadata['batch_size'] = args.batch_size
+        print(f"\nOverriding batch size from args: {args.batch_size}")
+    if hasattr(args, 'num_chunks') and args.num_chunks is not None:
+        metadata['num_chunks'] = args.num_chunks
+        print(f"\nOverriding num chunks from args: {args.num_chunks}")
+    return metadata
+def load_models(args,metadata):
+    """Load all required models and extract metadata."""
+    print("\nLoading models...")
+    try:
+        # Load embeddings model
+        print("\nLoading embeddings model...")
+        embed_path = parse_model_path(args.embed)
+        print(f"Loading from: {embed_path}")
+        embed_model = load_model(embed_path)
+        print("Embeddings model loaded successfully")
+        metadata = load_metadata(embed_model,args)
+        # Load LM head model
+        print("\nLoading LM head model...")
+        lmhead_path = parse_model_path(args.lmhead)
+        print(f"Loading from: {lmhead_path}")
+        lmhead_model = load_model(lmhead_path)
+        print("LM head model loaded successfully")
+        # Parse FFN path and find chunks if needed
+        print("\nLoading FFN+PREFILL model(s)...")
+        ffn_path = parse_model_path(args.ffn)
+        chunk_no, total_chunks = parse_ffn_filename(ffn_path)
+        ffn_models = []
+        if chunk_no and total_chunks:
+            print(f"\nDetected chunked FFN+PREFILL model ({total_chunks} chunks)")
+            # Find and load all chunks
+            chunk_paths = find_all_chunks(ffn_path)
+            if len(chunk_paths) != total_chunks:
+                raise ValueError(f"Found {len(chunk_paths)} chunks but filename indicates {total_chunks} chunks")
+            for chunk_path in chunk_paths:
+                print(f"\nLoading FFN+PREFILL chunk: {Path(chunk_path).name}")
+                try:
+                    # For chunked models, we need both infer and prefill functions
+                    ffn_models.append({
+                        'infer': load_model(chunk_path, function_name='infer'),
+                        'prefill': load_model(chunk_path, function_name='prefill')
+                    })
+                    print("Chunk loaded successfully")
+                except Exception as e:
+                    print(f"Error loading chunk {chunk_path}: {str(e)}")
+                    raise
+            metadata = load_metadata(ffn_models[0],args)
+        else:
+            print("\nLoading single FFN model...")
+            ffn_models.append(load_model(ffn_path))
+            print("FFN model loaded successfully")
+        return embed_model, ffn_models, lmhead_model, metadata
+    except Exception as e:
+        print(f"\nError loading models: {str(e)}")
+        print("\nPlease ensure all model files exist and are accessible.")
+        print("Expected files:")
+        print(f"  Embeddings: {args.embed}")
+        print(f"  LM Head: {args.lmhead}")
+        print(f"  FFN: {args.ffn}")
+        raise
+# At the top of the file, make this a default path
+def initialize_tokenizer(model_path=None):
+    """Initialize and configure the tokenizer."""
+    try:
+        tokenizer = AutoTokenizer.from_pretrained(
+            str(model_path),
+            use_fast=False,
+            trust_remote_code=True
+        )
+        print("\nTokenizer Configuration:")
+        print(f"Tokenizer type: {type(tokenizer)}")
+        print(f"Tokenizer name: {tokenizer.__class__.__name__}")
+        print(f"Vocabulary size: {len(tokenizer)}")
+        print(f"Model max length: {tokenizer.model_max_length}")
+        if tokenizer.pad_token is None:
+            tokenizer.pad_token = tokenizer.eos_token
+            tokenizer.pad_token_id = tokenizer.eos_token_id
+            print("Set PAD token to EOS token")
+        tokenizer.padding_side = "left"
+        print(f"\nSpecial Tokens:")
+        print(f"PAD token: '{tokenizer.pad_token}' (ID: {tokenizer.pad_token_id})")
+        print(f"EOS token: '{tokenizer.eos_token}' (ID: {tokenizer.eos_token_id})")
+        print(f"BOS token: '{tokenizer.bos_token}' (ID: {tokenizer.bos_token_id})")
+        print(f"UNK token: '{tokenizer.unk_token}' (ID: {tokenizer.unk_token_id})")
+        return tokenizer
+    except Exception as e:
+        print(f"\nError: Failed to load tokenizer from {model_path}")
+        print(f"Error details: {str(e)}")
+        print(f"Error type: {type(e)}")
+        print("\nThis code requires a Llama 3.2 model for chat template functionality.")
+        print("Please provide the path to a Llama 3.2 model directory.")
+        import traceback
+        traceback.print_exc()
+        raise
+def make_causal_mask(length, start):
+    """Create causal attention mask."""
+    mask = np.full((1, 1, length, length), -np.inf, dtype=np.float16)
+    row_indices = np.arange(length).reshape(length, 1)
+    col_indices = np.arange(length).reshape(1, length)
+    mask[:, :, col_indices <= (row_indices + start)] = 0
+    return mask
+def run_prefill(embed_model, ffn_models, input_ids, current_pos, context_length, batch_size, state, causal_mask):
+    """Run prefill on the input sequence."""
+    # Use provided causal mask or create one if not provided
+    if causal_mask is None:
+        causal_mask = make_causal_mask(context_length, 0)
+        causal_mask = torch.tensor(causal_mask, dtype=torch.float16)
+    # Process in batches
+    batch_pos = 0
+    while batch_pos < current_pos:
+        batch_end = min(batch_pos + batch_size, current_pos)
+        current_batch_size = batch_end - batch_pos
+        # Get current batch
+        batch_input = input_ids[:, batch_pos:batch_end]
+        # Always pad to full batch size for prefill
+        batch_input = F.pad(
+            batch_input,
+            (0, batch_size - current_batch_size),
+            value=0
+        )
+        # Generate position IDs for full batch size
+        position_ids = torch.arange(batch_size, dtype=torch.int32)  # Changed: Always use full batch size
+        batch_causal_mask = causal_mask[:, :, :batch_size, :]  # Changed: Use full batch size
+        # Run embeddings with proper batch size
+        hidden_states = torch.from_numpy(
+            embed_model.predict({
+                'input_ids': batch_input.numpy(),
+                'batch_size': np.array([batch_size], dtype=np.int32)  # Add batch_size parameter
+            })['hidden_states']
+        )
+        # Run through FFN chunks with state
+        for ffn_model in ffn_models:
+            if isinstance(ffn_model, dict):
+                inputs = {
+                    'hidden_states': hidden_states.numpy(),  # [1, 64, hidden_size]
+                    'position_ids': position_ids.numpy(),    # [64]
+                    'causal_mask': batch_causal_mask.numpy(), # [1, 1, 64, context_length]
+                    'current_pos': np.array([batch_pos], dtype=np.int32)  # [1]
+                }
+                output = ffn_model['prefill'].predict(inputs, state)
+                hidden_states = torch.from_numpy(output['output_hidden_states'])
+        batch_pos = batch_end
+    return torch.tensor([current_pos], dtype=torch.int32)
+def generate_next_token(embed_model, ffn_models, lmhead_model, input_ids, pos, context_length, state, causal_mask, temperature=0.0):
+    """Generate the next token."""
+    # Get current token
+    current_token = input_ids[:, pos-1:pos]
+    # Run embeddings
+    hidden_states = torch.from_numpy(
+        embed_model.predict({'input_ids': current_token.numpy()})['hidden_states']
+    )
+    # Create masks
+    update_mask = torch.zeros((1, 1, context_length, 1), dtype=torch.float16)
+    update_mask[0, 0, pos-1, 0] = 1.0
+    position_ids = torch.tensor([pos-1], dtype=torch.int32)
+    # Use the pre-initialized causal mask and extract the single position portion
+    single_causal_mask = causal_mask[:, :, pos-1:pos, :]
+    # Run through FFN chunks
+    for ffn_model in ffn_models:
+        if isinstance(ffn_model, dict):
+            inputs = {
+                'hidden_states': hidden_states.numpy(),
+                'update_mask': update_mask.numpy(),
+                'position_ids': position_ids.numpy(),
+                'causal_mask': single_causal_mask.numpy(),
+                'current_pos': position_ids.numpy()
+            }
+            output = ffn_model['infer'].predict(inputs, state)
+            hidden_states = torch.from_numpy(output['output_hidden_states'])
+    # Run LM head and get next token
+    lm_output = lmhead_model.predict({'hidden_states': hidden_states.numpy()})
+    if 'logits1' in lm_output:
+        logits_parts = []
+        for i in range(1, 9):
+            key = f'logits{i}'
+            if key in lm_output:
+                logits_parts.append(torch.from_numpy(lm_output[key]))
+        logits = torch.cat(logits_parts, dim=-1)
+    else:
+        logits = torch.from_numpy(lm_output['output_logits'])
+    if temperature > 0:
+        logits = logits / temperature
+        probs = F.softmax(logits[0, -1, :], dim=-1)
+        next_token = torch.multinomial(probs, num_samples=1).item()
+    else:
+        next_token = torch.argmax(logits[0, -1, :]).item()
+    return next_token
+def create_unified_state(ffn_models, context_length):
+    """Create unified KV cache state for transformer."""
+    if isinstance(ffn_models[0], dict):
+        # Use first FFN model's prefill function to create state
+        state = ffn_models[0]['prefill'].make_state()
+        print(f"\nCreated unified transformer state for {len(ffn_models)} chunks")
+        return state
+    else:
+        state = ffn_models[0].make_state()
+        print("\nCreated unified transformer state")
+        return state
+def initialize_causal_mask(context_length):
+    """Initialize causal mask for transformer attention."""
+    causal_mask = make_causal_mask(context_length, 0)
+    causal_mask = torch.tensor(causal_mask, dtype=torch.float16)
+    print(f"\nInitialized causal mask for context length {context_length}")
+    return causal_mask
+def get_user_input():
+    """Get input from user, handling special key combinations."""
+    global THINKING_MODE
+    try:
+        import termios
+        import tty
+        import sys
+        def _getch():
+            fd = sys.stdin.fileno()
+            old_settings = termios.tcgetattr(fd)
+            try:
+                tty.setraw(sys.stdin.fileno())
+                ch = sys.stdin.read(1)
+            finally:
+                termios.tcsetattr(fd, termios.TCSADRAIN, old_settings)
+            return ch
+        buffer = []
+        while True:
+            char = _getch()
+            # Debug: print the character code
+            print(f"\nKey pressed: {repr(char)} (hex: {hex(ord(char))})")
+            # Check for Enter key
+            if char == '\r' or char == '\n':
+                print()  # Move to next line
+                input_text = ''.join(buffer)
+                # Check if the command is /t
+                if input_text == '/t':
+                    THINKING_MODE = not THINKING_MODE
+                    print(f"Thinking mode {'ON' if THINKING_MODE else 'OFF'}")
+                    buffer = []  # Clear buffer
+                    print(f"\n{LIGHT_GREEN}You{' (thinking)' if THINKING_MODE else ''}:{RESET_COLOR}", end=' ', flush=True)
+                    continue
+                return input_text
+            # Handle backspace
+            if char == '\x7f':  # backspace
+                if buffer:
+                    buffer.pop()
+                    sys.stdout.write('\b \b')  # Erase character
+                    sys.stdout.flush()
+                continue
+            # Handle Ctrl-C
+            if char == '\x03':  # Ctrl-C
+                print("^C")
+                raise KeyboardInterrupt
+            # Print character and add to buffer
+            sys.stdout.write(char)
+            sys.stdout.flush()
+            buffer.append(char)
+    except ImportError:
+        # Fallback for systems without termios
+        return input("> ")
+def chat_loop(embed_model, ffn_models, lmhead_model, tokenizer, metadata, state, causal_mask, auto_prompt=None, warmup=False):
+    """Interactive chat loop."""
+    global THINKING_MODE
+    context_length = metadata.get('context_length')
+    batch_size = metadata.get('batch_size', 64)
+    if not warmup:
+        print(f"\nUsing context length: {context_length}")
+        print("\nStarting chat session. Press Ctrl+D to exit.")
+        print("Type your message and press Enter to chat. Use /t to toggle thinking mode.")
+        print(f"Thinking mode is {'ON' if THINKING_MODE else 'OFF'}")
+    # Keep track of conversation history
+    conversation = []
+    try:
+        while True:
+            try:
+                if not warmup:
+                    print(f"\n{LIGHT_GREEN}You{' (thinking)' if THINKING_MODE else ''}:{RESET_COLOR}", end=' ', flush=True)
+                if auto_prompt is not None:
+                    user_input = auto_prompt
+                    if not warmup:
+                        print(user_input)
+                else:
+                    user_input = input().strip()
+            except EOFError:
+                if not warmup:
+                    print("\nExiting chat...")
+                break
+            if not user_input:
+                continue
+            # Handle /t command
+            if user_input == "/t":
+                THINKING_MODE = not THINKING_MODE
+                print(f"Thinking mode {'ON' if THINKING_MODE else 'OFF'}")
+                continue
+            # Add user message to conversation
+            conversation.append({"role": "user", "content": user_input})
+            # Format using chat template with full history
+            if THINKING_MODE:
+                # Add thinking prompt to system message
+                conversation_with_thinking = [{"role": "system", "content": THINKING_PROMPT}] + conversation
+                base_input_ids = tokenizer.apply_chat_template(
+                    conversation_with_thinking,
+                    return_tensors="pt",
+                    add_generation_prompt=True
+                ).to(torch.int32)
+            else:
+                base_input_ids = tokenizer.apply_chat_template(
+                    conversation,
+                    return_tensors="pt",
+                    add_generation_prompt=True
+                ).to(torch.int32)
+            # Check if we need to trim history
+            while base_input_ids.size(1) > context_length - 100:  # Leave room for response
+                # Remove oldest message pair (user + assistant)
+                if len(conversation) > 2:
+                    conversation = conversation[2:]  # Remove oldest pair
+                    base_input_ids = tokenizer.apply_chat_template(
+                        conversation,
+                        return_tensors="pt",
+                        add_generation_prompt=True
+                    ).to(torch.int32)
+                else:
+                    # If only current message remains and still too long, truncate
+                    base_input_ids = base_input_ids[:, -context_length//2:]
+                    break
+            context_pos = base_input_ids.size(1)
+            # Pad sequence to context_size
+            input_ids = F.pad(
+                base_input_ids,
+                (0, context_length - context_pos),
+                value=0
+            )
+            if not warmup:
+                print(f"\n{LIGHT_BLUE}Assistant:{RESET_COLOR}", end=' ', flush=True)
+            # Initialize token printer and collect response
+            token_printer = TokenPrinter(tokenizer)
+            response_tokens = []
+            generation_start_time = time.time()
+            try:
+                # Run prefill on entire context
+                current_pos = run_prefill(
+                    embed_model,
+                    ffn_models,
+                    input_ids,
+                    context_pos,
+                    context_length,
+                    batch_size,
+                    state,
+                    causal_mask
+                )
+                #print(f"\n[DEBUG] After initial prefill - current_pos: {current_pos}")
+                # Generation loop
+                pos = context_pos
+                tokens_generated = 0
+                inference_start = time.time()  # Start inference timing
+                while True:
+                    # Check if we need to shift window
+                    if pos >= context_length - 2:
+                        # Calculate shift to maintain full batches
+                        batch_size = metadata.get('batch_size', 64)
+                        # Calculate max batches that fit in context
+                        max_batches = context_length // batch_size
+                        desired_batches = max(1, max_batches - 2)  # Leave room for new tokens
+                        new_size = min(desired_batches * batch_size, context_length - batch_size)
+                        # Create shifted input_ids
+                        tmp = torch.zeros((1, context_length), dtype=torch.int32)
+                        tmp[:,0:new_size] = input_ids[:,pos-new_size:pos]
+                        input_ids = tmp
+                        # Reset state and run prefill
+                        # keep the same state
+                        #state = create_unified_state(ffn_models, context_length)
+                        current_pos = run_prefill(
+                            embed_model,
+                            ffn_models,
+                            input_ids,
+                            new_size,  # Prefill the entire shifted content
+                            context_length,
+                            batch_size,
+                            state,
+                            causal_mask
+                        )
+                        # Start generating from the next position
+                        pos = new_size  # Don't back up, continue from where we left off
+                        #print(f"\n[DEBUG] After shift - next token will be at pos {pos}")
+                        #print(f"[DEBUG] Context before next token: {tokenizer.decode(input_ids[0, pos-40:pos])}")
+                        window_shifted = True
+                    # Generate next token
+                    next_token = generate_next_token(
+                        embed_model,
+                        ffn_models,
+                        lmhead_model,
+                        input_ids,
+                        pos,
+                        context_length,
+                        state,
+                        causal_mask
+                    )
+                    # Add token
+                    input_ids[0, pos] = next_token
+                    if not warmup:
+                        token_printer.add_token(next_token)
+                        token_printer.drain_buffer()
+                    response_tokens.append(next_token)
+                    pos += 1
+                    tokens_generated += 1
+                    # In warmup mode, limit tokens
+                    if warmup and tokens_generated >= WARMUP_TOKEN_LIMIT:
+                        break
+                    if next_token == tokenizer.eos_token_id:
+                        break
+                inference_time = time.time() - inference_start  # Calculate inference time
+                # Add assistant response to conversation
+                response_text = token_printer.stop()
+                conversation.append({"role": "assistant", "content": response_text})
+                # Print stats only if not in warmup
+                if not warmup:
+                    total_time = time.time() - generation_start_time
+                    prefill_time = total_time - inference_time
+                    inference_tokens_per_sec = len(response_tokens) / inference_time if inference_time > 0 else 0
+                    prefill_ms = prefill_time * 1000
+                    prefill_tokens_per_sec = context_pos / prefill_time if prefill_time > 0 else 0
+                    print(f"{DARK_BLUE}{inference_tokens_per_sec:.1f} t/s, "
+                          f"TTFT: {prefill_ms:.1f}ms ({prefill_tokens_per_sec:.1f} t/s), "
+                          f"{len(response_tokens)} tokens{RESET_COLOR}")
+                if auto_prompt is not None:
+                    break
+            except KeyboardInterrupt:
+                if not warmup:
+                    print("\nGeneration interrupted")
+                token_printer.stop()
+                continue
+    except Exception as e:
+        if not warmup:
+            print(f"\nError in chat loop: {str(e)}")
+            import traceback
+            traceback.print_exc()
+def main():
+    args = parse_args()
+    # Convert directory to absolute path
+    model_dir = Path(args.d).resolve()
+    if not model_dir.exists():
+        print(f"\nError: Model directory not found: {model_dir}")
+        return 1
+    print(f"\nUsing model directory: {model_dir}")
+    print(f"Context length: {args.context_length}")
+    try:
+        # Update paths to be relative to model directory
+        args.embed = str(model_dir / args.embed)
+        args.ffn = str(model_dir / args.ffn)
+        args.lmhead = str(model_dir / args.lmhead)
+        # Handle tokenizer path separately since it's not relative to model_dir
+        if args.tokenizer is None:
+            args.tokenizer = str(model_dir)
+        if not Path(args.tokenizer).exists():
+            print(f"\nError: Tokenizer directory not found: {args.tokenizer}")
+            return 1
+        args.tokenizer = str(Path(args.tokenizer).resolve())  # Convert to absolute path
+        print(f"Using tokenizer path: {args.tokenizer}")
+        metadata = {}
+        # Load models and extract metadata
+        embed_model, ffn_models, lmhead_model, metadata = load_models(args,metadata)
+        print(f"\nMetadata befor args.context_length: {metadata}")
+        # Override context length from command line if provided
+        if args.context_length is not None:
+            metadata['context_length'] = args.context_length
+            metadata['state_length'] = args.context_length  # Also update state_length
+            print(f"\nOverriding context length from command line: {args.context_length}")
+        print(f"\nMetadata after load_models: {metadata}")
+        # Load tokenizer with resolved path
+        tokenizer = initialize_tokenizer(args.tokenizer)
+        if tokenizer is None:
+            raise RuntimeError("Failed to initialize tokenizer")
+        # Create unified state once
+        state = create_unified_state(ffn_models, metadata['context_length'])
+        # Initialize causal mask once
+        causal_mask = initialize_causal_mask(metadata['context_length'])
+        # Warmup runs to prevent Python GIL issues with CoreML !
+        if not args.nw:
+            for i in range(2):
+                chat_loop(
+                    embed_model=embed_model,
+                    ffn_models=ffn_models,
+                    lmhead_model=lmhead_model,
+                    tokenizer=tokenizer,
+                    metadata=metadata,
+                    state=state,  # Pass the state
+                    causal_mask=causal_mask,  # Pass the causal mask
+                    warmup=True,
+                    auto_prompt="who are you?"
+                )
+        # Main run
+        chat_loop(
+            embed_model=embed_model,
+            ffn_models=ffn_models,
+            lmhead_model=lmhead_model,
+            tokenizer=tokenizer,
+            metadata=metadata,
+            state=state,  # Pass the state
+            causal_mask=causal_mask,  # Pass the causal mask
+            warmup=False,
+            auto_prompt=args.prompt
+        )
+    except Exception as e:
+        print(f"\nError: {str(e)}")
+        import traceback
+        traceback.print_exc()
+        return 1
+    return 0
+if __name__ == "__main__":
+    exit(main())

config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "tokenizer_class": "LlamaTokenizer",
+  "model_type": "llama"
+}

llama_FFN_PF_chunk_01of02.mlmodelc/analytics/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a3aa36502bb9a0d7bebabbbd190c20ec0dc75d2da0d134effd295edac9f0f055
+size 243

llama_FFN_PF_chunk_01of02.mlmodelc/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8b9e9177095ad681bd28dbe0164d4bf56fd63566cce5e1c043f409fcc30a02be
+size 951

llama_FFN_PF_chunk_01of02.mlmodelc/metadata.json ADDED Viewed

	@@ -0,0 +1,332 @@

+[
+  {
+    "metadataOutputVersion" : "3.0",
+    "userDefinedMetadata" : {
+      "com.github.apple.coremltools.version" : "8.2",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript",
+      "com.anemll.context_length" : "512",
+      "com.github.apple.coremltools.source" : "torch==2.5.0",
+      "com.anemll.num_chunks" : "2",
+      "com.anemll.batch_size" : "64",
+      "com.anemll.info" : "Converted with Anemll v0.3.0",
+      "com.anemll.chunk_no" : "1"
+    },
+    "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
+    },
+    "inputSchema" : [
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 2048]",
+        "name" : "hidden_states",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Int32",
+        "formattedType" : "MultiArray (Int32 1)",
+        "shortDescription" : "",
+        "shape" : "[1]",
+        "name" : "position_ids",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 1 × 512)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 1, 512]",
+        "name" : "causal_mask",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Int32",
+        "formattedType" : "MultiArray (Int32 1)",
+        "shortDescription" : "",
+        "shape" : "[1]",
+        "name" : "current_pos",
+        "type" : "MultiArray"
+      }
+    ],
+    "outputSchema" : [
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 2048]",
+        "name" : "output_hidden_states",
+        "type" : "MultiArray"
+      }
+    ],
+    "modelParameters" : [
+    ],
+    "storagePrecision" : "Float16",
+    "method" : "predict",
+    "functions" : [
+      {
+        "inputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 2048]",
+            "name" : "hidden_states",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 1)",
+            "shortDescription" : "",
+            "shape" : "[1]",
+            "name" : "position_ids",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 1 × 512)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 1, 512]",
+            "name" : "causal_mask",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 1)",
+            "shortDescription" : "",
+            "shape" : "[1]",
+            "name" : "current_pos",
+            "type" : "MultiArray"
+          }
+        ],
+        "computePrecision" : "Mixed (Float16, Int32)",
+        "storagePrecision" : "Float16",
+        "stateSchema" : [
+          {
+            "dataType" : "Float16",
+            "isOptional" : "0",
+            "formattedType" : "State (Float16 32 × 8 × 512 × 64)",
+            "shortDescription" : "",
+            "shape" : "[32, 8, 512, 64]",
+            "name" : "model_model_kv_cache_0",
+            "type" : "State"
+          }
+        ],
+        "outputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 2048]",
+            "name" : "output_hidden_states",
+            "type" : "MultiArray"
+          }
+        ],
+        "name" : "infer",
+        "mlProgramOperationTypeHistogram" : {
+          "Ios18.expandDims" : 32,
+          "Ios18.mul" : 80,
+          "Ios18.matmul" : 16,
+          "Identity" : 1,
+          "Ios16.reduceMean" : 16,
+          "Ios18.exp" : 8,
+          "Ios18.realDiv" : 8,
+          "Ios18.greaterEqual" : 1,
+          "Select" : 1,
+          "Ios18.readState" : 17,
+          "Tile" : 16,
+          "Ios18.gather" : 2,
+          "Ios18.add" : 42,
+          "Ios18.layerNorm" : 16,
+          "Ios18.sliceUpdate" : 16,
+          "Ios18.writeState" : 16,
+          "Ios18.reshape" : 50,
+          "Ios16.reduceMax" : 8,
+          "Ios16.reduceSum" : 8,
+          "Ios18.conv" : 48,
+          "Ios18.concat" : 48,
+          "Ios18.transpose" : 32,
+          "Ios18.sub" : 40,
+          "Ios18.linear" : 8,
+          "Ios18.silu" : 8,
+          "Ios18.sliceByIndex" : 50,
+          "Ios18.squeeze" : 24
+        }
+      },
+      {
+        "inputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 64 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 64, 2048]",
+            "name" : "hidden_states",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 64)",
+            "shortDescription" : "",
+            "shape" : "[64]",
+            "name" : "position_ids",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 64 × 512)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 64, 512]",
+            "name" : "causal_mask",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 1)",
+            "shortDescription" : "",
+            "shape" : "[1]",
+            "name" : "current_pos",
+            "type" : "MultiArray"
+          }
+        ],
+        "computePrecision" : "Mixed (Float16, Int32)",
+        "storagePrecision" : "Float16",
+        "stateSchema" : [
+          {
+            "dataType" : "Float16",
+            "isOptional" : "0",
+            "formattedType" : "State (Float16 32 × 8 × 512 × 64)",
+            "shortDescription" : "",
+            "shape" : "[32, 8, 512, 64]",
+            "name" : "model_model_kv_cache_0",
+            "type" : "State"
+          }
+        ],
+        "outputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 64 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 64, 2048]",
+            "name" : "output_hidden_states",
+            "type" : "MultiArray"
+          }
+        ],
+        "name" : "prefill",
+        "mlProgramOperationTypeHistogram" : {
+          "Ios18.expandDims" : 32,
+          "Ios18.mul" : 80,
+          "Ios18.matmul" : 16,
+          "Ios16.reduceMean" : 16,
+          "Ios18.exp" : 8,
+          "Ios18.realDiv" : 8,
+          "Ios18.greaterEqual" : 1,
+          "Select" : 1,
+          "Ios18.readState" : 17,
+          "Tile" : 16,
+          "Ios18.gather" : 2,
+          "Ios18.add" : 42,
+          "Ios18.layerNorm" : 16,
+          "Ios18.sliceUpdate" : 16,
+          "Ios18.writeState" : 16,
+          "Ios18.reshape" : 66,
+          "Ios16.reduceMax" : 8,
+          "Ios16.reduceSum" : 8,
+          "Ios18.conv" : 48,
+          "Ios18.concat" : 48,
+          "Ios18.transpose" : 58,
+          "Ios18.sub" : 40,
+          "Ios18.linear" : 8,
+          "Ios18.silu" : 8,
+          "Ios18.sliceByIndex" : 50,
+          "Ios18.squeeze" : 24
+        }
+      }
+    ],
+    "version" : "0.3.0",
+    "isUpdatable" : "0",
+    "defaultFunctionName" : "infer",
+    "specificationVersion" : 9,
+    "stateSchema" : [
+      {
+        "dataType" : "Float16",
+        "isOptional" : "0",
+        "formattedType" : "State (Float16 32 × 8 × 512 × 64)",
+        "shortDescription" : "",
+        "shape" : "[32, 8, 512, 64]",
+        "name" : "model_model_kv_cache_0",
+        "type" : "State"
+      }
+    ],
+    "computePrecision" : "Mixed (Float16, Int32)",
+    "mlProgramOperationTypeHistogram" : {
+      "Ios18.expandDims" : 32,
+      "Ios18.mul" : 80,
+      "Ios18.matmul" : 16,
+      "Identity" : 1,
+      "Ios16.reduceMean" : 16,
+      "Ios18.exp" : 8,
+      "Ios18.realDiv" : 8,
+      "Ios18.greaterEqual" : 1,
+      "Select" : 1,
+      "Ios18.readState" : 17,
+      "Tile" : 16,
+      "Ios18.gather" : 2,
+      "Ios18.add" : 42,
+      "Ios18.layerNorm" : 16,
+      "Ios18.sliceUpdate" : 16,
+      "Ios18.writeState" : 16,
+      "Ios18.reshape" : 50,
+      "Ios16.reduceMax" : 8,
+      "Ios16.reduceSum" : 8,
+      "Ios18.conv" : 48,
+      "Ios18.concat" : 48,
+      "Ios18.transpose" : 32,
+      "Ios18.sub" : 40,
+      "Ios18.linear" : 8,
+      "Ios18.silu" : 8,
+      "Ios18.sliceByIndex" : 50,
+      "Ios18.squeeze" : 24
+    },
+    "shortDescription" : "Anemll Model: Multifunction FFN+Prefill",
+    "generatedClassName" : "llama_FFN_PF_chunk_01of02",
+    "author" : "Converted with Anemll v0.3.0",
+    "modelType" : {
+      "name" : "MLModelType_mlProgram"
+    }
+  }
+]

llama_FFN_PF_chunk_01of02.mlmodelc/model.mil ADDED Viewed

The diff for this file is too large to render. See raw diff

llama_FFN_PF_chunk_01of02.mlmodelc/weights/weight.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:668bd8b220674c2b2bd2abd87095bcaba1075fe6ca806fef5c744702442a4b63
+size 1006707456

llama_FFN_PF_chunk_02of02.mlmodelc/analytics/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0f39a50160bba911d8ef28f9854cbb525548c6390db3eb35804b599a89889a07
+size 243

llama_FFN_PF_chunk_02of02.mlmodelc/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8617cece12f2a6c4fca63af54c3a3a9e87b83a08356a56404c1b8e4c743d9c87
+size 951

llama_FFN_PF_chunk_02of02.mlmodelc/metadata.json ADDED Viewed

	@@ -0,0 +1,332 @@

+[
+  {
+    "metadataOutputVersion" : "3.0",
+    "userDefinedMetadata" : {
+      "com.github.apple.coremltools.version" : "8.2",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript",
+      "com.anemll.context_length" : "512",
+      "com.github.apple.coremltools.source" : "torch==2.5.0",
+      "com.anemll.num_chunks" : "2",
+      "com.anemll.batch_size" : "64",
+      "com.anemll.info" : "Converted with Anemll v0.3.0",
+      "com.anemll.chunk_no" : "2"
+    },
+    "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
+    },
+    "inputSchema" : [
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 2048]",
+        "name" : "hidden_states",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Int32",
+        "formattedType" : "MultiArray (Int32 1)",
+        "shortDescription" : "",
+        "shape" : "[1]",
+        "name" : "position_ids",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 1 × 512)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 1, 512]",
+        "name" : "causal_mask",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Int32",
+        "formattedType" : "MultiArray (Int32 1)",
+        "shortDescription" : "",
+        "shape" : "[1]",
+        "name" : "current_pos",
+        "type" : "MultiArray"
+      }
+    ],
+    "outputSchema" : [
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 2048]",
+        "name" : "output_hidden_states",
+        "type" : "MultiArray"
+      }
+    ],
+    "modelParameters" : [
+    ],
+    "storagePrecision" : "Float16",
+    "method" : "predict",
+    "functions" : [
+      {
+        "inputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 2048]",
+            "name" : "hidden_states",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 1)",
+            "shortDescription" : "",
+            "shape" : "[1]",
+            "name" : "position_ids",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 1 × 512)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 1, 512]",
+            "name" : "causal_mask",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 1)",
+            "shortDescription" : "",
+            "shape" : "[1]",
+            "name" : "current_pos",
+            "type" : "MultiArray"
+          }
+        ],
+        "computePrecision" : "Mixed (Float16, Int32)",
+        "storagePrecision" : "Float16",
+        "stateSchema" : [
+          {
+            "dataType" : "Float16",
+            "isOptional" : "0",
+            "formattedType" : "State (Float16 32 × 8 × 512 × 64)",
+            "shortDescription" : "",
+            "shape" : "[32, 8, 512, 64]",
+            "name" : "model_model_kv_cache_0",
+            "type" : "State"
+          }
+        ],
+        "outputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 2048]",
+            "name" : "output_hidden_states",
+            "type" : "MultiArray"
+          }
+        ],
+        "name" : "infer",
+        "mlProgramOperationTypeHistogram" : {
+          "Ios18.expandDims" : 32,
+          "Ios18.mul" : 80,
+          "Ios18.matmul" : 16,
+          "Identity" : 1,
+          "Ios16.reduceMean" : 17,
+          "Ios18.exp" : 8,
+          "Ios18.realDiv" : 8,
+          "Ios18.greaterEqual" : 1,
+          "Select" : 1,
+          "Ios18.readState" : 17,
+          "Tile" : 16,
+          "Ios18.gather" : 2,
+          "Ios18.add" : 42,
+          "Ios18.layerNorm" : 17,
+          "Ios18.sliceUpdate" : 16,
+          "Ios18.writeState" : 16,
+          "Ios18.reshape" : 50,
+          "Ios16.reduceMax" : 8,
+          "Ios16.reduceSum" : 8,
+          "Ios18.conv" : 48,
+          "Ios18.concat" : 48,
+          "Ios18.transpose" : 32,
+          "Ios18.sub" : 41,
+          "Ios18.linear" : 8,
+          "Ios18.silu" : 8,
+          "Ios18.sliceByIndex" : 50,
+          "Ios18.squeeze" : 24
+        }
+      },
+      {
+        "inputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 64 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 64, 2048]",
+            "name" : "hidden_states",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 64)",
+            "shortDescription" : "",
+            "shape" : "[64]",
+            "name" : "position_ids",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 64 × 512)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 64, 512]",
+            "name" : "causal_mask",
+            "type" : "MultiArray"
+          },
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Int32",
+            "formattedType" : "MultiArray (Int32 1)",
+            "shortDescription" : "",
+            "shape" : "[1]",
+            "name" : "current_pos",
+            "type" : "MultiArray"
+          }
+        ],
+        "computePrecision" : "Mixed (Float16, Int32)",
+        "storagePrecision" : "Float16",
+        "stateSchema" : [
+          {
+            "dataType" : "Float16",
+            "isOptional" : "0",
+            "formattedType" : "State (Float16 32 × 8 × 512 × 64)",
+            "shortDescription" : "",
+            "shape" : "[32, 8, 512, 64]",
+            "name" : "model_model_kv_cache_0",
+            "type" : "State"
+          }
+        ],
+        "outputSchema" : [
+          {
+            "hasShapeFlexibility" : "0",
+            "isOptional" : "0",
+            "dataType" : "Float16",
+            "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+            "shortDescription" : "",
+            "shape" : "[1, 1, 2048]",
+            "name" : "output_hidden_states",
+            "type" : "MultiArray"
+          }
+        ],
+        "name" : "prefill",
+        "mlProgramOperationTypeHistogram" : {
+          "Ios18.expandDims" : 31,
+          "Ios18.mul" : 79,
+          "Ios18.matmul" : 16,
+          "Ios16.reduceMean" : 15,
+          "Ios18.exp" : 8,
+          "Ios18.realDiv" : 8,
+          "Ios18.greaterEqual" : 1,
+          "Select" : 1,
+          "Ios18.readState" : 17,
+          "Tile" : 16,
+          "Ios18.gather" : 2,
+          "Ios18.add" : 41,
+          "Ios18.layerNorm" : 15,
+          "Ios18.sliceUpdate" : 16,
+          "Ios18.writeState" : 16,
+          "Ios18.reshape" : 66,
+          "Ios16.reduceMax" : 8,
+          "Ios16.reduceSum" : 8,
+          "Ios18.conv" : 45,
+          "Ios18.concat" : 48,
+          "Ios18.transpose" : 56,
+          "Ios18.sub" : 39,
+          "Ios18.linear" : 8,
+          "Ios18.silu" : 7,
+          "Ios18.sliceByIndex" : 51,
+          "Ios18.squeeze" : 23
+        }
+      }
+    ],
+    "version" : "0.3.0",
+    "isUpdatable" : "0",
+    "defaultFunctionName" : "infer",
+    "specificationVersion" : 9,
+    "stateSchema" : [
+      {
+        "dataType" : "Float16",
+        "isOptional" : "0",
+        "formattedType" : "State (Float16 32 × 8 × 512 × 64)",
+        "shortDescription" : "",
+        "shape" : "[32, 8, 512, 64]",
+        "name" : "model_model_kv_cache_0",
+        "type" : "State"
+      }
+    ],
+    "computePrecision" : "Mixed (Float16, Int32)",
+    "mlProgramOperationTypeHistogram" : {
+      "Ios18.expandDims" : 32,
+      "Ios18.mul" : 80,
+      "Ios18.matmul" : 16,
+      "Identity" : 1,
+      "Ios16.reduceMean" : 17,
+      "Ios18.exp" : 8,
+      "Ios18.realDiv" : 8,
+      "Ios18.greaterEqual" : 1,
+      "Select" : 1,
+      "Ios18.readState" : 17,
+      "Tile" : 16,
+      "Ios18.gather" : 2,
+      "Ios18.add" : 42,
+      "Ios18.layerNorm" : 17,
+      "Ios18.sliceUpdate" : 16,
+      "Ios18.writeState" : 16,
+      "Ios18.reshape" : 50,
+      "Ios16.reduceMax" : 8,
+      "Ios16.reduceSum" : 8,
+      "Ios18.conv" : 48,
+      "Ios18.concat" : 48,
+      "Ios18.transpose" : 32,
+      "Ios18.sub" : 41,
+      "Ios18.linear" : 8,
+      "Ios18.silu" : 8,
+      "Ios18.sliceByIndex" : 50,
+      "Ios18.squeeze" : 24
+    },
+    "shortDescription" : "Anemll Model: Multifunction FFN+Prefill",
+    "generatedClassName" : "llama_FFN_PF_chunk_02of02",
+    "author" : "Converted with Anemll v0.3.0",
+    "modelType" : {
+      "name" : "MLModelType_mlProgram"
+    }
+  }
+]

llama_FFN_PF_chunk_02of02.mlmodelc/model.mil ADDED Viewed

The diff for this file is too large to render. See raw diff

llama_FFN_PF_chunk_02of02.mlmodelc/weights/weight.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6ecd8a42e94f71bfe30e803d8e1db739dcf3c1325ae82dab88c20f62b3dedc14
+size 1006711616

llama_embeddings.mlmodelc/analytics/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4a584b4103b6847031bf2b23d8a9947db58bffe4ba6a865290c8ef3d33a3ecff
+size 243

llama_embeddings.mlmodelc/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:dfb59657e6a7d770d6e215cd18b587be3083bac8f313bd2560d6e09f3779cb02
+size 498

llama_embeddings.mlmodelc/metadata.json ADDED Viewed

	@@ -0,0 +1,67 @@

+[
+  {
+    "shortDescription" : "Anemll Model (Embeddings) converted to CoreML",
+    "metadataOutputVersion" : "3.0",
+    "outputSchema" : [
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16)",
+        "shortDescription" : "",
+        "shape" : "[]",
+        "name" : "hidden_states",
+        "type" : "MultiArray"
+      }
+    ],
+    "version" : "0.3.0",
+    "modelParameters" : [
+    ],
+    "author" : "Converted with Anemll v0.3.0",
+    "specificationVersion" : 9,
+    "storagePrecision" : "Float16",
+    "mlProgramOperationTypeHistogram" : {
+      "Ios18.gather" : 1
+    },
+    "computePrecision" : "Mixed (Float16, Int32)",
+    "stateSchema" : [
+    ],
+    "isUpdatable" : "0",
+    "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
+    },
+    "modelType" : {
+      "name" : "MLModelType_mlProgram"
+    },
+    "inputSchema" : [
+      {
+        "shortDescription" : "",
+        "dataType" : "Int32",
+        "hasShapeFlexibility" : "1",
+        "isOptional" : "0",
+        "shapeFlexibility" : "1 × 1 | 1 × 64",
+        "formattedType" : "MultiArray (Int32 1 × 1)",
+        "type" : "MultiArray",
+        "shape" : "[1, 1]",
+        "name" : "input_ids",
+        "enumeratedShapes" : "[[1, 1], [1, 64]]"
+      }
+    ],
+    "userDefinedMetadata" : {
+      "com.github.apple.coremltools.source_dialect" : "TorchScript",
+      "com.github.apple.coremltools.version" : "8.2",
+      "com.github.apple.coremltools.source" : "torch==2.5.0",
+      "com.anemll.info" : "Converted with Anemll v0.3.0",
+      "com.anemll.context_length" : "512"
+    },
+    "generatedClassName" : "llama_embeddings",
+    "method" : "predict"
+  }
+]

llama_embeddings.mlmodelc/model.mil ADDED Viewed

	@@ -0,0 +1,11 @@

+program(1.3)
+[buildInfo = dict<string, string>({{"coremlc-component-MIL", "3404.16.1"}, {"coremlc-version", "3404.23.1"}, {"coremltools-component-torch", "2.5.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "8.2"}})]
+{
+    func main<ios18>(tensor<int32, [1, ?]> input_ids) [FlexibleShapeInformation = tuple<tuple<string, dict<string, tensor<int32, [?]>>>, tuple<string, dict<string, dict<string, tensor<int32, [?]>>>>>((("DefaultShapes", {{"input_ids", [1, 1]}}), ("EnumeratedShapes", {{"79ae981e", {{"input_ids", [1, 1]}}}, {"ed9b58c8", {{"input_ids", [1, 64]}}}})))] {
+            int32 hidden_states_axis_0 = const()[name = string("hidden_states_axis_0"), val = int32(0)];
+            int32 hidden_states_batch_dims_0 = const()[name = string("hidden_states_batch_dims_0"), val = int32(0)];
+            bool hidden_states_validate_indices_0 = const()[name = string("hidden_states_validate_indices_0"), val = bool(false)];
+            tensor<fp16, [128256, 2048]> embed_tokens_weight_to_fp16 = const()[name = string("embed_tokens_weight_to_fp16"), val = tensor<fp16, [128256, 2048]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))];
+            tensor<fp16, [1, ?, 2048]> hidden_states = gather(axis = hidden_states_axis_0, batch_dims = hidden_states_batch_dims_0, indices = input_ids, validate_indices = hidden_states_validate_indices_0, x = embed_tokens_weight_to_fp16)[name = string("hidden_states_cast_fp16")];
+        } -> (hidden_states);
+}

llama_embeddings.mlmodelc/weights/weight.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:73f76c5cbd933c0ee67f251d2278431346670fa90b5891d58ffd859af8e8003e
+size 525336704

llama_lm_head.mlmodelc/analytics/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:aa2e8e924a2534f234a920ffe4f9afc1629c3ff6c190980405a0bdf2884c4759
+size 243

llama_lm_head.mlmodelc/coremldata.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e822b459108468fb68f202bad1a864bf524496ad4dabf0f097da5a3d4ef43edd
+size 661

llama_lm_head.mlmodelc/metadata.json ADDED Viewed

	@@ -0,0 +1,138 @@

+[
+  {
+    "shortDescription" : "Anemll Model (LM Head) converted to CoreML",
+    "metadataOutputVersion" : "3.0",
+    "outputSchema" : [
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits1",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits2",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits3",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits4",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits5",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits6",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits7",
+        "type" : "MultiArray"
+      },
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 16032)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 16032]",
+        "name" : "logits8",
+        "type" : "MultiArray"
+      }
+    ],
+    "version" : "0.3.0",
+    "modelParameters" : [
+    ],
+    "author" : "Converted with Anemll v0.3.0",
+    "specificationVersion" : 9,
+    "storagePrecision" : "Float16",
+    "mlProgramOperationTypeHistogram" : {
+      "Ios18.transpose" : 9,
+      "Ios18.expandDims" : 1,
+      "Ios18.conv" : 8,
+      "Ios18.squeeze" : 8
+    },
+    "computePrecision" : "Mixed (Float16, Int32)",
+    "stateSchema" : [
+    ],
+    "isUpdatable" : "0",
+    "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
+    },
+    "modelType" : {
+      "name" : "MLModelType_mlProgram"
+    },
+    "inputSchema" : [
+      {
+        "hasShapeFlexibility" : "0",
+        "isOptional" : "0",
+        "dataType" : "Float16",
+        "formattedType" : "MultiArray (Float16 1 × 1 × 2048)",
+        "shortDescription" : "",
+        "shape" : "[1, 1, 2048]",
+        "name" : "hidden_states",
+        "type" : "MultiArray"
+      }
+    ],
+    "userDefinedMetadata" : {
+      "com.github.apple.coremltools.source_dialect" : "TorchScript",
+      "com.github.apple.coremltools.version" : "8.2",
+      "com.github.apple.coremltools.source" : "torch==2.5.0",
+      "com.anemll.info" : "Converted with Anemll v0.3.0",
+      "com.anemll.context_length" : "512"
+    },
+    "generatedClassName" : "llama_lm_head",
+    "method" : "predict"
+  }
+]

llama_lm_head.mlmodelc/model.mil ADDED Viewed

	@@ -0,0 +1,98 @@

+program(1.3)
+[buildInfo = dict<string, string>({{"coremlc-component-MIL", "3404.16.1"}, {"coremlc-version", "3404.23.1"}, {"coremltools-component-torch", "2.5.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "8.2"}})]
+{
+    func main<ios18>(tensor<fp16, [1, 1, 2048]> hidden_states) {
+            tensor<int32, [3]> var_5 = const()[name = string("op_5"), val = tensor<int32, [3]>([0, 2, 1])];
+            tensor<int32, [1]> input_axes_0 = const()[name = string("input_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 2048, 1]> var_6_cast_fp16 = transpose(perm = var_5, x = hidden_states)[name = string("transpose_8")];
+            tensor<fp16, [1, 2048, 1, 1]> input_cast_fp16 = expand_dims(axes = input_axes_0, x = var_6_cast_fp16)[name = string("input_cast_fp16")];
+            string var_29_pad_type_0 = const()[name = string("op_29_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_29_strides_0 = const()[name = string("op_29_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_29_pad_0 = const()[name = string("op_29_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_29_dilations_0 = const()[name = string("op_29_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_29_groups_0 = const()[name = string("op_29_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_9_promoted_to_fp16 = const()[name = string("op_9_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_29_cast_fp16 = conv(dilations = var_29_dilations_0, groups = var_29_groups_0, pad = var_29_pad_0, pad_type = var_29_pad_type_0, strides = var_29_strides_0, weight = var_9_promoted_to_fp16, x = input_cast_fp16)[name = string("op_29_cast_fp16")];
+            tensor<int32, [1]> var_31_axes_0 = const()[name = string("op_31_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_31_cast_fp16 = squeeze(axes = var_31_axes_0, x = var_29_cast_fp16)[name = string("op_31_cast_fp16")];
+            tensor<int32, [3]> var_34_perm_0 = const()[name = string("op_34_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            string var_55_pad_type_0 = const()[name = string("op_55_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_55_strides_0 = const()[name = string("op_55_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_55_pad_0 = const()[name = string("op_55_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_55_dilations_0 = const()[name = string("op_55_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_55_groups_0 = const()[name = string("op_55_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_35_promoted_to_fp16 = const()[name = string("op_35_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65667200)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_55_cast_fp16 = conv(dilations = var_55_dilations_0, groups = var_55_groups_0, pad = var_55_pad_0, pad_type = var_55_pad_type_0, strides = var_55_strides_0, weight = var_35_promoted_to_fp16, x = input_cast_fp16)[name = string("op_55_cast_fp16")];
+            tensor<int32, [1]> var_57_axes_0 = const()[name = string("op_57_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_57_cast_fp16 = squeeze(axes = var_57_axes_0, x = var_55_cast_fp16)[name = string("op_57_cast_fp16")];
+            tensor<int32, [3]> var_60_perm_0 = const()[name = string("op_60_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            string var_81_pad_type_0 = const()[name = string("op_81_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_81_strides_0 = const()[name = string("op_81_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_81_pad_0 = const()[name = string("op_81_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_81_dilations_0 = const()[name = string("op_81_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_81_groups_0 = const()[name = string("op_81_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_61_promoted_to_fp16 = const()[name = string("op_61_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(131334336)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_81_cast_fp16 = conv(dilations = var_81_dilations_0, groups = var_81_groups_0, pad = var_81_pad_0, pad_type = var_81_pad_type_0, strides = var_81_strides_0, weight = var_61_promoted_to_fp16, x = input_cast_fp16)[name = string("op_81_cast_fp16")];
+            tensor<int32, [1]> var_83_axes_0 = const()[name = string("op_83_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_83_cast_fp16 = squeeze(axes = var_83_axes_0, x = var_81_cast_fp16)[name = string("op_83_cast_fp16")];
+            tensor<int32, [3]> var_86_perm_0 = const()[name = string("op_86_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            string var_107_pad_type_0 = const()[name = string("op_107_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_107_strides_0 = const()[name = string("op_107_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_107_pad_0 = const()[name = string("op_107_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_107_dilations_0 = const()[name = string("op_107_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_107_groups_0 = const()[name = string("op_107_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_87_promoted_to_fp16 = const()[name = string("op_87_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197001472)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_107_cast_fp16 = conv(dilations = var_107_dilations_0, groups = var_107_groups_0, pad = var_107_pad_0, pad_type = var_107_pad_type_0, strides = var_107_strides_0, weight = var_87_promoted_to_fp16, x = input_cast_fp16)[name = string("op_107_cast_fp16")];
+            tensor<int32, [1]> var_109_axes_0 = const()[name = string("op_109_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_109_cast_fp16 = squeeze(axes = var_109_axes_0, x = var_107_cast_fp16)[name = string("op_109_cast_fp16")];
+            tensor<int32, [3]> var_112_perm_0 = const()[name = string("op_112_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            string var_133_pad_type_0 = const()[name = string("op_133_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_133_strides_0 = const()[name = string("op_133_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_133_pad_0 = const()[name = string("op_133_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_133_dilations_0 = const()[name = string("op_133_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_133_groups_0 = const()[name = string("op_133_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_113_promoted_to_fp16 = const()[name = string("op_113_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(262668608)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_133_cast_fp16 = conv(dilations = var_133_dilations_0, groups = var_133_groups_0, pad = var_133_pad_0, pad_type = var_133_pad_type_0, strides = var_133_strides_0, weight = var_113_promoted_to_fp16, x = input_cast_fp16)[name = string("op_133_cast_fp16")];
+            tensor<int32, [1]> var_135_axes_0 = const()[name = string("op_135_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_135_cast_fp16 = squeeze(axes = var_135_axes_0, x = var_133_cast_fp16)[name = string("op_135_cast_fp16")];
+            tensor<int32, [3]> var_138_perm_0 = const()[name = string("op_138_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            string var_159_pad_type_0 = const()[name = string("op_159_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_159_strides_0 = const()[name = string("op_159_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_159_pad_0 = const()[name = string("op_159_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_159_dilations_0 = const()[name = string("op_159_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_159_groups_0 = const()[name = string("op_159_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_139_promoted_to_fp16 = const()[name = string("op_139_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(328335744)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_159_cast_fp16 = conv(dilations = var_159_dilations_0, groups = var_159_groups_0, pad = var_159_pad_0, pad_type = var_159_pad_type_0, strides = var_159_strides_0, weight = var_139_promoted_to_fp16, x = input_cast_fp16)[name = string("op_159_cast_fp16")];
+            tensor<int32, [1]> var_161_axes_0 = const()[name = string("op_161_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_161_cast_fp16 = squeeze(axes = var_161_axes_0, x = var_159_cast_fp16)[name = string("op_161_cast_fp16")];
+            tensor<int32, [3]> var_164_perm_0 = const()[name = string("op_164_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            string var_185_pad_type_0 = const()[name = string("op_185_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_185_strides_0 = const()[name = string("op_185_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_185_pad_0 = const()[name = string("op_185_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_185_dilations_0 = const()[name = string("op_185_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_185_groups_0 = const()[name = string("op_185_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_165_promoted_to_fp16 = const()[name = string("op_165_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(394002880)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_185_cast_fp16 = conv(dilations = var_185_dilations_0, groups = var_185_groups_0, pad = var_185_pad_0, pad_type = var_185_pad_type_0, strides = var_185_strides_0, weight = var_165_promoted_to_fp16, x = input_cast_fp16)[name = string("op_185_cast_fp16")];
+            tensor<int32, [1]> var_187_axes_0 = const()[name = string("op_187_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_187_cast_fp16 = squeeze(axes = var_187_axes_0, x = var_185_cast_fp16)[name = string("op_187_cast_fp16")];
+            tensor<int32, [3]> var_190_perm_0 = const()[name = string("op_190_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            string var_211_pad_type_0 = const()[name = string("op_211_pad_type_0"), val = string("valid")];
+            tensor<int32, [2]> var_211_strides_0 = const()[name = string("op_211_strides_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [4]> var_211_pad_0 = const()[name = string("op_211_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
+            tensor<int32, [2]> var_211_dilations_0 = const()[name = string("op_211_dilations_0"), val = tensor<int32, [2]>([1, 1])];
+            int32 var_211_groups_0 = const()[name = string("op_211_groups_0"), val = int32(1)];
+            tensor<fp16, [16032, 2048, 1, 1]> var_191_promoted_to_fp16 = const()[name = string("op_191_promoted_to_fp16"), val = tensor<fp16, [16032, 2048, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(459670016)))];
+            tensor<fp16, [1, 16032, 1, 1]> var_211_cast_fp16 = conv(dilations = var_211_dilations_0, groups = var_211_groups_0, pad = var_211_pad_0, pad_type = var_211_pad_type_0, strides = var_211_strides_0, weight = var_191_promoted_to_fp16, x = input_cast_fp16)[name = string("op_211_cast_fp16")];
+            tensor<int32, [1]> var_213_axes_0 = const()[name = string("op_213_axes_0"), val = tensor<int32, [1]>([2])];
+            tensor<fp16, [1, 16032, 1]> var_213_cast_fp16 = squeeze(axes = var_213_axes_0, x = var_211_cast_fp16)[name = string("op_213_cast_fp16")];
+            tensor<int32, [3]> var_216_perm_0 = const()[name = string("op_216_perm_0"), val = tensor<int32, [3]>([0, 2, 1])];
+            tensor<fp16, [1, 1, 16032]> logits8 = transpose(perm = var_216_perm_0, x = var_213_cast_fp16)[name = string("transpose_0")];
+            tensor<fp16, [1, 1, 16032]> logits7 = transpose(perm = var_190_perm_0, x = var_187_cast_fp16)[name = string("transpose_1")];
+            tensor<fp16, [1, 1, 16032]> logits6 = transpose(perm = var_164_perm_0, x = var_161_cast_fp16)[name = string("transpose_2")];
+            tensor<fp16, [1, 1, 16032]> logits5 = transpose(perm = var_138_perm_0, x = var_135_cast_fp16)[name = string("transpose_3")];
+            tensor<fp16, [1, 1, 16032]> logits4 = transpose(perm = var_112_perm_0, x = var_109_cast_fp16)[name = string("transpose_4")];
+            tensor<fp16, [1, 1, 16032]> logits3 = transpose(perm = var_86_perm_0, x = var_83_cast_fp16)[name = string("transpose_5")];
+            tensor<fp16, [1, 1, 16032]> logits2 = transpose(perm = var_60_perm_0, x = var_57_cast_fp16)[name = string("transpose_6")];
+            tensor<fp16, [1, 1, 16032]> logits1 = transpose(perm = var_34_perm_0, x = var_31_cast_fp16)[name = string("transpose_7")];
+        } -> (logits1, logits2, logits3, logits4, logits5, logits6, logits7, logits8);
+}

llama_lm_head.mlmodelc/weights/weight.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:aa8e709ced9796bd998ebb1f358ebb9d865b85c921b757639b0f43bc5a59fc7a
+size 525337152

meta.yaml ADDED Viewed

	@@ -0,0 +1,23 @@

+model_info:
+  name: anemll-Meta-Llama-3.2-1B-ctx512
+  version: 0.3.0
+  description: |
+    Demonstarates running Meta-Llama-3.2-1B on Apple Neural Engine
+    Context length: 512
+    Batch size: 64
+    Chunks: 2
+  license: MIT
+  author: Anemll
+  framework: Core ML
+  language: Python
+  parameters:
+    context_length: 512
+    batch_size: 64
+    lut_embeddings: none
+    lut_ffn: none
+    lut_lmhead: none
+    num_chunks: 2
+    model_prefix: llama
+    embeddings: llama_embeddings.mlmodelc
+    lm_head: llama_lm_head.mlmodelc
+    ffn: llama_FFN_PF.mlmodelc

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,2062 @@

+{
+  "added_tokens_decoder": {
+    "128000": {
+      "content": "<|begin_of_text|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128001": {
+      "content": "<|end_of_text|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128002": {
+      "content": "<|reserved_special_token_0|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128003": {
+      "content": "<|reserved_special_token_1|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128004": {
+      "content": "<|finetune_right_pad_id|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128005": {
+      "content": "<|reserved_special_token_2|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128006": {
+      "content": "<|start_header_id|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128007": {
+      "content": "<|end_header_id|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128008": {
+      "content": "<|eom_id|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128009": {
+      "content": "<|eot_id|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128010": {
+      "content": "<|python_tag|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128011": {
+      "content": "<|reserved_special_token_3|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128012": {
+      "content": "<|reserved_special_token_4|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128013": {
+      "content": "<|reserved_special_token_5|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128014": {
+      "content": "<|reserved_special_token_6|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128015": {
+      "content": "<|reserved_special_token_7|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128016": {
+      "content": "<|reserved_special_token_8|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128017": {
+      "content": "<|reserved_special_token_9|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128018": {
+      "content": "<|reserved_special_token_10|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128019": {
+      "content": "<|reserved_special_token_11|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128020": {
+      "content": "<|reserved_special_token_12|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128021": {
+      "content": "<|reserved_special_token_13|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128022": {
+      "content": "<|reserved_special_token_14|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128023": {
+      "content": "<|reserved_special_token_15|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128024": {
+      "content": "<|reserved_special_token_16|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128025": {
+      "content": "<|reserved_special_token_17|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128026": {
+      "content": "<|reserved_special_token_18|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128027": {
+      "content": "<|reserved_special_token_19|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128028": {
+      "content": "<|reserved_special_token_20|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128029": {
+      "content": "<|reserved_special_token_21|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128030": {
+      "content": "<|reserved_special_token_22|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128031": {
+      "content": "<|reserved_special_token_23|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128032": {
+      "content": "<|reserved_special_token_24|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128033": {
+      "content": "<|reserved_special_token_25|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128034": {
+      "content": "<|reserved_special_token_26|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128035": {
+      "content": "<|reserved_special_token_27|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128036": {
+      "content": "<|reserved_special_token_28|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128037": {
+      "content": "<|reserved_special_token_29|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128038": {
+      "content": "<|reserved_special_token_30|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128039": {
+      "content": "<|reserved_special_token_31|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128040": {
+      "content": "<|reserved_special_token_32|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128041": {
+      "content": "<|reserved_special_token_33|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128042": {
+      "content": "<|reserved_special_token_34|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128043": {
+      "content": "<|reserved_special_token_35|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128044": {
+      "content": "<|reserved_special_token_36|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128045": {
+      "content": "<|reserved_special_token_37|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128046": {
+      "content": "<|reserved_special_token_38|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128047": {
+      "content": "<|reserved_special_token_39|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128048": {
+      "content": "<|reserved_special_token_40|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128049": {
+      "content": "<|reserved_special_token_41|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128050": {
+      "content": "<|reserved_special_token_42|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128051": {
+      "content": "<|reserved_special_token_43|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128052": {
+      "content": "<|reserved_special_token_44|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128053": {
+      "content": "<|reserved_special_token_45|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128054": {
+      "content": "<|reserved_special_token_46|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128055": {
+      "content": "<|reserved_special_token_47|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128056": {
+      "content": "<|reserved_special_token_48|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128057": {
+      "content": "<|reserved_special_token_49|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128058": {
+      "content": "<|reserved_special_token_50|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128059": {
+      "content": "<|reserved_special_token_51|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128060": {
+      "content": "<|reserved_special_token_52|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128061": {
+      "content": "<|reserved_special_token_53|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128062": {
+      "content": "<|reserved_special_token_54|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128063": {
+      "content": "<|reserved_special_token_55|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128064": {
+      "content": "<|reserved_special_token_56|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128065": {
+      "content": "<|reserved_special_token_57|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128066": {
+      "content": "<|reserved_special_token_58|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128067": {
+      "content": "<|reserved_special_token_59|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128068": {
+      "content": "<|reserved_special_token_60|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128069": {
+      "content": "<|reserved_special_token_61|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128070": {
+      "content": "<|reserved_special_token_62|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128071": {
+      "content": "<|reserved_special_token_63|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128072": {
+      "content": "<|reserved_special_token_64|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128073": {
+      "content": "<|reserved_special_token_65|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128074": {
+      "content": "<|reserved_special_token_66|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128075": {
+      "content": "<|reserved_special_token_67|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128076": {
+      "content": "<|reserved_special_token_68|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128077": {
+      "content": "<|reserved_special_token_69|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128078": {
+      "content": "<|reserved_special_token_70|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128079": {
+      "content": "<|reserved_special_token_71|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128080": {
+      "content": "<|reserved_special_token_72|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128081": {
+      "content": "<|reserved_special_token_73|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128082": {
+      "content": "<|reserved_special_token_74|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128083": {
+      "content": "<|reserved_special_token_75|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128084": {
+      "content": "<|reserved_special_token_76|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128085": {
+      "content": "<|reserved_special_token_77|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128086": {
+      "content": "<|reserved_special_token_78|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128087": {
+      "content": "<|reserved_special_token_79|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128088": {
+      "content": "<|reserved_special_token_80|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128089": {
+      "content": "<|reserved_special_token_81|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128090": {
+      "content": "<|reserved_special_token_82|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128091": {
+      "content": "<|reserved_special_token_83|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128092": {
+      "content": "<|reserved_special_token_84|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128093": {
+      "content": "<|reserved_special_token_85|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128094": {
+      "content": "<|reserved_special_token_86|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128095": {
+      "content": "<|reserved_special_token_87|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128096": {
+      "content": "<|reserved_special_token_88|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128097": {
+      "content": "<|reserved_special_token_89|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128098": {
+      "content": "<|reserved_special_token_90|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128099": {
+      "content": "<|reserved_special_token_91|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128100": {
+      "content": "<|reserved_special_token_92|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128101": {
+      "content": "<|reserved_special_token_93|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128102": {
+      "content": "<|reserved_special_token_94|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128103": {
+      "content": "<|reserved_special_token_95|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128104": {
+      "content": "<|reserved_special_token_96|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128105": {
+      "content": "<|reserved_special_token_97|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128106": {
+      "content": "<|reserved_special_token_98|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128107": {
+      "content": "<|reserved_special_token_99|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128108": {
+      "content": "<|reserved_special_token_100|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128109": {
+      "content": "<|reserved_special_token_101|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128110": {
+      "content": "<|reserved_special_token_102|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128111": {
+      "content": "<|reserved_special_token_103|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128112": {
+      "content": "<|reserved_special_token_104|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128113": {
+      "content": "<|reserved_special_token_105|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128114": {
+      "content": "<|reserved_special_token_106|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128115": {
+      "content": "<|reserved_special_token_107|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128116": {
+      "content": "<|reserved_special_token_108|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128117": {
+      "content": "<|reserved_special_token_109|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128118": {
+      "content": "<|reserved_special_token_110|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128119": {
+      "content": "<|reserved_special_token_111|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128120": {
+      "content": "<|reserved_special_token_112|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128121": {
+      "content": "<|reserved_special_token_113|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128122": {
+      "content": "<|reserved_special_token_114|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128123": {
+      "content": "<|reserved_special_token_115|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128124": {
+      "content": "<|reserved_special_token_116|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128125": {
+      "content": "<|reserved_special_token_117|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128126": {
+      "content": "<|reserved_special_token_118|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128127": {
+      "content": "<|reserved_special_token_119|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128128": {
+      "content": "<|reserved_special_token_120|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128129": {
+      "content": "<|reserved_special_token_121|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128130": {
+      "content": "<|reserved_special_token_122|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128131": {
+      "content": "<|reserved_special_token_123|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128132": {
+      "content": "<|reserved_special_token_124|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128133": {
+      "content": "<|reserved_special_token_125|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128134": {
+      "content": "<|reserved_special_token_126|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128135": {
+      "content": "<|reserved_special_token_127|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128136": {
+      "content": "<|reserved_special_token_128|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128137": {
+      "content": "<|reserved_special_token_129|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128138": {
+      "content": "<|reserved_special_token_130|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128139": {
+      "content": "<|reserved_special_token_131|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128140": {
+      "content": "<|reserved_special_token_132|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128141": {
+      "content": "<|reserved_special_token_133|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128142": {
+      "content": "<|reserved_special_token_134|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128143": {
+      "content": "<|reserved_special_token_135|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128144": {
+      "content": "<|reserved_special_token_136|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128145": {
+      "content": "<|reserved_special_token_137|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128146": {
+      "content": "<|reserved_special_token_138|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128147": {
+      "content": "<|reserved_special_token_139|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128148": {
+      "content": "<|reserved_special_token_140|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128149": {
+      "content": "<|reserved_special_token_141|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128150": {
+      "content": "<|reserved_special_token_142|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128151": {
+      "content": "<|reserved_special_token_143|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128152": {
+      "content": "<|reserved_special_token_144|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128153": {
+      "content": "<|reserved_special_token_145|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128154": {
+      "content": "<|reserved_special_token_146|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128155": {
+      "content": "<|reserved_special_token_147|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128156": {
+      "content": "<|reserved_special_token_148|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128157": {
+      "content": "<|reserved_special_token_149|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128158": {
+      "content": "<|reserved_special_token_150|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128159": {
+      "content": "<|reserved_special_token_151|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128160": {
+      "content": "<|reserved_special_token_152|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128161": {
+      "content": "<|reserved_special_token_153|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128162": {
+      "content": "<|reserved_special_token_154|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128163": {
+      "content": "<|reserved_special_token_155|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128164": {
+      "content": "<|reserved_special_token_156|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128165": {
+      "content": "<|reserved_special_token_157|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128166": {
+      "content": "<|reserved_special_token_158|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128167": {
+      "content": "<|reserved_special_token_159|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128168": {
+      "content": "<|reserved_special_token_160|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128169": {
+      "content": "<|reserved_special_token_161|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128170": {
+      "content": "<|reserved_special_token_162|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128171": {
+      "content": "<|reserved_special_token_163|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128172": {
+      "content": "<|reserved_special_token_164|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128173": {
+      "content": "<|reserved_special_token_165|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128174": {
+      "content": "<|reserved_special_token_166|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128175": {
+      "content": "<|reserved_special_token_167|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128176": {
+      "content": "<|reserved_special_token_168|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128177": {
+      "content": "<|reserved_special_token_169|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128178": {
+      "content": "<|reserved_special_token_170|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128179": {
+      "content": "<|reserved_special_token_171|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128180": {
+      "content": "<|reserved_special_token_172|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128181": {
+      "content": "<|reserved_special_token_173|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128182": {
+      "content": "<|reserved_special_token_174|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128183": {
+      "content": "<|reserved_special_token_175|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128184": {
+      "content": "<|reserved_special_token_176|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128185": {
+      "content": "<|reserved_special_token_177|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128186": {
+      "content": "<|reserved_special_token_178|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128187": {
+      "content": "<|reserved_special_token_179|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128188": {
+      "content": "<|reserved_special_token_180|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128189": {
+      "content": "<|reserved_special_token_181|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128190": {
+      "content": "<|reserved_special_token_182|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128191": {
+      "content": "<|reserved_special_token_183|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128192": {
+      "content": "<|reserved_special_token_184|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128193": {
+      "content": "<|reserved_special_token_185|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128194": {
+      "content": "<|reserved_special_token_186|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128195": {
+      "content": "<|reserved_special_token_187|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128196": {
+      "content": "<|reserved_special_token_188|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128197": {
+      "content": "<|reserved_special_token_189|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128198": {
+      "content": "<|reserved_special_token_190|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128199": {
+      "content": "<|reserved_special_token_191|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128200": {
+      "content": "<|reserved_special_token_192|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128201": {
+      "content": "<|reserved_special_token_193|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128202": {
+      "content": "<|reserved_special_token_194|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128203": {
+      "content": "<|reserved_special_token_195|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128204": {
+      "content": "<|reserved_special_token_196|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128205": {
+      "content": "<|reserved_special_token_197|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128206": {
+      "content": "<|reserved_special_token_198|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128207": {
+      "content": "<|reserved_special_token_199|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128208": {
+      "content": "<|reserved_special_token_200|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128209": {
+      "content": "<|reserved_special_token_201|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128210": {
+      "content": "<|reserved_special_token_202|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128211": {
+      "content": "<|reserved_special_token_203|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128212": {
+      "content": "<|reserved_special_token_204|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128213": {
+      "content": "<|reserved_special_token_205|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128214": {
+      "content": "<|reserved_special_token_206|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128215": {
+      "content": "<|reserved_special_token_207|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128216": {
+      "content": "<|reserved_special_token_208|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128217": {
+      "content": "<|reserved_special_token_209|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128218": {
+      "content": "<|reserved_special_token_210|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128219": {
+      "content": "<|reserved_special_token_211|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128220": {
+      "content": "<|reserved_special_token_212|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128221": {
+      "content": "<|reserved_special_token_213|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128222": {
+      "content": "<|reserved_special_token_214|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128223": {
+      "content": "<|reserved_special_token_215|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128224": {
+      "content": "<|reserved_special_token_216|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128225": {
+      "content": "<|reserved_special_token_217|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128226": {
+      "content": "<|reserved_special_token_218|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128227": {
+      "content": "<|reserved_special_token_219|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128228": {
+      "content": "<|reserved_special_token_220|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128229": {
+      "content": "<|reserved_special_token_221|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128230": {
+      "content": "<|reserved_special_token_222|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128231": {
+      "content": "<|reserved_special_token_223|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128232": {
+      "content": "<|reserved_special_token_224|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128233": {
+      "content": "<|reserved_special_token_225|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128234": {
+      "content": "<|reserved_special_token_226|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128235": {
+      "content": "<|reserved_special_token_227|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128236": {
+      "content": "<|reserved_special_token_228|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128237": {
+      "content": "<|reserved_special_token_229|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128238": {
+      "content": "<|reserved_special_token_230|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128239": {
+      "content": "<|reserved_special_token_231|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128240": {
+      "content": "<|reserved_special_token_232|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128241": {
+      "content": "<|reserved_special_token_233|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128242": {
+      "content": "<|reserved_special_token_234|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128243": {
+      "content": "<|reserved_special_token_235|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128244": {
+      "content": "<|reserved_special_token_236|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128245": {
+      "content": "<|reserved_special_token_237|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128246": {
+      "content": "<|reserved_special_token_238|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128247": {
+      "content": "<|reserved_special_token_239|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128248": {
+      "content": "<|reserved_special_token_240|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128249": {
+      "content": "<|reserved_special_token_241|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128250": {
+      "content": "<|reserved_special_token_242|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128251": {
+      "content": "<|reserved_special_token_243|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128252": {
+      "content": "<|reserved_special_token_244|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128253": {
+      "content": "<|reserved_special_token_245|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128254": {
+      "content": "<|reserved_special_token_246|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128255": {
+      "content": "<|reserved_special_token_247|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<|begin_of_text|>",
+  "chat_template": "{{- bos_token }}\n{%- if custom_tools is defined %}\n    {%- set tools = custom_tools %}\n{%- endif %}\n{%- if not tools_in_user_message is defined %}\n    {%- set tools_in_user_message = true %}\n{%- endif %}\n{%- if not date_string is defined %}\n    {%- if strftime_now is defined %}\n        {%- set date_string = strftime_now(\"%d %b %Y\") %}\n    {%- else %}\n        {%- set date_string = \"26 Jul 2024\" %}\n    {%- endif %}\n{%- endif %}\n{%- if not tools is defined %}\n    {%- set tools = none %}\n{%- endif %}\n\n{#- This block extracts the system message, so we can slot it into the right place. #}\n{%- if messages[0]['role'] == 'system' %}\n    {%- set system_message = messages[0]['content']|trim %}\n    {%- set messages = messages[1:] %}\n{%- else %}\n    {%- set system_message = \"\" %}\n{%- endif %}\n\n{#- System message #}\n{{- \"<|start_header_id|>system<|end_header_id|>\\n\\n\" }}\n{%- if tools is not none %}\n    {{- \"Environment: ipython\\n\" }}\n{%- endif %}\n{{- \"Cutting Knowledge Date: December 2023\\n\" }}\n{{- \"Today Date: \" + date_string + \"\\n\\n\" }}\n{%- if tools is not none and not tools_in_user_message %}\n    {{- \"You have access to the following functions. To call a function, please respond with JSON for a function call.\" }}\n    {{- 'Respond in the format {\"name\": function name, \"parameters\": dictionary of argument name and its value}.' }}\n    {{- \"Do not use variables.\\n\\n\" }}\n    {%- for t in tools %}\n        {{- t | tojson(indent=4) }}\n        {{- \"\\n\\n\" }}\n    {%- endfor %}\n{%- endif %}\n{{- system_message }}\n{{- \"<|eot_id|>\" }}\n\n{#- Custom tools are passed in a user message with some extra guidance #}\n{%- if tools_in_user_message and not tools is none %}\n    {#- Extract the first user message so we can plug it in here #}\n    {%- if messages | length != 0 %}\n        {%- set first_user_message = messages[0]['content']|trim %}\n        {%- set messages = messages[1:] %}\n    {%- else %}\n        {{- raise_exception(\"Cannot put tools in the first user message when there's no first user message!\") }}\n{%- endif %}\n    {{- '<|start_header_id|>user<|end_header_id|>\\n\\n' -}}\n    {{- \"Given the following functions, please respond with a JSON for a function call \" }}\n    {{- \"with its proper arguments that best answers the given prompt.\\n\\n\" }}\n    {{- 'Respond in the format {\"name\": function name, \"parameters\": dictionary of argument name and its value}.' }}\n    {{- \"Do not use variables.\\n\\n\" }}\n    {%- for t in tools %}\n        {{- t | tojson(indent=4) }}\n        {{- \"\\n\\n\" }}\n    {%- endfor %}\n    {{- first_user_message + \"<|eot_id|>\"}}\n{%- endif %}\n\n{%- for message in messages %}\n    {%- if not (message.role == 'ipython' or message.role == 'tool' or 'tool_calls' in message) %}\n        {{- '<|start_header_id|>' + message['role'] + '<|end_header_id|>\\n\\n'+ message['content'] | trim + '<|eot_id|>' }}\n    {%- elif 'tool_calls' in message %}\n        {%- if not message.tool_calls|length == 1 %}\n            {{- raise_exception(\"This model only supports single tool-calls at once!\") }}\n        {%- endif %}\n        {%- set tool_call = message.tool_calls[0].function %}\n        {{- '<|start_header_id|>assistant<|end_header_id|>\\n\\n' -}}\n        {{- '{\"name\": \"' + tool_call.name + '\", ' }}\n        {{- '\"parameters\": ' }}\n        {{- tool_call.arguments | tojson }}\n        {{- \"}\" }}\n        {{- \"<|eot_id|>\" }}\n    {%- elif message.role == \"tool\" or message.role == \"ipython\" %}\n        {{- \"<|start_header_id|>ipython<|end_header_id|>\\n\\n\" }}\n        {%- if message.content is mapping or message.content is iterable %}\n            {{- message.content | tojson }}\n        {%- else %}\n            {{- message.content }}\n        {%- endif %}\n        {{- \"<|eot_id|>\" }}\n    {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n    {{- '<|start_header_id|>assistant<|end_header_id|>\\n\\n' }}\n{%- endif %}\n",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "<|eot_id|>",
+  "model_input_names": [
+    "input_ids",
+    "attention_mask"
+  ],
+  "model_max_length": 131072,
+  "tokenizer_class": "PreTrainedTokenizerFast"
+}