Upload folder using huggingface_hub

Browse files

Files changed (15) hide show

rwkv7-0.4B-g1-respark-voice-tunable/__init__.py +0 -0
rwkv7-0.4B-g1-respark-voice-tunable/config.json +66 -0
rwkv7-0.4B-g1-respark-voice-tunable/configuration_rwkv7.py +91 -0
rwkv7-0.4B-g1-respark-voice-tunable/generation_config.json +6 -0
rwkv7-0.4B-g1-respark-voice-tunable/hf_rwkv_tokenizer.py +280 -0
rwkv7-0.4B-g1-respark-voice-tunable/model.safetensors +3 -0
rwkv7-0.4B-g1-respark-voice-tunable/model_converted.pth +3 -0
rwkv7-0.4B-g1-respark-voice-tunable/model_padded.pth +3 -0
rwkv7-0.4B-g1-respark-voice-tunable/modeling_rwkv7.py +4 -0
rwkv7-0.4B-g1-respark-voice-tunable/modeling_rwkvspeech.py +6 -0
rwkv7-0.4B-g1-respark-voice-tunable/rwkv_vocab_v20230424.txt +0 -0
rwkv7-0.4B-g1-respark-voice-tunable/spark_llm.py +202 -0
rwkv7-0.4B-g1-respark-voice-tunable/special_tokens_map.json +24 -0
rwkv7-0.4B-g1-respark-voice-tunable/tokenizer_config.json +836 -0
rwkv7-0.4B-g1-respark-voice-tunable/vocab.txt +0 -0

rwkv7-0.4B-g1-respark-voice-tunable/__init__.py ADDED Viewed

File without changes

rwkv7-0.4B-g1-respark-voice-tunable/config.json ADDED Viewed

	@@ -0,0 +1,66 @@

+{
+  "a_low_rank_dim": 64,
+  "architectures": [
+    "RWKV7ForSpeech"
+  ],
+  "attn": null,
+  "attn_mode": "chunk",
+  "audio_global_vocab_size": 4096,
+  "auto_map": {
+    "AutoConfig": "spark_llm.RWKV7SpeechConfig",
+    "AutoModel": "spark_llm.RWKV7Model",
+    "AutoModelForCausalLM": "spark_llm.RWKV7ForSpeech"
+  },
+  "bos_token_id": 0,
+  "decay_low_rank_dim": 64,
+  "eos_token_id": 0,
+  "fuse_cross_entropy": true,
+  "fuse_norm": false,
+  "gate_low_rank_dim": 128,
+  "head_dim": 64,
+  "hidden_act": "sqrelu",
+  "hidden_ratio": 4.0,
+  "hidden_size": 1024,
+  "initializer_range": 0.006,
+  "intermediate_size": 4096,
+  "max_position_embeddings": 2048,
+  "model_type": "rwkv7",
+  "norm_bias": true,
+  "norm_eps": 1e-05,
+  "norm_first": true,
+  "num_heads": 32,
+  "num_hidden_layers": 24,
+  "text_vocab_size": 65631,
+  "tie_word_embeddings": false,
+  "torch_dtype": "float32",
+  "transformers_version": "4.52.4",
+  "use_cache": true,
+  "v_low_rank_dim": 32,
+  "value_dim": [
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024,
+    1024
+  ],
+  "vocab_size": 8193
+}

rwkv7-0.4B-g1-respark-voice-tunable/configuration_rwkv7.py ADDED Viewed

	@@ -0,0 +1,91 @@

+# -*- coding: utf-8 -*-
+from typing import Dict, Optional
+from transformers.configuration_utils import PretrainedConfig
+class RWKV7Config(PretrainedConfig):
+    model_type = 'rwkv7'
+    keys_to_ignore_at_inference = ['past_key_values']
+    def __init__(
+        self,
+        attn_mode: str = "chunk",
+        hidden_size: int = 2048,
+        hidden_ratio: Optional[int] = 4,
+        intermediate_size: Optional[int] = None,
+        num_hidden_layers: int = 24,
+        head_dim: Optional[int] = 64,
+        num_heads: Optional[int] = None,
+        decay_low_rank_dim: int = 64,
+        gate_low_rank_dim: int = 128,
+        a_low_rank_dim: int = 64,
+        v_low_rank_dim: int = 16,
+        hidden_act: str = "sqrelu",
+        max_position_embeddings: int = 2048,
+        norm_first: bool = True,
+        norm_bias: bool = True,
+        norm_eps: float = 1e-5,
+        attn: Optional[Dict] = None,
+        use_cache: bool = True,
+        pad_token_id: int = None,
+        bos_token_id: int = 1,
+        eos_token_id: int = 2,
+        tie_word_embeddings: bool = False,
+        initializer_range: float = 0.006,
+        fuse_norm: bool = True,
+        fuse_cross_entropy: bool = True,
+        vocab_size: int = 32000,
+        **kwargs
+    ):
+        self.attn_mode = attn_mode
+        self.hidden_size = hidden_size
+        self.hidden_ratio = hidden_ratio
+        self.intermediate_size = intermediate_size
+        self.norm_first = norm_first
+        self.num_hidden_layers = num_hidden_layers
+        if head_dim is None and num_heads is not None:
+            head_dim = int(hidden_size // num_heads)
+        elif head_dim is not None and num_heads is None:
+            num_heads = int(hidden_size // head_dim)
+        self.head_dim = head_dim
+        self.num_heads = num_heads
+        self.decay_low_rank_dim = decay_low_rank_dim
+        self.gate_low_rank_dim = gate_low_rank_dim
+        self.a_low_rank_dim = a_low_rank_dim
+        self.v_low_rank_dim = v_low_rank_dim
+        self.hidden_act = hidden_act
+        self.max_position_embeddings = max_position_embeddings
+        self.norm_bias = norm_bias
+        self.norm_eps = norm_eps
+        self.attn = attn
+        self.use_cache = use_cache
+        self.initializer_range = initializer_range
+        self.fuse_norm = fuse_norm
+        self.fuse_cross_entropy = fuse_cross_entropy
+        self.vocab_size = vocab_size
+        if attn is not None:
+            if not isinstance(attn, Dict):
+                raise ValueError("attn must be a dictionary")
+            if 'layers' not in attn:
+                raise ValueError("Layer indices must be provided to initialize hybrid attention layers")
+            if 'num_heads' not in attn:
+                raise ValueError("Number of heads must be provided to initialize hybrid attention layers")
+            attn['num_kv_heads'] = attn.get('num_kv_heads', attn['num_heads'])
+            attn['qkv_bias'] = attn.get('qkv_bias', False)
+            attn['window_size'] = attn.get('window_size', None)
+            attn['rope_theta'] = attn.get('rope_theta', 10000.)
+        super().__init__(
+            pad_token_id=pad_token_id,
+            bos_token_id=bos_token_id,
+            eos_token_id=eos_token_id,
+            tie_word_embeddings=tie_word_embeddings,
+            **kwargs,
+        )

rwkv7-0.4B-g1-respark-voice-tunable/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 0,
+  "eos_token_id": 0,
+  "transformers_version": "4.52.4"
+}

rwkv7-0.4B-g1-respark-voice-tunable/hf_rwkv_tokenizer.py ADDED Viewed

	@@ -0,0 +1,280 @@

+# coding=utf-8
+# Copyright 2024 The HuggingFace Inc. team.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tokenization classes for RWKV."""
+import os
+import re
+from typing import TYPE_CHECKING, List, Optional, Tuple
+from transformers.tokenization_utils import AddedToken, PreTrainedTokenizer
+from transformers.utils import logging
+if TYPE_CHECKING:
+    pass
+logger = logging.get_logger(__name__)
+VOCAB_FILES_NAMES = {
+    "vocab_file": "rwkv_vocab_v20230424.txt",
+}
+class TRIE:
+    __slots__ = tuple("ch,to,values,front".split(","))
+    to: list
+    values: set
+    def __init__(self, front=None, ch=None):
+        self.ch = ch
+        self.to = [None for ch in range(256)]
+        self.values = set()
+        self.front = front
+    def __repr__(self):
+        fr = self
+        ret = []
+        while fr != None:
+            if fr.ch != None:
+                ret.append(fr.ch)
+            fr = fr.front
+        return "<TRIE %s %s>" % (ret[::-1], self.values)
+    def add(self, key: bytes, idx: int = 0, val=None):
+        if idx == len(key):
+            if val is None:
+                val = key
+            self.values.add(val)
+            return self
+        ch = key[idx]
+        if self.to[ch] is None:
+            self.to[ch] = TRIE(front=self, ch=ch)
+        return self.to[ch].add(key, idx=idx + 1, val=val)
+    def find_longest(self, key: bytes, idx: int = 0):
+        u: TRIE = self
+        ch: int = key[idx]
+        while u.to[ch] is not None:
+            u = u.to[ch]
+            idx += 1
+            if u.values:
+                ret = idx, u, u.values
+            if idx == len(key):
+                break
+            ch = key[idx]
+        return ret
+class RWKV_TOKENIZER:
+    def __init__(self, file_name):
+        self.idx2token = {}
+        sorted = []  # must be already sorted
+        with open(file_name, "r", encoding="utf-8") as f:
+            lines = f.readlines()
+        for l in lines:
+            idx = int(l[: l.index(" ")])
+            x = eval(l[l.index(" ") : l.rindex(" ")])
+            x = x.encode("utf-8") if isinstance(x, str) else x
+            assert isinstance(x, bytes)
+            assert len(x) == int(l[l.rindex(" ") :])
+            sorted += [x]
+            self.idx2token[idx] = x
+        self.token2idx = {}
+        for k, v in self.idx2token.items():
+            self.token2idx[v] = int(k)
+        self.root = TRIE()
+        for t, i in self.token2idx.items():
+            _ = self.root.add(t, val=(t, i))
+    def encodeBytes(self, src: bytes):
+        idx: int = 0
+        tokens = []
+        while idx < len(src):
+            _idx: int = idx
+            idx, _, values = self.root.find_longest(src, idx)
+            assert idx != _idx
+            _, token = next(iter(values))
+            tokens.append(token)
+        return tokens
+    def decodeBytes(self, tokens):
+        return b"".join(map(lambda i: self.idx2token[i], tokens))
+    def encode(self, src):
+        if isinstance(src, str):
+            return [self.encodeBytes(src.encode("utf-8"))]
+        elif isinstance(src, list):
+            return [self.encodeBytes(s.encode("utf-8")) for s in src]
+    def decode(self, tokens):
+        return [self.decodeBytes(batch).decode("utf-8") for batch in tokens]
+        # try:
+        #     return self.decodeBytes(tokens).decode('utf-8')
+        # except:
+        #     return '\ufffd' # bad utf-8
+    def printTokens(self, tokens):
+        for i in tokens:
+            s = self.idx2token[i]
+            try:
+                s = s.decode("utf-8")
+            except:
+                pass
+            print(f"{repr(s)}{i}", end=" ")
+        print()
+class RwkvTokenizer(PreTrainedTokenizer):
+    vocab_files_names = VOCAB_FILES_NAMES
+    model_input_names = ["input_ids", "attention_mask"]
+    def __init__(
+        self, vocab_file, bos_token="<|rwkv_tokenizer_end_of_text|>", eos_token="<|rwkv_tokenizer_end_of_text|>", unk_token="<|rwkv_tokenizer_end_of_text|>", **kwargs
+    ):
+        if not os.path.isfile(vocab_file):
+            raise ValueError(
+                f"Can't find a vocabulary file at path '{vocab_file}'."
+            )
+        with open(vocab_file, "r", encoding="utf-8") as reader:
+            tokens = reader.readlines()
+        if "add_bos_token" in kwargs:
+            self.add_bos_token = kwargs["add_bos_token"]
+        else:
+            self.add_bos_token = False
+        self.trie_tokenizer = RWKV_TOKENIZER(vocab_file)
+        vocab = self.trie_tokenizer.token2idx
+        self.encoder = vocab
+        self.decoder = {v: k for k, v in vocab.items()}
+        self._added_tokens_decoder = {0: AddedToken(str(bos_token))}
+        super().__init__(
+            bos_token=bos_token, eos_token=eos_token, unk_token=unk_token, **kwargs
+        )
+    @property
+    def vocab_size(self):
+        return len(self.encoder)
+    def get_vocab(self):
+        vocab = self.encoder
+        vocab.update(self.added_tokens_encoder)
+        vocab = dict(sorted(vocab.items(), key=lambda item: item[1]))
+        return vocab
+    def _tokenize(self, text, split_special_tokens=False):
+        # return self.wordpiece_tokenizer.tokenize(text.encode("utf-8"))
+        return self.trie_tokenizer.encode(text)[0]
+    def _convert_token_to_id(self, token):
+        return token
+    def _convert_id_to_token(self, index):
+        """Converts an index (integer) in a token (byte) using the vocab."""
+        token = self.decoder.get(index, self.unk_token)
+        if isinstance(token, (bytes)):
+            token = token.decode("utf-8", errors="replace")
+        return token
+    def convert_tokens_to_string(self, tokens):
+        """Converts a sequence of tokens (bytes) in a single string. Additional tokens are encoded to bytes"""
+        out_string = b"".join(
+            [k.encode(errors="replace") if isinstance(k, str) else k for k in tokens]
+        ).decode("utf-8")
+        return out_string
+    def save_vocabulary(
+        self, save_directory: str, filename_prefix: Optional[str] = None
+    ) -> Tuple[str]:
+        index = 0
+        if os.path.isdir(save_directory):
+            vocab_file = os.path.join(
+                save_directory,
+                (filename_prefix + "-" if filename_prefix else "") + "vocab.txt",
+            )
+        else:
+            vocab_file = (
+                filename_prefix + "-" if filename_prefix else ""
+            ) + save_directory
+        with open(vocab_file, "w", encoding="utf-8") as writer:
+            for token, token_index in sorted(
+                self.encoder.items(), key=lambda kv: kv[1]
+            ):
+                if index != token_index:
+                    logger.warning(
+                        f"Saving vocabulary to {vocab_file}: vocabulary indices are not consecutive."
+                        " Please check that the vocabulary is not corrupted!"
+                    )
+                    index = token_index
+                writer.write(str(token) + "\n")
+                index += 1
+        return (vocab_file,)
+    def build_inputs_with_special_tokens(self, token_ids_0, token_ids_1=None):
+        if self.add_bos_token:
+            bos_token_ids = [self.bos_token_id]
+        else:
+            bos_token_ids = []
+        output = bos_token_ids + token_ids_0
+        if token_ids_1 is None:
+            return output
+        return output + bos_token_ids + token_ids_1
+    def get_special_tokens_mask(
+        self,
+        token_ids_0: List[int],
+        token_ids_1: Optional[List[int]] = None,
+        already_has_special_tokens: bool = False,
+    ) -> List[int]:
+        """
+        Retrieves sequence ids from a token list that has no special tokens added. This method is called when adding
+        special tokens using the tokenizer `prepare_for_model` or `encode_plus` methods.
+        Args:
+            token_ids_0 (`List[int]`):
+                List of IDs.
+            token_ids_1 (`List[int]`, *optional*):
+                Optional second list of IDs for sequence pairs.
+            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
+                Whether or not the token list is already formatted with special tokens for the model.
+        Returns:
+            `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
+        """
+        if already_has_special_tokens:
+            return super().get_special_tokens_mask(
+                token_ids_0=token_ids_0,
+                token_ids_1=token_ids_1,
+                already_has_special_tokens=True,
+            )
+        if not self.add_bos_token:
+            return super().get_special_tokens_mask(
+                token_ids_0=token_ids_0,
+                token_ids_1=token_ids_1,
+                already_has_special_tokens=False,
+            )
+        if token_ids_1 is None:
+            return [1] + ([0] * len(token_ids_0))
+        return [1] + ([0] * len(token_ids_0)) + [1] + ([0] * len(token_ids_1))

rwkv7-0.4B-g1-respark-voice-tunable/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c1a5073f8de7273ae0e2ea9b9baa5c6417ffeb7fd75454abc0814e58ab94a011
+size 1619016168

rwkv7-0.4B-g1-respark-voice-tunable/model_converted.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4b12489fc40149dc3239169662acc66ff196f2ce81c2d469026afce3be4260d5
+size 1619174929

rwkv7-0.4B-g1-respark-voice-tunable/model_padded.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7aff221eee7ae652e6788b85dc3e32888d46da432729eda2a7f36a3723090878
+size 1904786606

rwkv7-0.4B-g1-respark-voice-tunable/modeling_rwkv7.py ADDED Viewed

	@@ -0,0 +1,4 @@

+from fla.models.rwkv7 import RWKV7ForCausalLM, RWKV7Model, RWKV7Config
+RWKV7ForCausalLM = RWKV7ForCausalLM
+RWKV7Model = RWKV7Model
+RWKV7Config = RWKV7Config

rwkv7-0.4B-g1-respark-voice-tunable/modeling_rwkvspeech.py ADDED Viewed

	@@ -0,0 +1,6 @@

+from spark_llm import RWKV7SpeechConfig,RWKV7ForSpeech
+from rwkvfla.models.rwkv7 import RWKV7Model
+RWKV7ForCausalLM = RWKV7ForSpeech
+RWKV7Model = RWKV7Model
+RWKV7Config = RWKV7SpeechConfig

rwkv7-0.4B-g1-respark-voice-tunable/rwkv_vocab_v20230424.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

rwkv7-0.4B-g1-respark-voice-tunable/spark_llm.py ADDED Viewed

	@@ -0,0 +1,202 @@

+import torch
+import torch.nn as nn
+from typing import Optional, Union, Tuple, Dict, Unpack
+from transformers.modeling_utils import PreTrainedModel
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from transformers.utils.deprecation import deprecate_kwarg
+from rwkvfla.models.rwkv7.modeling_rwkv7 import RWKV7Model, RWKV7PreTrainedModel, Cache,RWKV7ForCausalLM
+from rwkvfla.models.rwkv7.modeling_rwkv7 import FusedLinearCrossEntropyLoss, FusedCrossEntropyLoss
+from transformers.generation.utils import GenerationMixin
+from rwkvfla.models.rwkv7.configuration_rwkv7 import RWKV7Config
+class RWKV7SpeechConfig(RWKV7Config):
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+        self.text_vocab_size = kwargs.get("text_vocab_size", kwargs.get("text_vocab_size"))
+        self.audio_global_vocab_size = kwargs.get("audio_global_vocab_size", kwargs.get("audio_global_vocab_size"))
+class RWKV7ForSpeech(RWKV7ForCausalLM):
+    config_class = RWKV7SpeechConfig
+    def __init__(self, config: RWKV7SpeechConfig):
+        super().__init__(config)
+        self.model = RWKV7Model(config)
+        self.vocab_size = config.vocab_size
+        self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)#Spark 0.5B vocab size is 8192 + 1 for eos resulting in 8193
+        self.criterion = None
+        self.text_embedder = nn.Embedding(config.text_vocab_size, config.hidden_size)
+        self.global_embedder = nn.Embedding(config.audio_global_vocab_size, config.hidden_size)#Spark 0.5B global token size is 4096
+        #TTS Tag includes GLOBAL=0, SEMANTIC=1,START_TTS=2
+        self.tts_tag_embedder = nn.Embedding(3, config.hidden_size)
+        # Initialize weights and apply final processing
+        self.post_init()
+        self.dropout = torch.nn.Dropout(0.02)
+    def get_input_embeddings(self):
+        return self.model.embeddings
+    def set_input_embeddings(self, value):
+        self.model.embeddings = value
+    def get_output_embeddings(self):
+        return self.lm_head
+    def set_output_embeddings(self, new_embeddings):
+        self.lm_head = new_embeddings
+    def set_decoder(self, decoder):
+        self.model = decoder
+    def get_decoder(self):
+        return self.model
+    def generate(self, *args, **kwargs):
+        try:
+            return super().generate(*args, **kwargs)
+        except AttributeError as exception:
+            if 'past_key_values' in str(exception):
+                raise AttributeError(
+                    f"You tried to call `generate` with a decoding strategy that manipulates `past_key_values`, "
+                    f"which is not supported for {self.__class__.__name__}. "
+                    f"Try another generation strategy instead. "
+                    f"For the available generation strategies, check this doc: "
+                    f"https://huggingface.co/docs/transformers/en/generation_strategies#decoding-strategies"
+                )
+            else:
+                raise exception
+    @deprecate_kwarg("num_logits_to_keep", version="4.50", new_name="logits_to_keep")
+    def prepare_inputs_for_generation(
+        self,
+        input_ids: torch.LongTensor = None,
+        past_key_values: Optional[Cache] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        inputs_embeds: Optional[torch.Tensor] = None,
+        use_cache: bool = True,
+        logits_to_keep: Optional[int] = None,
+        **kwargs
+    ):
+        # only last token for `inputs_ids` if the `past_key_values` is not empty.
+        if past_key_values is not None and len(past_key_values) > 0:
+            input_ids = input_ids[:, -1:]
+        # if `inputs_embeds` are passed, we only want to use them in the 1st generation step
+        if inputs_embeds is not None and len(past_key_values) == 0:
+            model_inputs = {'inputs_embeds': inputs_embeds}
+        else:
+            # The `contiguous()` here is necessary to have a static stride during decoding. torchdynamo otherwise
+            # recompiles graphs as the stride of the inputs is a guard.
+            # Ref: https://github.com/huggingface/transformers/pull/29114
+            # TODO: use `next_tokens` directly instead.
+            model_inputs = {'input_ids': input_ids.contiguous()}
+        if logits_to_keep is not None:
+            model_inputs['logits_to_keep'] = logits_to_keep
+        model_inputs.update({
+            'past_key_values': past_key_values,
+            'use_cache': use_cache,
+            'attention_mask': attention_mask,
+            'logits_to_keep': logits_to_keep,
+        })
+        return model_inputs
+    @deprecate_kwarg("num_logits_to_keep", version="4.50", new_name="logits_to_keep")
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        inputs_embeds: Optional[torch.Tensor] = None,
+        past_key_values: Optional[Cache] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+        logits_to_keep: Optional[int] = 0,
+        **kwargs: Unpack[Dict]
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if self.training and inputs_embeds is not None:
+            inputs_embeds = self.dropout(inputs_embeds)
+        outputs = self.model(
+            input_ids=input_ids,
+            attention_mask=attention_mask,
+            inputs_embeds=inputs_embeds,
+            past_key_values=past_key_values,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            **kwargs
+        )
+        hidden_states = outputs[0]
+        fuse_linear_and_cross_entropy = self.config.fuse_cross_entropy and self.training
+        loss, logits = None, None
+        if not fuse_linear_and_cross_entropy or labels is None:
+            logits = self.lm_head(hidden_states if logits_to_keep is None else hidden_states[:, -logits_to_keep:])
+        if labels is not None:
+            if getattr(self, 'criterion', None) is None:
+                if fuse_linear_and_cross_entropy:
+                    criterion = FusedLinearCrossEntropyLoss()
+                elif self.config.fuse_cross_entropy:
+                    criterion = FusedCrossEntropyLoss(inplace_backward=True)
+                else:
+                    criterion = nn.CrossEntropyLoss()
+            else:
+                criterion = self.criterion
+            # Enable model parallelism
+            labels = labels.to(hidden_states.device)
+            labels = torch.cat((labels[..., 1:], torch.full_like(labels[:, :1], criterion.ignore_index)), 1)
+            if fuse_linear_and_cross_entropy:
+                loss = criterion(hidden_states, labels, self.lm_head.weight, self.lm_head.bias)
+            else:
+                loss = criterion(logits.view(labels.numel(), -1), labels.view(-1))
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+        return CausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+        )
+    def copy_state_dict(self, state_dict: dict):
+        """从源 state dict 复制参数到当前模型，排除 embeddings 和 lm_head
+        The state dict is from original RWKV7 language model
+        Args:
+            state_dict: 源 state dict
+        """
+        # 获取当前模型的 state dict
+        target_dict = self.state_dict()
+        # 创建新的 state dict 用于存储要复制的参数
+        new_state_dict = {}
+        # 遍历源 state dict 的键
+        for key in state_dict.keys():
+            # 跳过 embeddings 和 lm_head 相关的参数
+            if key == 'model.embeddings.weight':
+                new_state_dict['text_embedder.weight'] = state_dict[key]
+                continue
+            if 'embeddings' in key or 'lm_head' in key:
+                continue
+            # 如果键在当前模型中存在，则复制参数
+            if key in target_dict:
+                new_state_dict[key] = state_dict[key]
+        # 加载新的 state dict 到当前模型
+        info = self.load_state_dict(new_state_dict, strict=False)
+        print(info)
+        return self

rwkv7-0.4B-g1-respark-voice-tunable/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,24 @@

+{
+  "bos_token": {
+    "content": "<|rwkv_tokenizer_end_of_text|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": "\n\n",
+  "pad_token": {
+    "content": "<|rwkv_tokenizer_end_of_text|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<|rwkv_tokenizer_end_of_text|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

rwkv7-0.4B-g1-respark-voice-tunable/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,836 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<|rwkv_tokenizer_end_of_text|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "65530": {
+      "content": "\n\n",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "65531": {
+      "content": "SPCT_0",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65532": {
+      "content": "SPCT_1",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65533": {
+      "content": "SPCT_2",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65534": {
+      "content": "SPCT_3",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65535": {
+      "content": "SPCT_4",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65536": {
+      "content": "SPCT_5",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65537": {
+      "content": "SPCT_6",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65538": {
+      "content": "SPCT_7",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65539": {
+      "content": "SPCT_8",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65540": {
+      "content": "SPCT_9",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65541": {
+      "content": "SPCT_10",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65542": {
+      "content": "SPCT_11",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65543": {
+      "content": "SPCT_12",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65544": {
+      "content": "SPCT_13",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65545": {
+      "content": "SPCT_14",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65546": {
+      "content": "SPCT_15",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65547": {
+      "content": "SPCT_16",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65548": {
+      "content": "SPCT_17",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65549": {
+      "content": "SPCT_18",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65550": {
+      "content": "SPCT_19",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65551": {
+      "content": "SPCT_20",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65552": {
+      "content": "SPCT_21",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65553": {
+      "content": "SPCT_22",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65554": {
+      "content": "SPCT_23",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65555": {
+      "content": "SPCT_24",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65556": {
+      "content": "SPCT_25",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65557": {
+      "content": "SPCT_26",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65558": {
+      "content": "SPCT_27",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65559": {
+      "content": "SPCT_28",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65560": {
+      "content": "SPCT_29",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65561": {
+      "content": "SPCT_30",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65562": {
+      "content": "SPCT_31",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65563": {
+      "content": "SPCT_32",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65564": {
+      "content": "SPCT_33",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65565": {
+      "content": "SPCT_34",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65566": {
+      "content": "SPCT_35",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65567": {
+      "content": "SPCT_36",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65568": {
+      "content": "SPCT_37",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65569": {
+      "content": "SPCT_38",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65570": {
+      "content": "SPCT_39",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65571": {
+      "content": "SPCT_40",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65572": {
+      "content": "SPCT_41",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65573": {
+      "content": "SPCT_42",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65574": {
+      "content": "SPCT_43",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65575": {
+      "content": "SPCT_44",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65576": {
+      "content": "SPCT_45",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65577": {
+      "content": "SPCT_46",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65578": {
+      "content": "SPCT_47",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65579": {
+      "content": "SPCT_48",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65580": {
+      "content": "SPCT_49",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65581": {
+      "content": "SPCT_50",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65582": {
+      "content": "SPCT_51",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65583": {
+      "content": "SPCT_52",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65584": {
+      "content": "SPCT_53",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65585": {
+      "content": "SPCT_54",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65586": {
+      "content": "SPCT_55",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65587": {
+      "content": "SPCT_56",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65588": {
+      "content": "SPCT_57",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65589": {
+      "content": "SPCT_58",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65590": {
+      "content": "SPCT_59",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65591": {
+      "content": "SPCT_60",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65592": {
+      "content": "SPCT_61",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65593": {
+      "content": "SPCT_62",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65594": {
+      "content": "SPCT_63",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65595": {
+      "content": "SPCT_64",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65596": {
+      "content": "SPCT_65",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65597": {
+      "content": "SPCT_66",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65598": {
+      "content": "SPCT_67",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65599": {
+      "content": "SPCT_68",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65600": {
+      "content": "SPCT_69",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65601": {
+      "content": "SPCT_70",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65602": {
+      "content": "SPCT_71",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65603": {
+      "content": "SPCT_72",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65604": {
+      "content": "SPCT_73",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65605": {
+      "content": "SPCT_74",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65606": {
+      "content": "SPCT_75",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65607": {
+      "content": "SPCT_76",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65608": {
+      "content": "SPCT_77",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65609": {
+      "content": "SPCT_78",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65610": {
+      "content": "SPCT_79",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65611": {
+      "content": "SPCT_80",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65612": {
+      "content": "SPCT_81",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65613": {
+      "content": "SPCT_82",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65614": {
+      "content": "SPCT_83",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65615": {
+      "content": "SPCT_84",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65616": {
+      "content": "SPCT_85",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65617": {
+      "content": "SPCT_86",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65618": {
+      "content": "SPCT_87",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65619": {
+      "content": "SPCT_88",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65620": {
+      "content": "SPCT_89",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65621": {
+      "content": "SPCT_90",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65622": {
+      "content": "SPCT_91",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65623": {
+      "content": "SPCT_92",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65624": {
+      "content": "SPCT_93",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65625": {
+      "content": "SPCT_94",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65626": {
+      "content": "SPCT_95",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65627": {
+      "content": "SPCT_96",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65628": {
+      "content": "SPCT_97",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65629": {
+      "content": "SPCT_98",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "65630": {
+      "content": "SPCT_99",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "auto_map": {
+    "AutoTokenizer": [
+      "hf_rwkv_tokenizer.RwkvTokenizer",
+      null
+    ]
+  },
+  "bos_token": "<|rwkv_tokenizer_end_of_text|>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "\n\n",
+  "extra_special_tokens": {},
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<|rwkv_tokenizer_end_of_text|>",
+  "tokenizer_class": "RwkvTokenizer",
+  "unk_token": "<|rwkv_tokenizer_end_of_text|>",
+  "use_fast": false
+}

rwkv7-0.4B-g1-respark-voice-tunable/vocab.txt ADDED Viewed

The diff for this file is too large to render. See raw diff