mattChrisP commited on
Commit
28e721c
·
verified ·
1 Parent(s): c6ae099

release myna

Browse files
Files changed (47) hide show
  1. .gitattributes +2 -73
  2. .gitignore +7 -0
  3. MODEL_LINEAGE.json +22 -0
  4. README.md +68 -49
  5. processors/generation/chat_template.jinja → chat_template.jinja +0 -0
  6. processors/generation/chat_template.json → chat_template.json +0 -0
  7. config.json +208 -343
  8. configuration_mynahokkien.py +0 -41
  9. generation_config.json +6 -2
  10. inference.py +16 -16
  11. model.safetensors.index.json +0 -0
  12. modeling_mynahokkien.py +0 -115
  13. myna_hokkien.egg-info/PKG-INFO +15 -0
  14. myna_hokkien.egg-info/SOURCES.txt +9 -0
  15. myna_hokkien.egg-info/dependency_links.txt +1 -0
  16. myna_hokkien.egg-info/requires.txt +8 -0
  17. myna_hokkien.egg-info/top_level.txt +1 -0
  18. mynahokkien/__init__.py +3 -13
  19. mynahokkien/model.py +208 -0
  20. mynahokkien/models/__init__.py +0 -15
  21. mynahokkien/models/bridge1_decoder.py +0 -83
  22. mynahokkien/models/mynahokkien.py +0 -314
  23. processors/generation/preprocessor_config.json → preprocessor_config.json +0 -0
  24. processing_mynahokkien.py +0 -30
  25. processor_config.json +0 -5
  26. processors/generation/config.json +0 -589
  27. processors/generation/generation_config.json +0 -8
  28. processors/generation/processor_config.json +0 -109
  29. processors/reasoning/added_tokens.json +0 -24
  30. processors/reasoning/chat_template.jinja +0 -7
  31. processors/reasoning/chat_template.json +0 -3
  32. processors/reasoning/config.json +0 -650
  33. processors/reasoning/generation_config.json +0 -4
  34. processors/reasoning/merges.txt +0 -0
  35. processors/reasoning/preprocessor_config.json +0 -31
  36. processors/reasoning/special_tokens_map.json +0 -38
  37. processors/reasoning/tokenizer.json +0 -3
  38. processors/reasoning/tokenizer_config.json +0 -222
  39. processors/reasoning/video_preprocessor_config.json +0 -55
  40. processors/reasoning/vocab.json +0 -0
  41. pyproject.toml +8 -18
  42. release_config.json +18 -0
  43. requirements.txt +6 -9
  44. processors/generation/special_tokens_map.json → special_tokens_map.json +0 -0
  45. processors/generation/tokenizer.json → tokenizer.json +2 -2
  46. processors/generation/tokenizer_config.json → tokenizer_config.json +0 -4
  47. processors/generation/video_preprocessor_config.json → video_preprocessor_config.json +0 -0
.gitattributes CHANGED
@@ -1,76 +1,5 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
  *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
  *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
  *.wav filter=lfs diff=lfs merge=lfs -text
33
- *.xz filter=lfs diff=lfs merge=lfs -text
34
- *.zip filter=lfs diff=lfs merge=lfs -text
35
- *.zst filter=lfs diff=lfs merge=lfs -text
36
- *tfevents* filter=lfs diff=lfs merge=lfs -text
37
- q25_processor_config/tokenizer.json filter=lfs diff=lfs merge=lfs -text
38
- q3_processor_config/tokenizer.json filter=lfs diff=lfs merge=lfs -text
39
- demo_samples/01/01_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
40
- demo_samples/01/01_input.wav filter=lfs diff=lfs merge=lfs -text
41
- demo_samples/01/01_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
42
- demo_samples/02/02_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
43
- demo_samples/02/02_input.wav filter=lfs diff=lfs merge=lfs -text
44
- demo_samples/02/02_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
45
- demo_samples/02/02_qwen.wav filter=lfs diff=lfs merge=lfs -text
46
- demo_samples/03/03_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
47
- demo_samples/03/03_input.wav filter=lfs diff=lfs merge=lfs -text
48
- demo_samples/03/03_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
49
- demo_samples/03/03_qwen.wav filter=lfs diff=lfs merge=lfs -text
50
- demo_samples/04/04_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
51
- demo_samples/04/04_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
52
- demo_samples/04/04_qwen.wav filter=lfs diff=lfs merge=lfs -text
53
- demo_samples/05/05_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
54
- demo_samples/05/05_input.wav filter=lfs diff=lfs merge=lfs -text
55
- demo_samples/05/05_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
56
- demo_samples/05/05_qwen.wav filter=lfs diff=lfs merge=lfs -text
57
- demo_samples/06/06_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
58
- demo_samples/06/06_input.wav filter=lfs diff=lfs merge=lfs -text
59
- demo_samples/06/06_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
60
- demo_samples/06/06_qwen.wav filter=lfs diff=lfs merge=lfs -text
61
- demo_samples/07/07_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
62
- demo_samples/07/07_input.wav filter=lfs diff=lfs merge=lfs -text
63
- demo_samples/07/07_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
64
- demo_samples/07/07_qwen.wav filter=lfs diff=lfs merge=lfs -text
65
- demo_samples/08_en_text/08_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
66
- demo_samples/08_en_text/08_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
67
- demo_samples/08_en_text/08_qwen.wav filter=lfs diff=lfs merge=lfs -text
68
- demo_samples/09_en_text/09_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
69
- demo_samples/09_en_text/09_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
70
- demo_samples/09_en_text/09_qwen.wav filter=lfs diff=lfs merge=lfs -text
71
- demo_samples/10_zh_text/10_gptaudio.wav filter=lfs diff=lfs merge=lfs -text
72
- demo_samples/10_zh_text/10_mynahokkien.wav filter=lfs diff=lfs merge=lfs -text
73
- demo_samples/10_zh_text/10_qwen.wav filter=lfs diff=lfs merge=lfs -text
74
- demo_samples/04/04_input.wav filter=lfs diff=lfs merge=lfs -text
75
- processors/generation/tokenizer.json filter=lfs diff=lfs merge=lfs -text
76
- processors/reasoning/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
1
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
 
2
  *.bin filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  *.pt filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
4
  *.wav filter=lfs diff=lfs merge=lfs -text
5
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
.gitignore ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ .cache/
2
+ __pycache__/
3
+ *.py[cod]
4
+ build/
5
+ dist/
6
+ *.egg-info/
7
+ merge_info.json
MODEL_LINEAGE.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_model": "Qwen/Qwen3-Omni-30B-A3B-Instruct",
3
+ "release_family": "release_v2",
4
+ "thinker": {
5
+ "stage": 2,
6
+ "experiment": "thinker_stage2_meralion_asrqa_fullft_v1",
7
+ "training": "full-parameter continuation",
8
+ "state_during_talker_stage4": "frozen"
9
+ },
10
+ "talker": {
11
+ "stage": 4,
12
+ "experiment": "talker_stage4_currentbest_humanchat_s2s_v1",
13
+ "epoch": 1,
14
+ "training": "full Talker and MTP continuation"
15
+ },
16
+ "native_hidden_routing": {
17
+ "text_and_assistant_positions": 0,
18
+ "user_audio_positions": 24,
19
+ "runtime_ablation_passed": true
20
+ },
21
+ "speaker": "Ethan"
22
+ }
README.md CHANGED
@@ -1,59 +1,52 @@
1
  ---
2
  language:
3
  - nan
4
- library_name: mynahokkien
 
5
  pipeline_tag: audio-to-audio
 
6
  tags:
7
  - hokkien
8
  - singapore-hokkien
9
  - speech-to-speech
10
  - audio
 
11
  ---
12
 
13
  # Myna-Hokkien
14
 
15
- Myna-Hokkien is an open-source end-to-end conversational speech model for Singapore Hokkien.
16
- Users speak to the model in Hokkien and the model replies in natural-sounding Hokkien speech -
17
- no text round-trip required, though text input/output is also supported.
18
 
19
- Hokkien is the native language of tens of millions of speakers across Asia,
20
- yet it remains almost entirely absent from mainstream speech AI: no existing omni-style model (either open or closed) natively supports Hokkien.
21
- Myna-Hokkien is an attempt to close that gap, and to leave behind a reusable recipe for other low-resource language communities to do the same.
22
 
23
- Proudly built from [iNLP Lab](https://huggingface.co/iNLP-Lab)
24
 
 
25
 
26
- ## Model Details
27
- * Languages: Hokkien / Minnan (闽南语), primarily Singapore-accent in this release
28
- * Modalities: Audio in → Audio out: full spoken dialogue, text input/output also supported
29
- * Architecture: Built on two generations of Qwen-Omni via model editing: Qwen2.5-Omni-3B Thinker + trained Transformer Bridge + Qwen3-Omni Talker
30
- * License:
 
 
 
31
 
 
 
32
 
33
- ## Demo Samples
34
- Examples showcasing the model's Hokkien understanding & generation capabilities.
35
-
36
- For comparison, we also ran the same inputs through four frontier audio-to-audio models — GPT Audio, Qwen3.5-Omni-Plus, Gemini, and GLM-Voice —
37
- none of which officially supports Hokkien;
38
- we include them to illustrate how far mainstream omni models are from handling the language, not as an apples-to-apples benchmark.
39
-
40
- | Input | Myna-Hokkien | GPT Audio | Qwen3.5-Omni-Plus | Gemini | GLM-Voice |
41
- |-------|-------------|-----------|--------------------|--------|-----------|
42
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/01/01_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/01/01_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/01/01_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/01/01_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/01/01_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/01/01_glmvoice.wav"></audio> |
43
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/02/02_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/02/02_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/02/02_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/02/02_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/02/02_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/02/02_glmvoice.wav"></audio> |
44
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/03/03_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/03/03_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/03/03_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/03/03_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/03/03_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/03/03_glmvoice.wav"></audio> |
45
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/04/04_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/04/04_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/04/04_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/04/04_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/04/04_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/04/04_glmvoice.wav"></audio> |
46
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/05/05_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/05/05_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/05/05_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/05/05_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/05/05_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/05/05_glmvoice.wav"></audio> |
47
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/06/06_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/06/06_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/06/06_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/06/06_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/06/06_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/06/06_glmvoice.wav"></audio> |
48
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/07/07_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/07/07_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/07/07_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/07/07_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/07/07_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/07/07_glmvoice.wav"></audio> |
49
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/08/08_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/08/08_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/08/08_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/08/08_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/08/08_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/08/08_glmvoice.wav"></audio> |
50
- | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/09/09_input.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/09/09_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/09/09_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/09/09_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/09/09_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/09/09_glmvoice.wav"></audio> |
51
- | Draft an email for me to request to take a day off at work. | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/10/10_mynahokkien.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/10/10_gptaudio.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/10/10_qwen.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/10/10_gemini.wav"></audio> | <audio controls src="https://huggingface.co/iNLP-Lab/MynaHokkien/resolve/main/demo_samples/release_csv/10/10_glmvoice.wav"></audio> |
52
 
 
 
 
53
 
54
  ## Installation
55
 
56
- Download MynaHokkien into the standard Hugging Face cache, then install the included inference
57
  runtime:
58
 
59
  ```bash
@@ -63,6 +56,9 @@ MODEL_DIR="$(hf download iNLP-Lab/MynaHokkien --quiet)"
63
  pip install "$MODEL_DIR"
64
  ```
65
 
 
 
 
66
  ## Basic usage — spoken question
67
 
68
  ```python
@@ -78,7 +74,6 @@ model = MynaHokkien.from_pretrained(
78
 
79
  output = model.generate(
80
  audio="question.wav",
81
- prompt="Listen to this audio and reply naturally in Singaporean Hokkien.",
82
  language="nan",
83
  return_text=True,
84
  return_audio=True,
@@ -88,10 +83,13 @@ print(output.text)
88
  sf.write("output.wav", output.audio, output.sampling_rate)
89
  ```
90
 
 
 
 
91
  ## Text-query usage
92
 
93
  Text is treated as a question or instruction to the Hokkien assistant; it is not treated as a TTS
94
- transcript.
95
 
96
  ```python
97
  output = model.generate(
@@ -102,8 +100,8 @@ sf.write("output.wav", output.audio, output.sampling_rate)
102
  print(output.text)
103
  ```
104
 
105
- Exactly one of `audio=` and `text=` must be supplied. This release currently supports
106
- `language="nan"` and the `Ethan` voice.
107
 
108
  To return only one modality:
109
 
@@ -114,36 +112,57 @@ audio_only = model.generate(text="講一句歡迎詞。", return_text=False, ret
114
 
115
  ## Prompt behavior
116
 
117
- If `prompt=` is omitted for audio input, MynaHokkien uses this built-in prompt:
 
118
 
119
  ```text
120
- Answer or respond to what is said in the audio, in spoken Singapore Hokkien. Do NOT repeat or transcribe it.
121
  ```
122
 
123
- To override it, pass a different instruction through `prompt=`:
124
 
125
  ```python
126
  output = model.generate(
127
  audio="question.wav",
128
- prompt="Listen to this audio and reply naturally in Singaporean Hokkien.",
129
  language="nan",
130
  )
131
  ```
132
 
133
- For text input, put the instruction directly in `text=`.
 
 
134
 
 
135
 
136
- ## Citation
 
 
 
 
 
 
 
 
 
 
 
 
 
 
137
  ```
 
 
 
 
138
  @misc{myna-hokkien-2026,
139
  title = {Myna-Hokkien: An Open-Source End-to-End Hokkien Spoken Dialogue Model},
140
- author = {Matt, Ryner, Wenxuan},
141
- year = {2026},
142
- howpublished = {\url{xxx}}
143
  }
144
  ```
145
 
146
- ## Contact & collaboration
147
 
148
- This is an active, ongoing project — we're continuing to improve accent coverage, prosody, and expressiveness.
149
- We'd love to hear from you if you want to collaborate, have feedback, or run into issues: [Wenxuan Zhang](https://isakzhang.github.io/).
 
1
  ---
2
  language:
3
  - nan
4
+ license: apache-2.0
5
+ library_name: transformers
6
  pipeline_tag: audio-to-audio
7
+ base_model: Qwen/Qwen3-Omni-30B-A3B-Instruct
8
  tags:
9
  - hokkien
10
  - singapore-hokkien
11
  - speech-to-speech
12
  - audio
13
+ - qwen3-omni
14
  ---
15
 
16
  # Myna-Hokkien
17
 
18
+ Myna-Hokkien is an open-source conversational speech model for Singapore Hokkien. It accepts a
19
+ spoken or written question and returns Singapore-Hokkien text and 24 kHz speech.
 
20
 
21
+ This release is the native Qwen3-Omni `release_v2` model: the selected Stage-2 Thinker together
22
+ with the Stage-4 epoch-1 Talker and MTP. The public Python API is
23
+ `mynahokkien.MynaHokkien`.
24
 
25
+ Proudly built by [iNLP Lab](https://huggingface.co/iNLP-Lab) at SUTD.
26
 
27
+ ## Model details
28
 
29
+ - **Base architecture:** Qwen3-Omni-30B-A3B-Instruct, native Thinker–Talker
30
+ - **Thinker:** `thinker_stage2_meralion_asrqa_fullft_v1`
31
+ - **Talker:** `talker_stage4_currentbest_humanchat_s2s_v1`, epoch 1
32
+ - **Language:** Singapore Hokkien / Minnan (`nan`)
33
+ - **Voice:** Ethan
34
+ - **Input:** 16 kHz mono audio or text
35
+ - **Output:** text and 24 kHz mono audio
36
+ - **License:** Apache-2.0
37
 
38
+ The model is approximately 70.5 GB in BF16/FP16 weights. A large-memory CUDA GPU is strongly
39
+ recommended; the release workflow is validated on an H200.
40
 
41
+ ## Qualitative evaluation
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
42
 
43
+ The [playable unseen-audio probe](https://mattchrisp.github.io/omni-hokkien/) contains the
44
+ Stage-4 model outputs, prompt text and raw per-sample results. It is a qualitative listening
45
+ probe, not a scored benchmark: the recordings have no reference answers.
46
 
47
  ## Installation
48
 
49
+ Download Myna-Hokkien into the standard Hugging Face cache, then install the included inference
50
  runtime:
51
 
52
  ```bash
 
56
  pip install "$MODEL_DIR"
57
  ```
58
 
59
+ The runtime uses `transformers==4.57.1`, matching the version recorded in the released model
60
+ configuration.
61
+
62
  ## Basic usage — spoken question
63
 
64
  ```python
 
74
 
75
  output = model.generate(
76
  audio="question.wav",
 
77
  language="nan",
78
  return_text=True,
79
  return_audio=True,
 
83
  sf.write("output.wav", output.audio, output.sampling_rate)
84
  ```
85
 
86
+ Omitting `prompt=` is intentional: the release runtime supplies the default forced-Hokkien
87
+ instruction shown below.
88
+
89
  ## Text-query usage
90
 
91
  Text is treated as a question or instruction to the Hokkien assistant; it is not treated as a TTS
92
+ transcript. The runtime prefixes it with the corresponding forced-Hokkien text instruction.
93
 
94
  ```python
95
  output = model.generate(
 
100
  print(output.text)
101
  ```
102
 
103
+ Exactly one of `audio=` and `text=` must be supplied. This release supports `language="nan"` and
104
+ the `Ethan` voice.
105
 
106
  To return only one modality:
107
 
 
112
 
113
  ## Prompt behavior
114
 
115
+ The stock Qwen system turn is left unchanged. For audio input, the runtime places the following
116
+ instruction **after the audio in the same user turn**:
117
 
118
  ```text
119
+ Listen to the spoken Hokkien and reply naturally in concise Singapore Hokkien. Always answer in colloquial Singapore Hokkien written in Hanji. Never answer in Mandarin or English. Do not repeat or transcribe the input; respond to it directly.
120
  ```
121
 
122
+ To override the audio instruction, pass the complete replacement through `prompt=`:
123
 
124
  ```python
125
  output = model.generate(
126
  audio="question.wav",
127
+ prompt="Listen to the spoken Hokkien and reply naturally in concise Singapore Hokkien.",
128
  language="nan",
129
  )
130
  ```
131
 
132
+ For text queries, the runtime uses the corresponding direct-response text instruction recorded in
133
+ `release_config.json`, then appends the caller's `text=` in the same user turn. These are inference
134
+ instructions; their added constraint clauses were not separate training targets.
135
 
136
+ ## Limitations
137
 
138
+ - This is an active research release, not a production assistant.
139
+ - The model can mirror or restate the speaker instead of answering them.
140
+ - The prompt strongly requests Hokkien but does not provide a hard decoding guarantee.
141
+ - Automatic language-marker screens are incomplete and should not replace listening evaluation.
142
+ - The model is large and has only been validated in the release workflow on H200-class hardware.
143
+
144
+ ## Included CLI
145
+
146
+ After installation, or from the downloaded model directory:
147
+
148
+ ```bash
149
+ python inference.py \
150
+ --audio question.wav \
151
+ --output-wav answer.wav \
152
+ --output-text answer.txt
153
  ```
154
+
155
+ ## Citation
156
+
157
+ ```bibtex
158
  @misc{myna-hokkien-2026,
159
  title = {Myna-Hokkien: An Open-Source End-to-End Hokkien Spoken Dialogue Model},
160
+ author = {Matt and Ryner and Wenxuan},
161
+ year = {2026}
 
162
  }
163
  ```
164
 
165
+ ## Contact and collaboration
166
 
167
+ This is an ongoing project. For collaboration, feedback or runtime issues, contact
168
+ [Wenxuan Zhang](https://isakzhang.github.io/).
processors/generation/chat_template.jinja → chat_template.jinja RENAMED
File without changes
processors/generation/chat_template.json → chat_template.json RENAMED
File without changes
config.json CHANGED
@@ -1,133 +1,112 @@
1
  {
2
  "architectures": [
3
- "MynaHokkienForConditionalGeneration"
4
  ],
5
- "bridge_config": {
6
- "ffn": 6144,
7
- "heads": 12,
8
- "inner": 1536,
9
- "layers": 4,
10
- "q25_dim": 2048
11
- },
12
- "default_language": "nan",
13
- "default_speaker": "Ethan",
14
- "generation_embedding_config": {
15
- "embedding_dim": 2048,
16
- "num_embeddings": 152064
17
- },
18
- "generation_processor_path": "processors/generation",
19
- "input_sampling_rate": 16000,
20
- "model_type": "mynahokkien",
21
- "reasoning_adapter_config": {
22
- "alora_invocation_tokens": null,
23
- "alpha_pattern": {},
24
- "arrow_config": null,
25
- "auto_mapping": null,
26
- "base_model_name_or_path": "Qwen2.5-Omni-3B/thinker",
27
- "bias": "none",
28
- "corda_config": null,
29
- "ensure_weight_tying": false,
30
- "eva_config": null,
31
- "exclude_modules": null,
32
- "fan_in_fan_out": false,
33
- "inference_mode": true,
34
- "init_lora_weights": true,
35
- "layer_replication": null,
36
- "layers_pattern": null,
37
- "layers_to_transform": null,
38
- "loftq_config": {},
39
- "lora_alpha": 32,
40
- "lora_bias": false,
41
- "lora_dropout": 0.05,
42
- "lora_ga_config": null,
43
- "megatron_config": null,
44
- "megatron_core": "megatron.core",
45
- "modules_to_save": null,
46
- "peft_type": "LORA",
47
- "peft_version": "0.19.1",
48
- "qalora_group_size": 16,
49
- "r": 16,
50
- "rank_pattern": {},
51
- "revision": null,
52
- "target_modules": [
53
- "q_proj",
54
- "gate_proj",
55
- "up_proj",
56
- "down_proj",
57
- "k_proj",
58
- "v_proj",
59
- "o_proj"
60
  ],
61
- "target_parameters": null,
62
- "task_type": "CAUSAL_LM",
63
- "trainable_token_indices": null,
64
- "use_bdlora": null,
65
- "use_dora": false,
66
- "use_qalora": false,
67
- "use_rslora": false
68
- },
69
- "reasoning_model_config": {
70
- "_attn_implementation_autoset": true,
71
- "_name_or_path": "Qwen2.5-Omni-3B/thinker",
72
- "architectures": [
73
- "Qwen2OmniNaViTThinkerForConditionalGeneration"
74
  ],
75
- "audio_config": {
76
- "_attn_implementation_autoset": true,
 
 
 
 
 
 
 
 
 
 
 
 
77
  "_name_or_path": "",
78
- "activation_dropout": 0.0,
79
- "activation_function": "gelu",
80
  "add_cross_attention": false,
81
  "architectures": null,
82
- "attention_dropout": 0.0,
 
83
  "bad_words_ids": null,
84
  "begin_suppress_tokens": null,
85
  "bos_token_id": null,
86
  "chunk_size_feed_forward": 0,
87
  "cross_attention_hidden_size": null,
88
- "d_model": 1280,
89
  "decoder_start_token_id": null,
90
  "diversity_penalty": 0.0,
91
  "do_sample": false,
92
- "dropout": 0.0,
93
  "dtype": null,
94
  "early_stopping": false,
95
- "encoder_attention_heads": 20,
96
- "encoder_ffn_dim": 5120,
97
- "encoder_layerdrop": 0.0,
98
- "encoder_layers": 32,
99
  "encoder_no_repeat_ngram_size": 0,
100
  "eos_token_id": null,
101
  "exponential_decay_length_penalty": null,
102
  "finetuning_task": null,
103
  "forced_bos_token_id": null,
104
  "forced_eos_token_id": null,
 
 
 
105
  "id2label": {
106
  "0": "LABEL_0",
107
  "1": "LABEL_1"
108
  },
109
- "init_std": 0.02,
110
  "initializer_range": 0.02,
 
111
  "is_decoder": false,
112
  "is_encoder_decoder": false,
113
  "label2id": {
114
  "LABEL_0": 0,
115
  "LABEL_1": 1
116
  },
 
 
 
 
 
 
 
117
  "length_penalty": 1.0,
118
  "max_length": 20,
119
- "max_source_positions": 1500,
 
120
  "min_length": 0,
121
- "model_type": "qwen2_5_omni_audio_encoder",
122
- "n_window": 100,
123
  "no_repeat_ngram_size": 0,
 
124
  "num_beam_groups": 1,
125
  "num_beams": 1,
126
- "num_hidden_layers": 32,
127
- "num_mel_bins": 128,
 
128
  "num_return_sequences": 1,
129
  "output_attentions": false,
130
- "output_dim": 2048,
131
  "output_hidden_states": false,
132
  "output_scores": false,
133
  "pad_token_id": null,
@@ -138,45 +117,58 @@
138
  "repetition_penalty": 1.0,
139
  "return_dict": true,
140
  "return_dict_in_generate": false,
141
- "scale_embedding": false,
 
 
142
  "sep_token_id": null,
 
143
  "suppress_tokens": null,
144
  "task_specific_params": null,
145
  "temperature": 1.0,
146
  "tf_legacy_loss": false,
147
  "tie_encoder_decoder": false,
148
- "tie_word_embeddings": true,
149
  "tokenizer_class": null,
150
  "top_k": 50,
151
  "top_p": 1.0,
152
  "torchscript": false,
153
  "typical_p": 1.0,
154
- "use_bfloat16": false
 
 
 
155
  },
156
- "audio_end_token_id": 151648,
157
- "audio_start_token_id": 151647,
158
- "audio_token_index": 151646,
159
- "bos_token_id": 151644,
 
 
160
  "dtype": "bfloat16",
161
- "eos_token_id": 151645,
162
- "ignore_index": -100,
163
- "image_token_index": 151655,
164
- "init_std": 0.02,
165
- "initializer_range": 0.02,
166
- "model_type": "qwen2_5_omni_thinker",
167
- "pad_token_id": 151643,
168
- "position_id_per_seconds": 25,
169
  "seconds_per_chunk": 2,
 
 
 
 
 
 
170
  "text_config": {
171
  "_name_or_path": "",
172
  "add_cross_attention": false,
173
  "architectures": null,
174
- "attention_dropout": 0.0,
 
175
  "bad_words_ids": null,
176
  "begin_suppress_tokens": null,
177
  "bos_token_id": null,
178
  "chunk_size_feed_forward": 0,
179
  "cross_attention_hidden_size": null,
 
180
  "decoder_start_token_id": null,
181
  "diversity_penalty": 0.0,
182
  "do_sample": false,
@@ -188,74 +180,41 @@
188
  "finetuning_task": null,
189
  "forced_bos_token_id": null,
190
  "forced_eos_token_id": null,
 
191
  "hidden_act": "silu",
192
- "hidden_size": 2048,
193
  "id2label": {
194
  "0": "LABEL_0",
195
  "1": "LABEL_1"
196
  },
197
- "init_std": 0.02,
198
  "initializer_range": 0.02,
199
- "intermediate_size": 11008,
200
  "is_decoder": false,
201
  "is_encoder_decoder": false,
202
  "label2id": {
203
  "LABEL_0": 0,
204
  "LABEL_1": 1
205
  },
206
- "layer_types": [
207
- "full_attention",
208
- "full_attention",
209
- "full_attention",
210
- "full_attention",
211
- "full_attention",
212
- "full_attention",
213
- "full_attention",
214
- "full_attention",
215
- "full_attention",
216
- "full_attention",
217
- "full_attention",
218
- "full_attention",
219
- "full_attention",
220
- "full_attention",
221
- "full_attention",
222
- "full_attention",
223
- "full_attention",
224
- "full_attention",
225
- "full_attention",
226
- "full_attention",
227
- "full_attention",
228
- "full_attention",
229
- "full_attention",
230
- "full_attention",
231
- "full_attention",
232
- "full_attention",
233
- "full_attention",
234
- "full_attention",
235
- "full_attention",
236
- "full_attention",
237
- "full_attention",
238
- "full_attention",
239
- "full_attention",
240
- "full_attention",
241
- "full_attention",
242
- "full_attention"
243
- ],
244
  "length_penalty": 1.0,
245
  "max_length": 20,
246
- "max_position_embeddings": 32768,
247
- "max_window_layers": 70,
248
  "min_length": 0,
249
- "model_type": "qwen2_5_omni_text",
 
 
250
  "no_repeat_ngram_size": 0,
 
251
  "num_attention_heads": 16,
252
  "num_beam_groups": 1,
253
  "num_beams": 1,
254
- "num_hidden_layers": 36,
 
 
255
  "num_key_value_heads": 2,
256
  "num_return_sequences": 1,
257
  "output_attentions": false,
258
  "output_hidden_states": false,
 
259
  "output_scores": false,
260
  "pad_token_id": null,
261
  "prefix": null,
@@ -266,29 +225,20 @@
266
  "return_dict": true,
267
  "return_dict_in_generate": false,
268
  "rms_norm_eps": 1e-06,
269
- "rope_parameters": {
270
  "interleaved": true,
271
- "mrope_interleaved": true,
272
  "mrope_section": [
273
  24,
274
  20,
275
  20
276
  ],
277
- "rope_theta": 1000000,
278
  "rope_type": "default",
279
  "type": "default"
280
  },
281
- "rope_scaling": {
282
- "mrope_section": [
283
- 16,
284
- 24,
285
- 24
286
- ],
287
- "rope_type": "default",
288
- "type": "default"
289
- },
290
- "rope_theta": 1000000.0,
291
  "sep_token_id": null,
 
292
  "sliding_window": null,
293
  "suppress_tokens": null,
294
  "task_specific_params": null,
@@ -304,52 +254,48 @@
304
  "use_bfloat16": false,
305
  "use_cache": true,
306
  "use_sliding_window": false,
307
- "vocab_size": 151936
308
  },
309
- "tie_word_embeddings": false,
310
- "use_cache": true,
311
- "user_token_id": 872,
312
- "video_token_index": 151656,
313
- "vision_config": {
314
- "_attn_implementation_autoset": true,
315
  "_name_or_path": "",
 
 
316
  "add_cross_attention": false,
317
  "architectures": null,
 
318
  "bad_words_ids": null,
319
  "begin_suppress_tokens": null,
320
  "bos_token_id": null,
321
  "chunk_size_feed_forward": 0,
 
322
  "cross_attention_hidden_size": null,
 
323
  "decoder_start_token_id": null,
324
- "depth": 32,
325
  "diversity_penalty": 0.0,
326
  "do_sample": false,
 
 
327
  "dtype": null,
328
  "early_stopping": false,
329
- "embed_dim": 1280,
 
 
330
  "encoder_no_repeat_ngram_size": 0,
331
  "eos_token_id": null,
332
  "exponential_decay_length_penalty": null,
333
  "finetuning_task": null,
334
  "forced_bos_token_id": null,
335
  "forced_eos_token_id": null,
336
- "fullatt_block_indexes": [
337
- 7,
338
- 15,
339
- 23,
340
- 31
341
- ],
342
- "hidden_act": "silu",
343
- "hidden_size": 1280,
344
  "id2label": {
345
  "0": "LABEL_0",
346
  "1": "LABEL_1"
347
  },
348
- "in_channels": 3,
349
- "in_chans": 3,
350
- "init_std": 0.02,
351
  "initializer_range": 0.02,
352
- "intermediate_size": 3420,
353
  "is_decoder": false,
354
  "is_encoder_decoder": false,
355
  "label2id": {
@@ -358,19 +304,22 @@
358
  },
359
  "length_penalty": 1.0,
360
  "max_length": 20,
 
361
  "min_length": 0,
362
- "model_type": "qwen2_5_omni_vision_encoder",
 
 
363
  "no_repeat_ngram_size": 0,
364
  "num_beam_groups": 1,
365
  "num_beams": 1,
366
- "num_heads": 16,
 
367
  "num_return_sequences": 1,
368
- "out_hidden_size": 2048,
369
  "output_attentions": false,
 
370
  "output_hidden_states": false,
371
  "output_scores": false,
372
  "pad_token_id": null,
373
- "patch_size": 14,
374
  "prefix": null,
375
  "problem_type": null,
376
  "pruned_heads": {},
@@ -378,53 +327,46 @@
378
  "repetition_penalty": 1.0,
379
  "return_dict": true,
380
  "return_dict_in_generate": false,
 
381
  "sep_token_id": null,
382
- "spatial_merge_size": 2,
383
- "spatial_patch_size": 14,
384
  "suppress_tokens": null,
385
  "task_specific_params": null,
386
  "temperature": 1.0,
387
- "temporal_patch_size": 2,
388
  "tf_legacy_loss": false,
389
  "tie_encoder_decoder": false,
390
  "tie_word_embeddings": true,
391
  "tokenizer_class": null,
392
- "tokens_per_second": 25,
393
  "top_k": 50,
394
  "top_p": 1.0,
395
  "torchscript": false,
396
  "typical_p": 1.0,
397
- "use_bfloat16": false,
398
- "window_size": 112
399
  },
400
- "vision_end_token_id": 151653,
401
- "vision_start_token_id": 151652,
402
- "vision_token_id": 151654
403
- },
404
- "reasoning_processor_path": "processors/reasoning",
405
- "release_epoch": 3,
406
- "sampling_rate": 24000,
407
- "speech_generator_config": {
408
- "accept_hidden_layer": 24,
409
  "audio_end_token_id": 151670,
410
  "audio_start_token_id": 151669,
411
  "audio_token_id": 151675,
412
- "audio_tower_config": {},
413
- "code_predictor_config": {
 
 
 
 
 
414
  "_name_or_path": "",
415
  "add_cross_attention": false,
416
  "architectures": null,
417
  "attention_bias": false,
418
- "attention_dropout": 0,
419
  "bad_words_ids": null,
420
  "begin_suppress_tokens": null,
421
  "bos_token_id": null,
422
  "chunk_size_feed_forward": 0,
423
  "cross_attention_hidden_size": null,
 
424
  "decoder_start_token_id": null,
425
  "diversity_penalty": 0.0,
426
  "do_sample": false,
427
- "dtype": "bfloat16",
428
  "early_stopping": false,
429
  "encoder_no_repeat_ngram_size": 0,
430
  "eos_token_id": null,
@@ -434,42 +376,39 @@
434
  "forced_eos_token_id": null,
435
  "head_dim": 128,
436
  "hidden_act": "silu",
437
- "hidden_size": 1024,
438
  "id2label": {
439
  "0": "LABEL_0",
440
  "1": "LABEL_1"
441
  },
442
  "initializer_range": 0.02,
443
- "intermediate_size": 3072,
444
  "is_decoder": false,
445
  "is_encoder_decoder": false,
446
  "label2id": {
447
  "LABEL_0": 0,
448
  "LABEL_1": 1
449
  },
450
- "layer_types": [
451
- "full_attention",
452
- "full_attention",
453
- "full_attention",
454
- "full_attention",
455
- "full_attention"
456
- ],
457
  "length_penalty": 1.0,
458
  "max_length": 20,
459
- "max_position_embeddings": 32768,
460
- "max_window_layers": 28,
461
  "min_length": 0,
462
- "model_type": "qwen3_omni_moe_talker_code_predictor",
 
 
463
  "no_repeat_ngram_size": 0,
464
- "num_attention_heads": 16,
 
465
  "num_beam_groups": 1,
466
  "num_beams": 1,
467
- "num_code_groups": 16,
468
- "num_hidden_layers": 5,
469
- "num_key_value_heads": 8,
 
470
  "num_return_sequences": 1,
471
  "output_attentions": false,
472
  "output_hidden_states": false,
 
473
  "output_scores": false,
474
  "pad_token_id": null,
475
  "prefix": null,
@@ -480,13 +419,21 @@
480
  "return_dict": true,
481
  "return_dict_in_generate": false,
482
  "rms_norm_eps": 1e-06,
483
- "rope_parameters": {
484
- "rope_theta": 1000000,
485
- "rope_type": "default"
 
 
 
 
 
 
 
486
  },
487
- "rope_scaling": null,
488
- "rope_theta": 10000,
489
  "sep_token_id": null,
 
490
  "sliding_window": null,
491
  "suppress_tokens": null,
492
  "task_specific_params": null,
@@ -501,46 +448,32 @@
501
  "typical_p": 1.0,
502
  "use_bfloat16": false,
503
  "use_cache": true,
 
504
  "use_sliding_window": false,
505
- "vocab_size": 2048
506
  },
507
- "codec_bos_id": 2149,
508
- "codec_config": {},
509
- "codec_eos_token_id": 2150,
510
- "codec_nothink_id": 2155,
511
- "codec_pad_id": 2148,
512
- "codec_think_bos_id": 2156,
513
- "codec_think_eos_id": 2157,
514
- "dtype": "bfloat16",
515
- "image_token_id": 151655,
516
- "initializer_range": 0.02,
517
- "model_type": "",
518
- "num_code_groups": 16,
519
- "output_router_logits": false,
520
- "position_id_per_seconds": 13,
521
- "seconds_per_chunk": 2,
522
- "spatial_merge_size": 2,
523
- "speaker_id": {
524
- "aiden": 2303,
525
- "chelsie": 2301,
526
- "ethan": 2302
527
- },
528
- "text_config": {
529
  "_name_or_path": "",
530
  "add_cross_attention": false,
 
531
  "architectures": null,
532
- "attention_bias": false,
533
- "attention_dropout": 0,
534
  "bad_words_ids": null,
535
  "begin_suppress_tokens": null,
536
  "bos_token_id": null,
537
  "chunk_size_feed_forward": 0,
538
  "cross_attention_hidden_size": null,
539
- "decoder_sparse_step": 1,
540
  "decoder_start_token_id": null,
 
 
 
 
 
 
541
  "diversity_penalty": 0.0,
542
  "do_sample": false,
543
- "dtype": "bfloat16",
544
  "early_stopping": false,
545
  "encoder_no_repeat_ngram_size": 0,
546
  "eos_token_id": null,
@@ -548,15 +481,17 @@
548
  "finetuning_task": null,
549
  "forced_bos_token_id": null,
550
  "forced_eos_token_id": null,
551
- "head_dim": 128,
552
- "hidden_act": "silu",
553
- "hidden_size": 1024,
554
  "id2label": {
555
  "0": "LABEL_0",
556
  "1": "LABEL_1"
557
  },
 
 
 
558
  "initializer_range": 0.02,
559
- "intermediate_size": 2048,
560
  "is_decoder": false,
561
  "is_encoder_decoder": false,
562
  "label2id": {
@@ -565,27 +500,20 @@
565
  },
566
  "length_penalty": 1.0,
567
  "max_length": 20,
568
- "max_position_embeddings": 65536,
569
  "min_length": 0,
570
- "mlp_only_layers": [],
571
- "model_type": "qwen3_omni_moe_talker_text",
572
- "moe_intermediate_size": 384,
573
  "no_repeat_ngram_size": 0,
574
- "norm_topk_prob": true,
575
- "num_attention_heads": 16,
576
  "num_beam_groups": 1,
577
  "num_beams": 1,
578
- "num_experts": 128,
579
- "num_experts_per_tok": 6,
580
- "num_hidden_layers": 20,
581
- "num_key_value_heads": 2,
582
- "num_local_experts": 128,
583
  "num_return_sequences": 1,
 
584
  "output_attentions": false,
585
  "output_hidden_states": false,
586
- "output_router_logits": false,
587
  "output_scores": false,
588
  "pad_token_id": null,
 
589
  "prefix": null,
590
  "problem_type": null,
591
  "pruned_heads": {},
@@ -593,93 +521,30 @@
593
  "repetition_penalty": 1.0,
594
  "return_dict": true,
595
  "return_dict_in_generate": false,
596
- "rms_norm_eps": 1e-06,
597
- "rope_parameters": {
598
- "interleaved": true,
599
- "mrope_section": [
600
- 24,
601
- 20,
602
- 20
603
- ],
604
- "rope_theta": 1000000,
605
- "rope_type": "default",
606
- "type": "default"
607
- },
608
- "rope_scaling": {
609
- "interleaved": true,
610
- "mrope_section": [
611
- 24,
612
- 20,
613
- 20
614
- ],
615
- "rope_type": "default",
616
- "type": "default"
617
- },
618
- "rope_theta": 10000,
619
- "router_aux_loss_coef": 0.001,
620
  "sep_token_id": null,
621
- "shared_expert_intermediate_size": 768,
622
- "sliding_window": null,
623
  "suppress_tokens": null,
624
  "task_specific_params": null,
625
  "temperature": 1.0,
 
626
  "tf_legacy_loss": false,
627
  "tie_encoder_decoder": false,
628
- "tie_word_embeddings": false,
629
  "tokenizer_class": null,
 
630
  "top_k": 50,
631
  "top_p": 1.0,
632
  "torchscript": false,
633
  "typical_p": 1.0,
634
- "use_bfloat16": false,
635
- "use_cache": true,
636
- "use_sliding_window": false,
637
- "vocab_size": 3072
638
  },
639
- "thinker_hidden_size": 2048,
640
- "tie_word_embeddings": false,
641
- "video_token_id": 151656,
642
  "vision_start_token_id": 151652
643
  },
644
- "torch_dtype": "float16",
645
  "transformers_version": "4.57.1",
646
- "waveform_decoder_config": {
647
- "attention_bias": false,
648
- "attention_dropout": 0.0,
649
- "codebook_dim": 512,
650
- "codebook_size": 2048,
651
- "decoder_dim": 1536,
652
- "dtype": "bfloat16",
653
- "hidden_act": "silu",
654
- "hidden_size": 1024,
655
- "initializer_range": 0.02,
656
- "intermediate_size": 3072,
657
- "layer_scale_initial_scale": 0.01,
658
- "max_position_embeddings": 8000,
659
- "model_type": "",
660
- "num_attention_heads": 16,
661
- "num_hidden_layers": 8,
662
- "num_key_value_heads": 16,
663
- "num_quantizers": 16,
664
- "num_semantic_quantizers": 1,
665
- "rms_norm_eps": 1e-05,
666
- "rope_parameters": {
667
- "rope_theta": 10000,
668
- "rope_type": "default"
669
- },
670
- "rope_theta": 10000,
671
- "semantic_codebook_size": 4096,
672
- "sliding_window": 72,
673
- "upsample_rates": [
674
- 8,
675
- 5,
676
- 4,
677
- 3
678
- ],
679
- "upsampling_ratios": [
680
- 2,
681
- 2
682
- ],
683
- "vector_quantization_hidden_dimension": 512
684
- }
685
  }
 
1
  {
2
  "architectures": [
3
+ "Qwen3OmniMoeForConditionalGeneration"
4
  ],
5
+ "assistant_token_id": 77091,
6
+ "code2wav_config": {
7
+ "attention_bias": false,
8
+ "attention_dropout": 0.0,
9
+ "codebook_dim": 512,
10
+ "codebook_size": 2048,
11
+ "decoder_dim": 1536,
12
+ "dtype": "bfloat16",
13
+ "hidden_act": "silu",
14
+ "hidden_size": 1024,
15
+ "intermediate_size": 3072,
16
+ "layer_scale_initial_scale": 0.01,
17
+ "max_position_embeddings": 8000,
18
+ "model_type": "",
19
+ "num_attention_heads": 16,
20
+ "num_hidden_layers": 8,
21
+ "num_key_value_heads": 16,
22
+ "num_quantizers": 16,
23
+ "num_semantic_quantizers": 1,
24
+ "rms_norm_eps": 1e-05,
25
+ "rope_theta": 10000,
26
+ "semantic_codebook_size": 4096,
27
+ "sliding_window": 72,
28
+ "upsample_rates": [
29
+ 8,
30
+ 5,
31
+ 4,
32
+ 3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
33
  ],
34
+ "upsampling_ratios": [
35
+ 2,
36
+ 2
 
 
 
 
 
 
 
 
 
 
37
  ],
38
+ "vector_quantization_hidden_dimension": 512
39
+ },
40
+ "dtype": "bfloat16",
41
+ "enable_audio_output": true,
42
+ "im_end_token_id": 151645,
43
+ "im_start_token_id": 151644,
44
+ "model_type": "qwen3_omni_moe",
45
+ "system_token_id": 8948,
46
+ "talker_config": {
47
+ "accept_hidden_layer": 24,
48
+ "audio_end_token_id": 151670,
49
+ "audio_start_token_id": 151669,
50
+ "audio_token_id": 151675,
51
+ "code_predictor_config": {
52
  "_name_or_path": "",
 
 
53
  "add_cross_attention": false,
54
  "architectures": null,
55
+ "attention_bias": false,
56
+ "attention_dropout": 0,
57
  "bad_words_ids": null,
58
  "begin_suppress_tokens": null,
59
  "bos_token_id": null,
60
  "chunk_size_feed_forward": 0,
61
  "cross_attention_hidden_size": null,
 
62
  "decoder_start_token_id": null,
63
  "diversity_penalty": 0.0,
64
  "do_sample": false,
 
65
  "dtype": null,
66
  "early_stopping": false,
 
 
 
 
67
  "encoder_no_repeat_ngram_size": 0,
68
  "eos_token_id": null,
69
  "exponential_decay_length_penalty": null,
70
  "finetuning_task": null,
71
  "forced_bos_token_id": null,
72
  "forced_eos_token_id": null,
73
+ "head_dim": 128,
74
+ "hidden_act": "silu",
75
+ "hidden_size": 1024,
76
  "id2label": {
77
  "0": "LABEL_0",
78
  "1": "LABEL_1"
79
  },
 
80
  "initializer_range": 0.02,
81
+ "intermediate_size": 3072,
82
  "is_decoder": false,
83
  "is_encoder_decoder": false,
84
  "label2id": {
85
  "LABEL_0": 0,
86
  "LABEL_1": 1
87
  },
88
+ "layer_types": [
89
+ "full_attention",
90
+ "full_attention",
91
+ "full_attention",
92
+ "full_attention",
93
+ "full_attention"
94
+ ],
95
  "length_penalty": 1.0,
96
  "max_length": 20,
97
+ "max_position_embeddings": 32768,
98
+ "max_window_layers": 28,
99
  "min_length": 0,
100
+ "model_type": "qwen3_omni_moe_talker_code_predictor",
 
101
  "no_repeat_ngram_size": 0,
102
+ "num_attention_heads": 16,
103
  "num_beam_groups": 1,
104
  "num_beams": 1,
105
+ "num_code_groups": 16,
106
+ "num_hidden_layers": 5,
107
+ "num_key_value_heads": 8,
108
  "num_return_sequences": 1,
109
  "output_attentions": false,
 
110
  "output_hidden_states": false,
111
  "output_scores": false,
112
  "pad_token_id": null,
 
117
  "repetition_penalty": 1.0,
118
  "return_dict": true,
119
  "return_dict_in_generate": false,
120
+ "rms_norm_eps": 1e-06,
121
+ "rope_scaling": null,
122
+ "rope_theta": 1000000,
123
  "sep_token_id": null,
124
+ "sliding_window": null,
125
  "suppress_tokens": null,
126
  "task_specific_params": null,
127
  "temperature": 1.0,
128
  "tf_legacy_loss": false,
129
  "tie_encoder_decoder": false,
130
+ "tie_word_embeddings": false,
131
  "tokenizer_class": null,
132
  "top_k": 50,
133
  "top_p": 1.0,
134
  "torchscript": false,
135
  "typical_p": 1.0,
136
+ "use_bfloat16": false,
137
+ "use_cache": true,
138
+ "use_sliding_window": false,
139
+ "vocab_size": 2048
140
  },
141
+ "codec_bos_id": 2149,
142
+ "codec_eos_token_id": 2150,
143
+ "codec_nothink_id": 2155,
144
+ "codec_pad_id": 2148,
145
+ "codec_think_bos_id": 2156,
146
+ "codec_think_eos_id": 2157,
147
  "dtype": "bfloat16",
148
+ "image_token_id": 151655,
149
+ "model_type": "",
150
+ "num_code_groups": 16,
151
+ "output_router_logits": false,
152
+ "position_id_per_seconds": 13,
 
 
 
153
  "seconds_per_chunk": 2,
154
+ "spatial_merge_size": 2,
155
+ "speaker_id": {
156
+ "aiden": 2303,
157
+ "chelsie": 2301,
158
+ "ethan": 2302
159
+ },
160
  "text_config": {
161
  "_name_or_path": "",
162
  "add_cross_attention": false,
163
  "architectures": null,
164
+ "attention_bias": false,
165
+ "attention_dropout": 0,
166
  "bad_words_ids": null,
167
  "begin_suppress_tokens": null,
168
  "bos_token_id": null,
169
  "chunk_size_feed_forward": 0,
170
  "cross_attention_hidden_size": null,
171
+ "decoder_sparse_step": 1,
172
  "decoder_start_token_id": null,
173
  "diversity_penalty": 0.0,
174
  "do_sample": false,
 
180
  "finetuning_task": null,
181
  "forced_bos_token_id": null,
182
  "forced_eos_token_id": null,
183
+ "head_dim": 128,
184
  "hidden_act": "silu",
185
+ "hidden_size": 1024,
186
  "id2label": {
187
  "0": "LABEL_0",
188
  "1": "LABEL_1"
189
  },
 
190
  "initializer_range": 0.02,
191
+ "intermediate_size": 2048,
192
  "is_decoder": false,
193
  "is_encoder_decoder": false,
194
  "label2id": {
195
  "LABEL_0": 0,
196
  "LABEL_1": 1
197
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
198
  "length_penalty": 1.0,
199
  "max_length": 20,
200
+ "max_position_embeddings": 65536,
 
201
  "min_length": 0,
202
+ "mlp_only_layers": [],
203
+ "model_type": "qwen3_omni_moe_talker_text",
204
+ "moe_intermediate_size": 384,
205
  "no_repeat_ngram_size": 0,
206
+ "norm_topk_prob": true,
207
  "num_attention_heads": 16,
208
  "num_beam_groups": 1,
209
  "num_beams": 1,
210
+ "num_experts": 128,
211
+ "num_experts_per_tok": 6,
212
+ "num_hidden_layers": 20,
213
  "num_key_value_heads": 2,
214
  "num_return_sequences": 1,
215
  "output_attentions": false,
216
  "output_hidden_states": false,
217
+ "output_router_logits": false,
218
  "output_scores": false,
219
  "pad_token_id": null,
220
  "prefix": null,
 
225
  "return_dict": true,
226
  "return_dict_in_generate": false,
227
  "rms_norm_eps": 1e-06,
228
+ "rope_scaling": {
229
  "interleaved": true,
 
230
  "mrope_section": [
231
  24,
232
  20,
233
  20
234
  ],
 
235
  "rope_type": "default",
236
  "type": "default"
237
  },
238
+ "rope_theta": 1000000,
239
+ "router_aux_loss_coef": 0.001,
 
 
 
 
 
 
 
 
240
  "sep_token_id": null,
241
+ "shared_expert_intermediate_size": 768,
242
  "sliding_window": null,
243
  "suppress_tokens": null,
244
  "task_specific_params": null,
 
254
  "use_bfloat16": false,
255
  "use_cache": true,
256
  "use_sliding_window": false,
257
+ "vocab_size": 3072
258
  },
259
+ "thinker_hidden_size": 2048,
260
+ "video_token_id": 151656,
261
+ "vision_start_token_id": 151652
262
+ },
263
+ "thinker_config": {
264
+ "audio_config": {
265
  "_name_or_path": "",
266
+ "activation_dropout": 0,
267
+ "activation_function": "gelu",
268
  "add_cross_attention": false,
269
  "architectures": null,
270
+ "attention_dropout": 0,
271
  "bad_words_ids": null,
272
  "begin_suppress_tokens": null,
273
  "bos_token_id": null,
274
  "chunk_size_feed_forward": 0,
275
+ "conv_chunksize": 500,
276
  "cross_attention_hidden_size": null,
277
+ "d_model": 1280,
278
  "decoder_start_token_id": null,
 
279
  "diversity_penalty": 0.0,
280
  "do_sample": false,
281
+ "downsample_hidden_size": 480,
282
+ "dropout": 0,
283
  "dtype": null,
284
  "early_stopping": false,
285
+ "encoder_attention_heads": 20,
286
+ "encoder_ffn_dim": 5120,
287
+ "encoder_layers": 32,
288
  "encoder_no_repeat_ngram_size": 0,
289
  "eos_token_id": null,
290
  "exponential_decay_length_penalty": null,
291
  "finetuning_task": null,
292
  "forced_bos_token_id": null,
293
  "forced_eos_token_id": null,
 
 
 
 
 
 
 
 
294
  "id2label": {
295
  "0": "LABEL_0",
296
  "1": "LABEL_1"
297
  },
 
 
 
298
  "initializer_range": 0.02,
 
299
  "is_decoder": false,
300
  "is_encoder_decoder": false,
301
  "label2id": {
 
304
  },
305
  "length_penalty": 1.0,
306
  "max_length": 20,
307
+ "max_source_positions": 1500,
308
  "min_length": 0,
309
+ "model_type": "qwen3_omni_moe_audio_encoder",
310
+ "n_window": 50,
311
+ "n_window_infer": 800,
312
  "no_repeat_ngram_size": 0,
313
  "num_beam_groups": 1,
314
  "num_beams": 1,
315
+ "num_hidden_layers": 32,
316
+ "num_mel_bins": 128,
317
  "num_return_sequences": 1,
 
318
  "output_attentions": false,
319
+ "output_dim": 2048,
320
  "output_hidden_states": false,
321
  "output_scores": false,
322
  "pad_token_id": null,
 
323
  "prefix": null,
324
  "problem_type": null,
325
  "pruned_heads": {},
 
327
  "repetition_penalty": 1.0,
328
  "return_dict": true,
329
  "return_dict_in_generate": false,
330
+ "scale_embedding": false,
331
  "sep_token_id": null,
 
 
332
  "suppress_tokens": null,
333
  "task_specific_params": null,
334
  "temperature": 1.0,
 
335
  "tf_legacy_loss": false,
336
  "tie_encoder_decoder": false,
337
  "tie_word_embeddings": true,
338
  "tokenizer_class": null,
 
339
  "top_k": 50,
340
  "top_p": 1.0,
341
  "torchscript": false,
342
  "typical_p": 1.0,
343
+ "use_bfloat16": false
 
344
  },
 
 
 
 
 
 
 
 
 
345
  "audio_end_token_id": 151670,
346
  "audio_start_token_id": 151669,
347
  "audio_token_id": 151675,
348
+ "dtype": "bfloat16",
349
+ "image_token_id": 151655,
350
+ "initializer_range": 0.02,
351
+ "model_type": "qwen3_omni_moe_thinker",
352
+ "position_id_per_seconds": 13,
353
+ "seconds_per_chunk": 2,
354
+ "text_config": {
355
  "_name_or_path": "",
356
  "add_cross_attention": false,
357
  "architectures": null,
358
  "attention_bias": false,
359
+ "attention_dropout": 0.0,
360
  "bad_words_ids": null,
361
  "begin_suppress_tokens": null,
362
  "bos_token_id": null,
363
  "chunk_size_feed_forward": 0,
364
  "cross_attention_hidden_size": null,
365
+ "decoder_sparse_step": 1,
366
  "decoder_start_token_id": null,
367
  "diversity_penalty": 0.0,
368
  "do_sample": false,
369
+ "dtype": null,
370
  "early_stopping": false,
371
  "encoder_no_repeat_ngram_size": 0,
372
  "eos_token_id": null,
 
376
  "forced_eos_token_id": null,
377
  "head_dim": 128,
378
  "hidden_act": "silu",
379
+ "hidden_size": 2048,
380
  "id2label": {
381
  "0": "LABEL_0",
382
  "1": "LABEL_1"
383
  },
384
  "initializer_range": 0.02,
385
+ "intermediate_size": 768,
386
  "is_decoder": false,
387
  "is_encoder_decoder": false,
388
  "label2id": {
389
  "LABEL_0": 0,
390
  "LABEL_1": 1
391
  },
 
 
 
 
 
 
 
392
  "length_penalty": 1.0,
393
  "max_length": 20,
394
+ "max_position_embeddings": 65536,
 
395
  "min_length": 0,
396
+ "mlp_only_layers": [],
397
+ "model_type": "qwen3_omni_moe_text",
398
+ "moe_intermediate_size": 768,
399
  "no_repeat_ngram_size": 0,
400
+ "norm_topk_prob": true,
401
+ "num_attention_heads": 32,
402
  "num_beam_groups": 1,
403
  "num_beams": 1,
404
+ "num_experts": 128,
405
+ "num_experts_per_tok": 8,
406
+ "num_hidden_layers": 48,
407
+ "num_key_value_heads": 4,
408
  "num_return_sequences": 1,
409
  "output_attentions": false,
410
  "output_hidden_states": false,
411
+ "output_router_logits": false,
412
  "output_scores": false,
413
  "pad_token_id": null,
414
  "prefix": null,
 
419
  "return_dict": true,
420
  "return_dict_in_generate": false,
421
  "rms_norm_eps": 1e-06,
422
+ "rope_scaling": {
423
+ "interleaved": true,
424
+ "mrope_interleaved": true,
425
+ "mrope_section": [
426
+ 24,
427
+ 20,
428
+ 20
429
+ ],
430
+ "rope_type": "default",
431
+ "type": "default"
432
  },
433
+ "rope_theta": 1000000,
434
+ "router_aux_loss_coef": 0.001,
435
  "sep_token_id": null,
436
+ "shared_expert_intermediate_size": 0,
437
  "sliding_window": null,
438
  "suppress_tokens": null,
439
  "task_specific_params": null,
 
448
  "typical_p": 1.0,
449
  "use_bfloat16": false,
450
  "use_cache": true,
451
+ "use_qk_norm": true,
452
  "use_sliding_window": false,
453
+ "vocab_size": 152064
454
  },
455
+ "user_token_id": 872,
456
+ "video_token_id": 151656,
457
+ "vision_config": {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
458
  "_name_or_path": "",
459
  "add_cross_attention": false,
460
+ "apply_vit_abs_pos_embed": true,
461
  "architectures": null,
 
 
462
  "bad_words_ids": null,
463
  "begin_suppress_tokens": null,
464
  "bos_token_id": null,
465
  "chunk_size_feed_forward": 0,
466
  "cross_attention_hidden_size": null,
 
467
  "decoder_start_token_id": null,
468
+ "deepstack_visual_indexes": [
469
+ 8,
470
+ 16,
471
+ 24
472
+ ],
473
+ "depth": 27,
474
  "diversity_penalty": 0.0,
475
  "do_sample": false,
476
+ "dtype": null,
477
  "early_stopping": false,
478
  "encoder_no_repeat_ngram_size": 0,
479
  "eos_token_id": null,
 
481
  "finetuning_task": null,
482
  "forced_bos_token_id": null,
483
  "forced_eos_token_id": null,
484
+ "hidden_act": "gelu_pytorch_tanh",
485
+ "hidden_size": 1152,
 
486
  "id2label": {
487
  "0": "LABEL_0",
488
  "1": "LABEL_1"
489
  },
490
+ "image_size": 768,
491
+ "in_channels": 3,
492
+ "in_chans": 3,
493
  "initializer_range": 0.02,
494
+ "intermediate_size": 4304,
495
  "is_decoder": false,
496
  "is_encoder_decoder": false,
497
  "label2id": {
 
500
  },
501
  "length_penalty": 1.0,
502
  "max_length": 20,
 
503
  "min_length": 0,
504
+ "model_type": "qwen3_omni_moe_vision_encoder",
 
 
505
  "no_repeat_ngram_size": 0,
 
 
506
  "num_beam_groups": 1,
507
  "num_beams": 1,
508
+ "num_heads": 16,
509
+ "num_position_embeddings": 2304,
 
 
 
510
  "num_return_sequences": 1,
511
+ "out_hidden_size": 2048,
512
  "output_attentions": false,
513
  "output_hidden_states": false,
 
514
  "output_scores": false,
515
  "pad_token_id": null,
516
+ "patch_size": 16,
517
  "prefix": null,
518
  "problem_type": null,
519
  "pruned_heads": {},
 
521
  "repetition_penalty": 1.0,
522
  "return_dict": true,
523
  "return_dict_in_generate": false,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
524
  "sep_token_id": null,
525
+ "spatial_merge_size": 2,
526
+ "spatial_patch_size": 16,
527
  "suppress_tokens": null,
528
  "task_specific_params": null,
529
  "temperature": 1.0,
530
+ "temporal_patch_size": 2,
531
  "tf_legacy_loss": false,
532
  "tie_encoder_decoder": false,
533
+ "tie_word_embeddings": true,
534
  "tokenizer_class": null,
535
+ "tokens_per_second": 2,
536
  "top_k": 50,
537
  "top_p": 1.0,
538
  "torchscript": false,
539
  "typical_p": 1.0,
540
+ "use_bfloat16": false
 
 
 
541
  },
542
+ "vision_end_token_id": 151653,
 
 
543
  "vision_start_token_id": 151652
544
  },
 
545
  "transformers_version": "4.57.1",
546
+ "tts_bos_token_id": 151672,
547
+ "tts_eos_token_id": 151673,
548
+ "tts_pad_token_id": 151671,
549
+ "user_token_id": 872
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
550
  }
configuration_mynahokkien.py DELETED
@@ -1,41 +0,0 @@
1
- from __future__ import annotations
2
-
3
- from transformers import PretrainedConfig
4
-
5
-
6
- class MynaHokkienConfig(PretrainedConfig):
7
- """Configuration for the unified MynaHokkien inference checkpoint."""
8
-
9
- model_type = "mynahokkien"
10
-
11
- def __init__(
12
- self,
13
- *,
14
- reasoning_model_config=None,
15
- reasoning_adapter_config=None,
16
- speech_generator_config=None,
17
- waveform_decoder_config=None,
18
- generation_embedding_config=None,
19
- bridge_config=None,
20
- reasoning_processor_path="processors/reasoning",
21
- generation_processor_path="processors/generation",
22
- sampling_rate=24000,
23
- input_sampling_rate=16000,
24
- default_language="nan",
25
- default_speaker="Ethan",
26
- **kwargs,
27
- ):
28
- super().__init__(**kwargs)
29
- self.reasoning_model_config = reasoning_model_config or {}
30
- self.reasoning_adapter_config = reasoning_adapter_config or {}
31
- self.speech_generator_config = speech_generator_config or {}
32
- self.waveform_decoder_config = waveform_decoder_config or {}
33
- self.generation_embedding_config = generation_embedding_config or {}
34
- self.bridge_config = bridge_config or {}
35
- self.reasoning_processor_path = reasoning_processor_path
36
- self.generation_processor_path = generation_processor_path
37
- self.sampling_rate = sampling_rate
38
- self.input_sampling_rate = input_sampling_rate
39
- self.default_language = default_language
40
- self.default_speaker = default_speaker
41
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
generation_config.json CHANGED
@@ -1,4 +1,8 @@
1
  {
2
- "max_audio_tokens": 4096,
3
- "max_new_tokens": 512
 
 
 
 
4
  }
 
1
  {
2
+ "talker_max_new_tokens": 4096,
3
+ "talker_repetition_penalty": 1.05,
4
+ "talker_temperature": 0.9,
5
+ "talker_top_k": 50,
6
+ "talker_top_p": 1.0,
7
+ "transformers_version": "4.57.1"
8
  }
inference.py CHANGED
@@ -1,5 +1,5 @@
1
  #!/usr/bin/env python3
2
- """Simple one-request inference CLI for the unified Myna-Hokkien release."""
3
  from __future__ import annotations
4
 
5
  import argparse
@@ -14,16 +14,18 @@ from mynahokkien import MynaHokkien
14
  def parse_args() -> argparse.Namespace:
15
  parser = argparse.ArgumentParser()
16
  parser.add_argument("--model", default="iNLP-Lab/MynaHokkien")
17
- modality = parser.add_mutually_exclusive_group(required=True)
18
- modality.add_argument("--audio", type=Path)
19
- modality.add_argument("--text")
20
- parser.add_argument("--prompt")
21
  parser.add_argument("--output-wav", type=Path, default=Path("output.wav"))
22
  parser.add_argument("--output-text", type=Path)
23
  parser.add_argument("--device", default="cuda:0")
 
24
  parser.add_argument("--seed", type=int, default=1234)
25
- parser.add_argument("--max-new-tokens", type=int, default=512)
26
  parser.add_argument("--max-audio-tokens", type=int, default=4096)
 
27
  parser.add_argument("--local-files-only", action="store_true")
28
  return parser.parse_args()
29
 
@@ -31,14 +33,13 @@ def parse_args() -> argparse.Namespace:
31
  def main() -> None:
32
  args = parse_args()
33
  if args.prompt is not None and args.audio is None:
34
- raise ValueError("--prompt is only valid with --audio; include the instruction in --text")
35
- if args.audio is not None and not args.audio.is_file():
36
- raise FileNotFoundError(args.audio)
37
-
38
  model = MynaHokkien.from_pretrained(
39
  args.model,
40
  device_map=args.device,
41
- dtype=torch.float16,
 
42
  local_files_only=args.local_files_only,
43
  )
44
  output = model.generate(
@@ -46,20 +47,19 @@ def main() -> None:
46
  text=args.text,
47
  prompt=args.prompt,
48
  language="nan",
 
49
  seed=args.seed,
50
  max_new_tokens=args.max_new_tokens,
51
  max_audio_tokens=args.max_audio_tokens,
52
- return_text=True,
53
- return_audio=True,
54
  )
55
- if not output.text or output.audio is None:
56
- raise RuntimeError("model did not return both text and audio")
57
 
58
  args.output_wav.parent.mkdir(parents=True, exist_ok=True)
59
  sf.write(args.output_wav, output.audio, output.sampling_rate)
60
  if args.output_text is not None:
61
  args.output_text.parent.mkdir(parents=True, exist_ok=True)
62
- args.output_text.write_text(output.text.strip() + "\n", encoding="utf-8")
63
  print(output.text)
64
  print(f"[audio] {args.output_wav} ({output.sampling_rate} Hz)")
65
 
 
1
  #!/usr/bin/env python3
2
+ """Run one request through the released Myna-Hokkien model."""
3
  from __future__ import annotations
4
 
5
  import argparse
 
14
  def parse_args() -> argparse.Namespace:
15
  parser = argparse.ArgumentParser()
16
  parser.add_argument("--model", default="iNLP-Lab/MynaHokkien")
17
+ source = parser.add_mutually_exclusive_group(required=True)
18
+ source.add_argument("--audio", type=Path)
19
+ source.add_argument("--text")
20
+ parser.add_argument("--prompt", help="audio-user-turn override")
21
  parser.add_argument("--output-wav", type=Path, default=Path("output.wav"))
22
  parser.add_argument("--output-text", type=Path)
23
  parser.add_argument("--device", default="cuda:0")
24
+ parser.add_argument("--dtype", choices=("float16", "bfloat16"), default="float16")
25
  parser.add_argument("--seed", type=int, default=1234)
26
+ parser.add_argument("--max-new-tokens", type=int, default=256)
27
  parser.add_argument("--max-audio-tokens", type=int, default=4096)
28
+ parser.add_argument("--revision")
29
  parser.add_argument("--local-files-only", action="store_true")
30
  return parser.parse_args()
31
 
 
33
  def main() -> None:
34
  args = parse_args()
35
  if args.prompt is not None and args.audio is None:
36
+ raise ValueError("--prompt is only valid with --audio")
37
+ dtype = torch.float16 if args.dtype == "float16" else torch.bfloat16
 
 
38
  model = MynaHokkien.from_pretrained(
39
  args.model,
40
  device_map=args.device,
41
+ dtype=dtype,
42
+ revision=args.revision,
43
  local_files_only=args.local_files_only,
44
  )
45
  output = model.generate(
 
47
  text=args.text,
48
  prompt=args.prompt,
49
  language="nan",
50
+ speaker="Ethan",
51
  seed=args.seed,
52
  max_new_tokens=args.max_new_tokens,
53
  max_audio_tokens=args.max_audio_tokens,
 
 
54
  )
55
+ if output.text is None or output.audio is None:
56
+ raise RuntimeError("expected both text and audio")
57
 
58
  args.output_wav.parent.mkdir(parents=True, exist_ok=True)
59
  sf.write(args.output_wav, output.audio, output.sampling_rate)
60
  if args.output_text is not None:
61
  args.output_text.parent.mkdir(parents=True, exist_ok=True)
62
+ args.output_text.write_text(output.text + "\n", encoding="utf-8")
63
  print(output.text)
64
  print(f"[audio] {args.output_wav} ({output.sampling_rate} Hz)")
65
 
model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
modeling_mynahokkien.py DELETED
@@ -1,115 +0,0 @@
1
- from __future__ import annotations
2
-
3
- import inspect
4
- from pathlib import Path
5
-
6
- import torch
7
- from torch import nn
8
- from transformers import PreTrainedModel
9
-
10
- from configuration_mynahokkien import MynaHokkienConfig
11
- from mynahokkien.models.bridge1_decoder import Bridge1Decoder
12
-
13
-
14
- class MynaHokkienForConditionalGeneration(PreTrainedModel):
15
- """The single module whose state dict is stored in the unified checkpoint."""
16
-
17
- config_class = MynaHokkienConfig
18
- base_model_prefix = "mynahokkien"
19
- main_input_name = "input_ids"
20
- _supports_assign_param_buffer = True
21
-
22
- def __init__(self, config: MynaHokkienConfig):
23
- super().__init__(config)
24
- from peft import LoraConfig, get_peft_model
25
- from transformers.models.qwen3_omni_moe.configuration_qwen3_omni_moe import Qwen3OmniMoeCode2WavConfig
26
- from transformers import (
27
- Qwen2_5OmniThinkerForConditionalGeneration,
28
- Qwen3OmniMoeCode2Wav,
29
- Qwen3OmniMoeTalkerForConditionalGeneration,
30
- )
31
-
32
- reasoning_config = (
33
- Qwen2_5OmniThinkerForConditionalGeneration.config_class.from_dict(
34
- config.reasoning_model_config
35
- )
36
- )
37
- reasoning_base = Qwen2_5OmniThinkerForConditionalGeneration(reasoning_config)
38
-
39
- # PEFT config files can be produced by a newer PEFT release than the
40
- # runtime. Keep only constructor parameters supported by the pinned
41
- # runtime while preserving the actual LoRA topology and scaling.
42
- adapter_values = dict(config.reasoning_adapter_config)
43
- allowed = set(inspect.signature(LoraConfig.__init__).parameters)
44
- adapter_values = {
45
- key: value for key, value in adapter_values.items()
46
- if key in allowed and key != "self"
47
- }
48
- adapter_values["inference_mode"] = True
49
- self.reasoning_model = get_peft_model(
50
- reasoning_base,
51
- LoraConfig(**adapter_values),
52
- )
53
-
54
- speech_config = Qwen3OmniMoeTalkerForConditionalGeneration.config_class.from_dict(
55
- config.speech_generator_config
56
- )
57
- waveform_config = Qwen3OmniMoeCode2WavConfig.from_dict(
58
- config.waveform_decoder_config
59
- )
60
- self.speech_generator = Qwen3OmniMoeTalkerForConditionalGeneration(speech_config)
61
- self.waveform_decoder = Qwen3OmniMoeCode2Wav(waveform_config)
62
-
63
- embedding = config.generation_embedding_config
64
- self.generation_embedding = nn.Embedding(
65
- int(embedding["num_embeddings"]),
66
- int(embedding["embedding_dim"]),
67
- )
68
- bridge = config.bridge_config
69
- self.bridge = Bridge1Decoder(
70
- q3_vocab=int(embedding["num_embeddings"]),
71
- q3_emb_dim=int(embedding["embedding_dim"]),
72
- q25_dim=int(bridge["q25_dim"]),
73
- inner=int(bridge["inner"]),
74
- heads=int(bridge["heads"]),
75
- layers=int(bridge["layers"]),
76
- ffn=int(bridge["ffn"]),
77
- )
78
-
79
- @classmethod
80
- def load_unified(
81
- cls,
82
- root: Path,
83
- *,
84
- device: str,
85
- dtype: torch.dtype,
86
- ) -> "MynaHokkienForConditionalGeneration":
87
- """Low-memory load of the root sharded checkpoint onto one device."""
88
- from accelerate import init_empty_weights, load_checkpoint_and_dispatch
89
-
90
- config = MynaHokkienConfig.from_pretrained(root, local_files_only=True)
91
- with init_empty_weights():
92
- model = cls(config)
93
- model = load_checkpoint_and_dispatch(
94
- model,
95
- checkpoint=str(root),
96
- device_map={"": device},
97
- dtype=None,
98
- strict=True,
99
- )
100
-
101
- # Match the original release runtime: foundation components use the
102
- # requested inference dtype, while Bridge1 remains FP32.
103
- model.reasoning_model.to(dtype=dtype)
104
- model.speech_generator.to(dtype=dtype)
105
- model.waveform_decoder.to(dtype=dtype)
106
- model.generation_embedding.to(dtype=dtype)
107
- model.bridge.to(dtype=torch.float32)
108
- model.eval()
109
- return model
110
-
111
- def forward(self, *args, **kwargs):
112
- raise NotImplementedError(
113
- "Use the high-level MynaHokkien.generate(audio=... | text=...) interface."
114
- )
115
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
myna_hokkien.egg-info/PKG-INFO ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Metadata-Version: 2.4
2
+ Name: myna-hokkien
3
+ Version: 0.2.0
4
+ Summary: iNLP-Lab Myna-Hokkien native Qwen3-Omni inference runtime
5
+ Project-URL: Homepage, https://huggingface.co/iNLP-Lab/MynaHokkien
6
+ Project-URL: Repository, https://huggingface.co/iNLP-Lab/MynaHokkien
7
+ Requires-Python: >=3.10
8
+ Requires-Dist: torch>=2.6
9
+ Requires-Dist: transformers==4.57.1
10
+ Requires-Dist: accelerate<2,>=1.10
11
+ Requires-Dist: huggingface_hub<2,>=0.36
12
+ Requires-Dist: safetensors>=0.4
13
+ Requires-Dist: numpy<3,>=1.26
14
+ Requires-Dist: librosa<1,>=0.11
15
+ Requires-Dist: soundfile<1,>=0.13
myna_hokkien.egg-info/SOURCES.txt ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ README.md
2
+ pyproject.toml
3
+ myna_hokkien.egg-info/PKG-INFO
4
+ myna_hokkien.egg-info/SOURCES.txt
5
+ myna_hokkien.egg-info/dependency_links.txt
6
+ myna_hokkien.egg-info/requires.txt
7
+ myna_hokkien.egg-info/top_level.txt
8
+ mynahokkien/__init__.py
9
+ mynahokkien/model.py
myna_hokkien.egg-info/dependency_links.txt ADDED
@@ -0,0 +1 @@
 
 
1
+
myna_hokkien.egg-info/requires.txt ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ torch>=2.6
2
+ transformers==4.57.1
3
+ accelerate<2,>=1.10
4
+ huggingface_hub<2,>=0.36
5
+ safetensors>=0.4
6
+ numpy<3,>=1.26
7
+ librosa<1,>=0.11
8
+ soundfile<1,>=0.13
myna_hokkien.egg-info/top_level.txt ADDED
@@ -0,0 +1 @@
 
 
1
+ mynahokkien
mynahokkien/__init__.py CHANGED
@@ -1,15 +1,5 @@
1
- __all__ = [
2
- "MynaHokkien",
3
- "MynaHokkienOutput",
4
- ]
5
 
 
6
 
7
- def __getattr__(name):
8
- if name in __all__:
9
- from .models.mynahokkien import MynaHokkien, MynaHokkienOutput
10
-
11
- return {
12
- "MynaHokkien": MynaHokkien,
13
- "MynaHokkienOutput": MynaHokkienOutput,
14
- }[name]
15
- raise AttributeError(name)
 
1
+ """Public inference API for the Myna-Hokkien release."""
 
 
 
2
 
3
+ from .model import MynaHokkien, MynaHokkienOutput
4
 
5
+ __all__ = ["MynaHokkien", "MynaHokkienOutput"]
 
 
 
 
 
 
 
 
mynahokkien/model.py ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Small, stable wrapper around the native Qwen3-Omni release."""
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import dataclass
5
+ from pathlib import Path
6
+ from typing import Optional, Union
7
+
8
+ import librosa
9
+ import numpy as np
10
+ import torch
11
+ from transformers import Qwen3OmniMoeForConditionalGeneration, Qwen3OmniMoeProcessor
12
+
13
+
14
+ SYSTEM_PROMPT = (
15
+ "You are Qwen, a virtual human developed by the Qwen Team, Alibaba Group, "
16
+ "capable of perceiving auditory inputs and generating text and speech."
17
+ )
18
+ DEFAULT_AUDIO_PROMPT = (
19
+ "Listen to the spoken Hokkien and reply naturally in concise Singapore Hokkien. "
20
+ "Always answer in colloquial Singapore Hokkien written in Hanji. Never answer in "
21
+ "Mandarin or English. Do not repeat or transcribe the input; respond to it directly."
22
+ )
23
+ DEFAULT_TEXT_PROMPT = (
24
+ "Read the text and reply naturally in concise Singapore Hokkien. Always answer in "
25
+ "colloquial Singapore Hokkien written in Hanji. Never answer in Mandarin or English. "
26
+ "Do not repeat the input; respond to it directly."
27
+ )
28
+
29
+
30
+ @dataclass
31
+ class MynaHokkienOutput:
32
+ text: Optional[str]
33
+ audio: Optional[np.ndarray]
34
+ sampling_rate: int = 24000
35
+
36
+
37
+ class MynaHokkien:
38
+ """Audio/text question to Singapore-Hokkien text and speech."""
39
+
40
+ input_sampling_rate = 16000
41
+ sampling_rate = 24000
42
+
43
+ def __init__(self, model, processor) -> None:
44
+ self.model = model.eval()
45
+ self.processor = processor
46
+ self.last_rendered_prompt: Optional[str] = None
47
+
48
+ @classmethod
49
+ def from_pretrained(
50
+ cls,
51
+ model_id_or_path: Union[str, Path],
52
+ *,
53
+ device_map: str = "cuda:0",
54
+ dtype: torch.dtype = torch.float16,
55
+ token: Optional[str] = None,
56
+ revision: Optional[str] = None,
57
+ local_files_only: bool = False,
58
+ attn_implementation: str = "sdpa",
59
+ ) -> "MynaHokkien":
60
+ common = {
61
+ "token": token,
62
+ "revision": revision,
63
+ "local_files_only": local_files_only,
64
+ }
65
+ model = Qwen3OmniMoeForConditionalGeneration.from_pretrained(
66
+ model_id_or_path,
67
+ device_map=device_map,
68
+ dtype=dtype,
69
+ attn_implementation=attn_implementation,
70
+ **common,
71
+ )
72
+ processor = Qwen3OmniMoeProcessor.from_pretrained(model_id_or_path, **common)
73
+ return cls(model, processor)
74
+
75
+ @staticmethod
76
+ def _check_request(
77
+ audio: Optional[Union[str, Path]],
78
+ text: Optional[str],
79
+ prompt: Optional[str],
80
+ language: str,
81
+ speaker: str,
82
+ return_text: bool,
83
+ return_audio: bool,
84
+ ) -> None:
85
+ if (audio is None) == (text is None):
86
+ raise ValueError("provide exactly one of audio=... or text=...")
87
+ if prompt is not None and audio is None:
88
+ raise ValueError("prompt= is only valid with audio=...")
89
+ if language != "nan":
90
+ raise ValueError("this release currently supports language='nan' only")
91
+ if speaker.lower() != "ethan":
92
+ raise ValueError("this release currently supports speaker='Ethan' only")
93
+ if not return_text and not return_audio:
94
+ raise ValueError("at least one output modality must be requested")
95
+
96
+ def _conversation(
97
+ self,
98
+ *,
99
+ audio: Optional[Union[str, Path]],
100
+ text: Optional[str],
101
+ prompt: Optional[str],
102
+ ) -> tuple[list[dict], Optional[np.ndarray]]:
103
+ waveform = None
104
+ if audio is not None:
105
+ audio_path = Path(audio).expanduser()
106
+ if not audio_path.is_file():
107
+ raise FileNotFoundError(audio_path)
108
+ waveform = librosa.load(
109
+ str(audio_path), sr=self.input_sampling_rate, mono=True
110
+ )[0].astype(np.float32)
111
+ user_content = [
112
+ {"type": "audio", "audio": waveform},
113
+ {"type": "text", "text": prompt or DEFAULT_AUDIO_PROMPT},
114
+ ]
115
+ else:
116
+ query = text.strip() if text is not None else ""
117
+ if not query:
118
+ raise ValueError("text= must not be empty")
119
+ # This is the exact prefix form used by probe/prompts.json.
120
+ user_content = [
121
+ {"type": "text", "text": f"{DEFAULT_TEXT_PROMPT}\n\n{query}"}
122
+ ]
123
+ conversation = [
124
+ {"role": "system", "content": [{"type": "text", "text": SYSTEM_PROMPT}]},
125
+ {"role": "user", "content": user_content},
126
+ ]
127
+ return conversation, waveform
128
+
129
+ def generate(
130
+ self,
131
+ *,
132
+ audio: Optional[Union[str, Path]] = None,
133
+ text: Optional[str] = None,
134
+ prompt: Optional[str] = None,
135
+ language: str = "nan",
136
+ speaker: str = "Ethan",
137
+ seed: Optional[int] = 1234,
138
+ max_new_tokens: int = 256,
139
+ max_audio_tokens: int = 4096,
140
+ return_text: bool = True,
141
+ return_audio: bool = True,
142
+ ) -> MynaHokkienOutput:
143
+ self._check_request(
144
+ audio, text, prompt, language, speaker, return_text, return_audio
145
+ )
146
+ if seed is not None:
147
+ torch.manual_seed(seed)
148
+ if torch.cuda.is_available():
149
+ torch.cuda.manual_seed_all(seed)
150
+
151
+ conversation, waveform = self._conversation(audio=audio, text=text, prompt=prompt)
152
+ rendered = self.processor.apply_chat_template(
153
+ conversation, add_generation_prompt=True, tokenize=False
154
+ )
155
+ self.last_rendered_prompt = rendered
156
+ processor_kwargs = {
157
+ "text": [rendered],
158
+ "return_tensors": "pt",
159
+ "padding": True,
160
+ "use_audio_in_video": False,
161
+ }
162
+ if waveform is not None:
163
+ processor_kwargs.update(
164
+ {"audio": [waveform], "sampling_rate": self.input_sampling_rate}
165
+ )
166
+ inputs = self.processor(**processor_kwargs)
167
+ inputs = inputs.to(self.model.device).to(self.model.dtype)
168
+ prompt_tokens = inputs["input_ids"].shape[1]
169
+
170
+ generation_kwargs = {
171
+ **inputs,
172
+ "speaker": speaker,
173
+ "return_audio": return_audio,
174
+ "thinker_do_sample": False,
175
+ "thinker_return_dict_in_generate": True,
176
+ "thinker_max_new_tokens": max_new_tokens,
177
+ "talker_do_sample": False,
178
+ "talker_max_new_tokens": max_audio_tokens,
179
+ "use_audio_in_video": False,
180
+ }
181
+ with torch.inference_mode():
182
+ generated = self.model.generate(**generation_kwargs)
183
+
184
+ if return_audio:
185
+ text_result, generated_audio = generated
186
+ else:
187
+ text_result, generated_audio = generated, None
188
+ sequences = (
189
+ text_result.sequences if hasattr(text_result, "sequences") else text_result
190
+ )
191
+ decoded = self.processor.batch_decode(
192
+ sequences[:, prompt_tokens:],
193
+ skip_special_tokens=True,
194
+ clean_up_tokenization_spaces=False,
195
+ )[0].strip()
196
+ if not decoded:
197
+ raise RuntimeError("the Thinker returned an empty answer")
198
+
199
+ audio_array = None
200
+ if generated_audio is not None:
201
+ audio_array = (
202
+ generated_audio.reshape(-1).float().cpu().numpy().astype(np.float32)
203
+ )
204
+ return MynaHokkienOutput(
205
+ text=decoded if return_text else None,
206
+ audio=audio_array,
207
+ sampling_rate=self.sampling_rate,
208
+ )
mynahokkien/models/__init__.py DELETED
@@ -1,15 +0,0 @@
1
- __all__ = [
2
- "MynaHokkien",
3
- "MynaHokkienOutput",
4
- ]
5
-
6
-
7
- def __getattr__(name):
8
- if name in __all__:
9
- from .mynahokkien import MynaHokkien, MynaHokkienOutput
10
-
11
- return {
12
- "MynaHokkien": MynaHokkien,
13
- "MynaHokkienOutput": MynaHokkienOutput,
14
- }[name]
15
- raise AttributeError(name)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
mynahokkien/models/bridge1_decoder.py DELETED
@@ -1,83 +0,0 @@
1
- """Bridge1 runtime used by the standalone MynaHokkien inference package."""
2
- from __future__ import annotations
3
-
4
- import math
5
-
6
- import torch
7
- import torch.nn as nn
8
- import torch.nn.functional as F
9
-
10
-
11
- def sinusoidal_pe(n, dim, device, dtype):
12
- pos = torch.arange(n, device=device, dtype=torch.float32).unsqueeze(1)
13
- div = torch.exp(torch.arange(0, dim, 2, device=device, dtype=torch.float32) * (-math.log(10000.0) / dim))
14
- pe = torch.zeros(n, dim, device=device, dtype=torch.float32)
15
- pe[:, 0::2] = torch.sin(pos * div)
16
- pe[:, 1::2] = torch.cos(pos * div)
17
- return pe.to(dtype)
18
-
19
-
20
- class Bridge1Decoder(nn.Module):
21
- def __init__(self, q3_vocab, q3_emb_dim=2048, q25_dim=2048, inner=1536, heads=12, layers=4, ffn=6144):
22
- super().__init__()
23
- self.in_proj = nn.Linear(q3_emb_dim, inner)
24
- self.mem_proj = nn.Linear(q25_dim, inner)
25
- self.start = nn.Parameter(torch.zeros(1, 1, inner))
26
- nn.init.normal_(self.start, std=0.02)
27
- self.blocks = nn.ModuleList(
28
- [
29
- nn.TransformerDecoderLayer(
30
- d_model=inner,
31
- nhead=heads,
32
- dim_feedforward=ffn,
33
- dropout=0.0,
34
- activation="gelu",
35
- batch_first=True,
36
- norm_first=True,
37
- )
38
- for _ in range(layers)
39
- ]
40
- )
41
- self.out_norm = nn.LayerNorm(inner)
42
- self.out_proj = nn.Linear(inner, q3_emb_dim)
43
- self.length_head = nn.Linear(inner, 1)
44
-
45
- def _embed_tokens(self, ids, q3_emb_weight):
46
- return self.in_proj(F.embedding(ids, q3_emb_weight).to(self.in_proj.weight.dtype))
47
-
48
- def _logits(self, hidden, q3_emb_weight):
49
- projected = self.out_proj(self.out_norm(hidden))
50
- return F.linear(projected, q3_emb_weight.to(projected.dtype))
51
-
52
- def _run(self, target, memory):
53
- target = target + sinusoidal_pe(
54
- target.shape[1], target.shape[-1], target.device, target.dtype
55
- ).unsqueeze(0)
56
- memory = memory + sinusoidal_pe(
57
- memory.shape[1], memory.shape[-1], memory.device, memory.dtype
58
- ).unsqueeze(0)
59
- causal = torch.triu(
60
- torch.full((target.shape[1], target.shape[1]), float("-inf"), device=target.device),
61
- diagonal=1,
62
- )
63
- hidden = target
64
- for block in self.blocks:
65
- hidden = block(tgt=hidden, memory=memory, tgt_mask=causal)
66
- return hidden
67
-
68
- @torch.no_grad()
69
- def generate(self, q25_memory, q3_emb_weight, eos_id, max_len=512):
70
- memory = self.mem_proj(q25_memory)
71
- emitted = []
72
- current = self.start.to(memory.dtype)
73
- for _ in range(max_len):
74
- hidden = self._run(current, memory)
75
- token = int(self._logits(hidden[:, -1:], q3_emb_weight)[0, -1].argmax().item())
76
- if token == eos_id:
77
- break
78
- emitted.append(token)
79
- next_embed = self._embed_tokens(
80
- torch.tensor([[token]], device=memory.device), q3_emb_weight
81
- )
82
- current = torch.cat([current, next_embed], dim=1)
83
- return torch.tensor(emitted, dtype=torch.long, device=memory.device)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
mynahokkien/models/mynahokkien.py DELETED
@@ -1,314 +0,0 @@
1
- from __future__ import annotations
2
-
3
- from dataclasses import dataclass
4
- from pathlib import Path
5
- from typing import Optional, Union
6
-
7
- import numpy as np
8
- import torch
9
-
10
- from modeling_mynahokkien import MynaHokkienForConditionalGeneration
11
- from processing_mynahokkien import MynaHokkienProcessor
12
-
13
-
14
- @dataclass
15
- class MynaHokkienOutput:
16
- text: Optional[str] = None
17
- audio: Optional[np.ndarray] = None
18
- sampling_rate: int = 24000
19
-
20
-
21
- class MynaHokkien:
22
- """Unified inference interface for the implicit Hokkien S2S model."""
23
-
24
- sampling_rate = 24000
25
- input_sampling_rate = 16000
26
-
27
- def __init__(self):
28
- self.last_text: Optional[str] = None
29
-
30
- @staticmethod
31
- def _resolve_model_path(model_id_or_path: Union[str, Path], **hub_kwargs) -> Path:
32
- local = Path(model_id_or_path).expanduser()
33
- if local.is_dir():
34
- return local.resolve()
35
- from huggingface_hub import snapshot_download
36
-
37
- return Path(snapshot_download(repo_id=str(model_id_or_path), **hub_kwargs))
38
-
39
- @classmethod
40
- def from_pretrained(
41
- cls,
42
- model_id_or_path: Union[str, Path],
43
- *,
44
- device_map: str = "cuda:0",
45
- dtype: torch.dtype = torch.float16,
46
- token: Optional[str] = None,
47
- revision: Optional[str] = None,
48
- local_files_only: bool = False,
49
- ) -> "MynaHokkien":
50
- root = cls._resolve_model_path(
51
- model_id_or_path,
52
- token=token,
53
- revision=revision,
54
- local_files_only=local_files_only,
55
- )
56
- self = cls()
57
- self.root = root
58
- self.device = torch.device(device_map)
59
- self.dtype = dtype
60
- self.core = MynaHokkienForConditionalGeneration.load_unified(
61
- root,
62
- device=device_map,
63
- dtype=dtype,
64
- )
65
- self.sampling_rate = int(self.core.config.sampling_rate)
66
- self.input_sampling_rate = int(self.core.config.input_sampling_rate)
67
- processors = MynaHokkienProcessor.from_pretrained(root)
68
- self.q25_processor = processors.reasoning_processor
69
- self.q3_processor = processors.generation_processor
70
- from transformers import Qwen3OmniMoeConfig
71
-
72
- q3_processor_path = root / self.core.config.generation_processor_path
73
- self.q3_omni_config = Qwen3OmniMoeConfig.from_pretrained(
74
- q3_processor_path,
75
- local_files_only=True,
76
- )
77
-
78
- self.q25 = self.core.reasoning_model
79
- self.talker = self.core.speech_generator
80
- self.code2wav = self.core.waveform_decoder
81
- self.q3_embedding = self.core.generation_embedding
82
- self.bridge = self.core.bridge
83
- self.q3_config = self.q3_processor_config = self.q3_omni_config.to_dict()
84
-
85
- self.q25.eval()
86
- self.talker.eval()
87
- self.code2wav.eval()
88
- self.bridge.eval()
89
- return self
90
-
91
- def _q25_answer_memory(
92
- self,
93
- text: Optional[str],
94
- audio: Optional[Union[str, Path]],
95
- prompt: Optional[str],
96
- max_new_tokens: int,
97
- ):
98
- system = (
99
- "You are a helpful Singapore Hokkien speaker. Always reply in colloquial Singapore "
100
- "Hokkien written in Hanji. Keep it short."
101
- )
102
- sg_prompt = (
103
- "Answer or respond to what is said in the audio, in spoken Singapore Hokkien. "
104
- "Do NOT repeat or transcribe it."
105
- )
106
- wav = None
107
- if audio is not None:
108
- import librosa
109
-
110
- wav = librosa.load(str(audio), sr=self.input_sampling_rate, mono=True)[0].astype(np.float32)
111
- content = [{"type": "audio", "audio": wav}, {"type": "text", "text": prompt or sg_prompt}]
112
- elif text:
113
- content = [{"type": "text", "text": text}]
114
- else:
115
- raise ValueError("provide exactly one of audio=... or text=...")
116
-
117
- conversation = [
118
- {"role": "system", "content": [{"type": "text", "text": system}]},
119
- {"role": "user", "content": content},
120
- ]
121
- rendered = self.q25_processor.apply_chat_template(
122
- conversation, add_generation_prompt=True, tokenize=False
123
- )
124
- if wav is None:
125
- inputs = self.q25_processor(text=[rendered], return_tensors="pt", padding=True)
126
- else:
127
- inputs = self.q25_processor(
128
- text=[rendered],
129
- audio=[wav],
130
- sampling_rate=self.input_sampling_rate,
131
- return_tensors="pt",
132
- padding=True,
133
- )
134
- inputs = {
135
- key: value.to(self.device) if torch.is_tensor(value) else value
136
- for key, value in inputs.items()
137
- }
138
- prompt_len = inputs["input_ids"].shape[1]
139
- with torch.no_grad():
140
- generated = self.q25.generate(
141
- **inputs, do_sample=False, max_new_tokens=max_new_tokens
142
- )
143
- answer_ids = generated[0, prompt_len:].tolist()
144
- eos = self.q25_processor.tokenizer.eos_token_id
145
- if eos in answer_ids:
146
- answer_ids = answer_ids[: answer_ids.index(eos)]
147
- if not answer_ids:
148
- raise RuntimeError("Q2.5 thinker returned an empty answer")
149
- self.last_text = self.q25_processor.tokenizer.decode(
150
- answer_ids, skip_special_tokens=True
151
- ).strip()
152
- answer = torch.tensor([answer_ids], device=self.device)
153
- embedding = self.q25.get_input_embeddings()
154
- if hasattr(embedding, "base_layer"):
155
- embedding = embedding.base_layer
156
- memory = embedding(answer).to(dtype=torch.float32)
157
- return memory
158
-
159
- def _assistant_part(self, start, end, speaker_id, hidden, pad, bos, eos):
160
- cfg = self.q3_config
161
- talker_cfg = cfg["talker_config"]
162
- assistant_hidden = self.talker.text_projection(hidden[:, start:end]).to(self.device)
163
- text_hidden = torch.cat(
164
- (assistant_hidden[:, :3], pad.expand(-1, 4, -1), bos, assistant_hidden[:, 3:4]),
165
- dim=1,
166
- )
167
- special_ids = torch.tensor(
168
- [[
169
- talker_cfg["codec_nothink_id"],
170
- talker_cfg["codec_think_bos_id"],
171
- talker_cfg["codec_think_eos_id"],
172
- speaker_id,
173
- talker_cfg["codec_pad_id"],
174
- talker_cfg["codec_bos_id"],
175
- ]],
176
- device=self.device,
177
- )
178
- codec_hidden = torch.cat(
179
- (
180
- torch.zeros((1, 3, text_hidden.shape[-1]), device=self.device, dtype=self.dtype),
181
- self.talker.get_input_embeddings()(special_ids).to(self.device),
182
- ),
183
- dim=1,
184
- )
185
- trailing = torch.cat((assistant_hidden[:, 4:], eos), dim=1)
186
- input_ids = torch.full(
187
- (1, text_hidden.shape[1]),
188
- fill_value=cfg["tts_pad_token_id"],
189
- dtype=torch.long,
190
- device=self.device,
191
- )
192
- return text_hidden + codec_hidden, input_ids, trailing
193
-
194
- def _speak(self, predicted_ids: list[int], speaker: str, max_audio_tokens: int) -> np.ndarray:
195
- cfg = self.q3_config
196
- tok = self.q3_processor.tokenizer
197
- tts_system = (
198
- "You are a high-quality Text-to-Speech (TTS) model. Your task is to convert text "
199
- "into natural, fluent, and realistic speech."
200
- )
201
- tts_user = "Read this Hokkien text aloud naturally and clearly."
202
- head = self.q3_processor.apply_chat_template(
203
- [{"role": "system", "content": tts_system}, {"role": "user", "content": tts_user}],
204
- add_generation_prompt=True,
205
- tokenize=False,
206
- )
207
- head_ids = tok(head, add_special_tokens=False)["input_ids"]
208
- full_ids = torch.tensor(
209
- [head_ids + predicted_ids + [cfg["im_end_token_id"]]],
210
- device=self.device,
211
- )
212
- hidden = self.q3_embedding(full_ids).to(dtype=self.dtype)
213
- special = torch.tensor(
214
- [[cfg["tts_bos_token_id"], cfg["tts_eos_token_id"], cfg["tts_pad_token_id"]]],
215
- device=self.device,
216
- )
217
- bos, eos, pad = self.talker.text_projection(self.q3_embedding(special)).chunk(3, dim=1)
218
- speaker_id = cfg["talker_config"]["speaker_id"].get(speaker.lower())
219
- if speaker_id is None:
220
- raise ValueError(f"unknown speaker {speaker!r}")
221
-
222
- starts = torch.nonzero(full_ids[0] == cfg["im_start_token_id"]).view(-1)
223
- ends = torch.cat([starts, torch.tensor([full_ids.shape[1]], device=self.device)])
224
- parts, id_parts, trailing = [], [], None
225
- for i in range(len(ends) - 1):
226
- start, end = int(ends[i]), int(ends[i + 1])
227
- role = int(full_ids[0, start + 1])
228
- if role == cfg["user_token_id"]:
229
- parts.append(self.talker.text_projection(hidden[:, start:end]))
230
- id_parts.append(full_ids[:, start:end])
231
- elif role == cfg["assistant_token_id"] and i == len(ends) - 2:
232
- part, ids, trailing = self._assistant_part(
233
- start, end, speaker_id, hidden, pad, bos, eos
234
- )
235
- parts.append(part)
236
- id_parts.append(ids)
237
- if trailing is None:
238
- raise RuntimeError("failed to construct Q3 assistant Talker prefix")
239
-
240
- talker_cfg = cfg["talker_config"]
241
- vocab = talker_cfg["text_config"]["vocab_size"]
242
- codec_eos = talker_cfg["codec_eos_token_id"]
243
- suppress = [i for i in range(vocab - 1024, vocab) if i != codec_eos]
244
- with torch.no_grad():
245
- result = self.talker.generate(
246
- inputs_embeds=torch.cat(parts, dim=1),
247
- trailing_text_hidden=trailing,
248
- tts_pad_embed=pad,
249
- talker_input_ids=torch.cat(id_parts, dim=1),
250
- max_new_tokens=max_audio_tokens,
251
- do_sample=True,
252
- top_k=50,
253
- top_p=1.0,
254
- temperature=0.9,
255
- repetition_penalty=1.05,
256
- eos_token_id=codec_eos,
257
- suppress_tokens=suppress,
258
- output_hidden_states=True,
259
- return_dict_in_generate=True,
260
- )
261
- codes = torch.stack(
262
- [states[-1] for states in result.hidden_states if states[-1] is not None],
263
- dim=1,
264
- ).transpose(1, 2)
265
- wav = self.code2wav.chunked_decode(
266
- codes.to(next(self.code2wav.parameters()).device),
267
- chunk_size=300,
268
- left_context_size=25,
269
- )
270
- return wav.float().squeeze().cpu().numpy().astype(np.float32)
271
-
272
- def generate(
273
- self,
274
- *,
275
- audio: Optional[Union[str, Path]] = None,
276
- text: Optional[str] = None,
277
- language: str = "nan",
278
- speaker: str = "Ethan",
279
- prompt: Optional[str] = None,
280
- max_new_tokens: int = 512,
281
- max_audio_tokens: int = 4096,
282
- seed: Optional[int] = None,
283
- return_text: bool = True,
284
- return_audio: bool = True,
285
- ) -> MynaHokkienOutput:
286
- if language != "nan":
287
- raise ValueError("this release currently supports language='nan' only")
288
- if audio is not None and text is not None:
289
- raise ValueError("audio and text are mutually exclusive")
290
- if prompt is not None and audio is None:
291
- raise ValueError("prompt= is only used with audio=; include instructions directly in text=")
292
- if not return_text and not return_audio:
293
- raise ValueError("at least one of return_text or return_audio must be True")
294
- if seed is not None:
295
- torch.manual_seed(seed)
296
- if torch.cuda.is_available():
297
- torch.cuda.manual_seed_all(seed)
298
- memory = self._q25_answer_memory(text, audio, prompt, max_new_tokens)
299
- reasoning_text = self.last_text
300
- q3_tok = self.q3_processor.tokenizer
301
- predicted = self.bridge.generate(
302
- memory,
303
- self.q3_embedding.weight,
304
- eos_id=q3_tok.eos_token_id,
305
- max_len=max_new_tokens,
306
- ).tolist()
307
- if not predicted:
308
- raise RuntimeError("Bridge1 returned an empty Q3 token sequence")
309
- waveform = self._speak(predicted, speaker, max_audio_tokens) if return_audio else None
310
- return MynaHokkienOutput(
311
- text=reasoning_text if return_text else None,
312
- audio=waveform,
313
- sampling_rate=self.sampling_rate,
314
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/generation/preprocessor_config.json → preprocessor_config.json RENAMED
File without changes
processing_mynahokkien.py DELETED
@@ -1,30 +0,0 @@
1
- from __future__ import annotations
2
-
3
- from pathlib import Path
4
-
5
- from configuration_mynahokkien import MynaHokkienConfig
6
-
7
-
8
- class MynaHokkienProcessor:
9
- """Private two-processor implementation behind the unified public API."""
10
-
11
- def __init__(self, reasoning_processor, generation_processor):
12
- self.reasoning_processor = reasoning_processor
13
- self.generation_processor = generation_processor
14
-
15
- @classmethod
16
- def from_pretrained(cls, root: str | Path):
17
- from transformers import Qwen2_5OmniProcessor, Qwen3OmniMoeProcessor
18
-
19
- root = Path(root)
20
- config = MynaHokkienConfig.from_pretrained(root, local_files_only=True)
21
- reasoning = Qwen2_5OmniProcessor.from_pretrained(
22
- root / config.reasoning_processor_path,
23
- local_files_only=True,
24
- )
25
- generation = Qwen3OmniMoeProcessor.from_pretrained(
26
- root / config.generation_processor_path,
27
- local_files_only=True,
28
- )
29
- return cls(reasoning, generation)
30
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processor_config.json DELETED
@@ -1,5 +0,0 @@
1
- {
2
- "generation_processor_path": "processors/generation",
3
- "processor_class": "MynaHokkienProcessor",
4
- "reasoning_processor_path": "processors/reasoning"
5
- }
 
 
 
 
 
 
processors/generation/config.json DELETED
@@ -1,589 +0,0 @@
1
- {
2
- "architectures": [
3
- "Qwen3OmniMoeForConditionalGeneration"
4
- ],
5
- "assistant_token_id": 77091,
6
- "code2wav_config": {
7
- "attention_bias": false,
8
- "attention_dropout": 0.0,
9
- "codebook_dim": 512,
10
- "codebook_size": 2048,
11
- "decoder_dim": 1536,
12
- "dtype": "bfloat16",
13
- "hidden_act": "silu",
14
- "hidden_size": 1024,
15
- "initializer_range": 0.02,
16
- "intermediate_size": 3072,
17
- "layer_scale_initial_scale": 0.01,
18
- "max_position_embeddings": 8000,
19
- "model_type": "",
20
- "num_attention_heads": 16,
21
- "num_hidden_layers": 8,
22
- "num_key_value_heads": 16,
23
- "num_quantizers": 16,
24
- "num_semantic_quantizers": 1,
25
- "rms_norm_eps": 1e-05,
26
- "rope_parameters": {
27
- "rope_theta": 10000,
28
- "rope_type": "default"
29
- },
30
- "rope_theta": 10000,
31
- "semantic_codebook_size": 4096,
32
- "sliding_window": 72,
33
- "upsample_rates": [
34
- 8,
35
- 5,
36
- 4,
37
- 3
38
- ],
39
- "upsampling_ratios": [
40
- 2,
41
- 2
42
- ],
43
- "vector_quantization_hidden_dimension": 512
44
- },
45
- "dtype": "bfloat16",
46
- "enable_audio_output": true,
47
- "im_end_token_id": 151645,
48
- "im_start_token_id": 151644,
49
- "initializer_range": 0.02,
50
- "model_type": "qwen3_omni_moe",
51
- "system_token_id": 8948,
52
- "talker_config": {
53
- "accept_hidden_layer": 24,
54
- "audio_end_token_id": 151670,
55
- "audio_start_token_id": 151669,
56
- "audio_token_id": 151675,
57
- "audio_tower_config": {},
58
- "code_predictor_config": {
59
- "_name_or_path": "",
60
- "add_cross_attention": false,
61
- "architectures": null,
62
- "attention_bias": false,
63
- "attention_dropout": 0,
64
- "bad_words_ids": null,
65
- "begin_suppress_tokens": null,
66
- "bos_token_id": null,
67
- "chunk_size_feed_forward": 0,
68
- "cross_attention_hidden_size": null,
69
- "decoder_start_token_id": null,
70
- "diversity_penalty": 0.0,
71
- "do_sample": false,
72
- "dtype": "bfloat16",
73
- "early_stopping": false,
74
- "encoder_no_repeat_ngram_size": 0,
75
- "eos_token_id": null,
76
- "exponential_decay_length_penalty": null,
77
- "finetuning_task": null,
78
- "forced_bos_token_id": null,
79
- "forced_eos_token_id": null,
80
- "head_dim": 128,
81
- "hidden_act": "silu",
82
- "hidden_size": 1024,
83
- "id2label": {
84
- "0": "LABEL_0",
85
- "1": "LABEL_1"
86
- },
87
- "initializer_range": 0.02,
88
- "intermediate_size": 3072,
89
- "is_decoder": false,
90
- "is_encoder_decoder": false,
91
- "label2id": {
92
- "LABEL_0": 0,
93
- "LABEL_1": 1
94
- },
95
- "layer_types": [
96
- "full_attention",
97
- "full_attention",
98
- "full_attention",
99
- "full_attention",
100
- "full_attention"
101
- ],
102
- "length_penalty": 1.0,
103
- "max_length": 20,
104
- "max_position_embeddings": 32768,
105
- "max_window_layers": 28,
106
- "min_length": 0,
107
- "model_type": "qwen3_omni_moe_talker_code_predictor",
108
- "no_repeat_ngram_size": 0,
109
- "num_attention_heads": 16,
110
- "num_beam_groups": 1,
111
- "num_beams": 1,
112
- "num_code_groups": 16,
113
- "num_hidden_layers": 5,
114
- "num_key_value_heads": 8,
115
- "num_return_sequences": 1,
116
- "output_attentions": false,
117
- "output_hidden_states": false,
118
- "output_scores": false,
119
- "pad_token_id": null,
120
- "prefix": null,
121
- "problem_type": null,
122
- "pruned_heads": {},
123
- "remove_invalid_values": false,
124
- "repetition_penalty": 1.0,
125
- "return_dict": true,
126
- "return_dict_in_generate": false,
127
- "rms_norm_eps": 1e-06,
128
- "rope_parameters": {
129
- "rope_theta": 1000000,
130
- "rope_type": "default"
131
- },
132
- "rope_scaling": null,
133
- "rope_theta": 10000,
134
- "sep_token_id": null,
135
- "sliding_window": null,
136
- "suppress_tokens": null,
137
- "task_specific_params": null,
138
- "temperature": 1.0,
139
- "tf_legacy_loss": false,
140
- "tie_encoder_decoder": false,
141
- "tie_word_embeddings": false,
142
- "tokenizer_class": null,
143
- "top_k": 50,
144
- "top_p": 1.0,
145
- "torchscript": false,
146
- "typical_p": 1.0,
147
- "use_bfloat16": false,
148
- "use_cache": true,
149
- "use_sliding_window": false,
150
- "vocab_size": 2048
151
- },
152
- "codec_bos_id": 2149,
153
- "codec_config": {},
154
- "codec_eos_token_id": 2150,
155
- "codec_nothink_id": 2155,
156
- "codec_pad_id": 2148,
157
- "codec_think_bos_id": 2156,
158
- "codec_think_eos_id": 2157,
159
- "dtype": "bfloat16",
160
- "image_token_id": 151655,
161
- "initializer_range": 0.02,
162
- "model_type": "",
163
- "num_code_groups": 16,
164
- "output_router_logits": false,
165
- "position_id_per_seconds": 13,
166
- "seconds_per_chunk": 2,
167
- "spatial_merge_size": 2,
168
- "speaker_id": {
169
- "aiden": 2303,
170
- "chelsie": 2301,
171
- "ethan": 2302
172
- },
173
- "text_config": {
174
- "_name_or_path": "",
175
- "add_cross_attention": false,
176
- "architectures": null,
177
- "attention_bias": false,
178
- "attention_dropout": 0,
179
- "bad_words_ids": null,
180
- "begin_suppress_tokens": null,
181
- "bos_token_id": null,
182
- "chunk_size_feed_forward": 0,
183
- "cross_attention_hidden_size": null,
184
- "decoder_sparse_step": 1,
185
- "decoder_start_token_id": null,
186
- "diversity_penalty": 0.0,
187
- "do_sample": false,
188
- "dtype": "bfloat16",
189
- "early_stopping": false,
190
- "encoder_no_repeat_ngram_size": 0,
191
- "eos_token_id": null,
192
- "exponential_decay_length_penalty": null,
193
- "finetuning_task": null,
194
- "forced_bos_token_id": null,
195
- "forced_eos_token_id": null,
196
- "head_dim": 128,
197
- "hidden_act": "silu",
198
- "hidden_size": 1024,
199
- "id2label": {
200
- "0": "LABEL_0",
201
- "1": "LABEL_1"
202
- },
203
- "initializer_range": 0.02,
204
- "intermediate_size": 2048,
205
- "is_decoder": false,
206
- "is_encoder_decoder": false,
207
- "label2id": {
208
- "LABEL_0": 0,
209
- "LABEL_1": 1
210
- },
211
- "length_penalty": 1.0,
212
- "max_length": 20,
213
- "max_position_embeddings": 65536,
214
- "min_length": 0,
215
- "mlp_only_layers": [],
216
- "model_type": "qwen3_omni_moe_talker_text",
217
- "moe_intermediate_size": 384,
218
- "no_repeat_ngram_size": 0,
219
- "norm_topk_prob": true,
220
- "num_attention_heads": 16,
221
- "num_beam_groups": 1,
222
- "num_beams": 1,
223
- "num_experts": 128,
224
- "num_experts_per_tok": 6,
225
- "num_hidden_layers": 20,
226
- "num_key_value_heads": 2,
227
- "num_local_experts": 128,
228
- "num_return_sequences": 1,
229
- "output_attentions": false,
230
- "output_hidden_states": false,
231
- "output_router_logits": false,
232
- "output_scores": false,
233
- "pad_token_id": null,
234
- "prefix": null,
235
- "problem_type": null,
236
- "pruned_heads": {},
237
- "remove_invalid_values": false,
238
- "repetition_penalty": 1.0,
239
- "return_dict": true,
240
- "return_dict_in_generate": false,
241
- "rms_norm_eps": 1e-06,
242
- "rope_parameters": {
243
- "interleaved": true,
244
- "mrope_section": [
245
- 24,
246
- 20,
247
- 20
248
- ],
249
- "rope_theta": 1000000,
250
- "rope_type": "default",
251
- "type": "default"
252
- },
253
- "rope_scaling": {
254
- "interleaved": true,
255
- "mrope_section": [
256
- 24,
257
- 20,
258
- 20
259
- ],
260
- "rope_type": "default",
261
- "type": "default"
262
- },
263
- "rope_theta": 10000,
264
- "router_aux_loss_coef": 0.001,
265
- "sep_token_id": null,
266
- "shared_expert_intermediate_size": 768,
267
- "sliding_window": null,
268
- "suppress_tokens": null,
269
- "task_specific_params": null,
270
- "temperature": 1.0,
271
- "tf_legacy_loss": false,
272
- "tie_encoder_decoder": false,
273
- "tie_word_embeddings": false,
274
- "tokenizer_class": null,
275
- "top_k": 50,
276
- "top_p": 1.0,
277
- "torchscript": false,
278
- "typical_p": 1.0,
279
- "use_bfloat16": false,
280
- "use_cache": true,
281
- "use_sliding_window": false,
282
- "vocab_size": 3072
283
- },
284
- "thinker_hidden_size": 2048,
285
- "tie_word_embeddings": false,
286
- "video_token_id": 151656,
287
- "vision_start_token_id": 151652
288
- },
289
- "thinker_config": {
290
- "audio_config": {
291
- "_name_or_path": "",
292
- "activation_dropout": 0,
293
- "activation_function": "gelu",
294
- "add_cross_attention": false,
295
- "architectures": null,
296
- "attention_dropout": 0,
297
- "bad_words_ids": null,
298
- "begin_suppress_tokens": null,
299
- "bos_token_id": null,
300
- "chunk_size_feed_forward": 0,
301
- "conv_chunksize": 500,
302
- "cross_attention_hidden_size": null,
303
- "d_model": 1280,
304
- "decoder_start_token_id": null,
305
- "diversity_penalty": 0.0,
306
- "do_sample": false,
307
- "downsample_hidden_size": 480,
308
- "dropout": 0,
309
- "dtype": "bfloat16",
310
- "early_stopping": false,
311
- "encoder_attention_heads": 20,
312
- "encoder_ffn_dim": 5120,
313
- "encoder_layers": 32,
314
- "encoder_no_repeat_ngram_size": 0,
315
- "eos_token_id": null,
316
- "exponential_decay_length_penalty": null,
317
- "finetuning_task": null,
318
- "forced_bos_token_id": null,
319
- "forced_eos_token_id": null,
320
- "id2label": {
321
- "0": "LABEL_0",
322
- "1": "LABEL_1"
323
- },
324
- "initializer_range": 0.02,
325
- "is_decoder": false,
326
- "is_encoder_decoder": false,
327
- "label2id": {
328
- "LABEL_0": 0,
329
- "LABEL_1": 1
330
- },
331
- "length_penalty": 1.0,
332
- "max_length": 20,
333
- "max_source_positions": 1500,
334
- "min_length": 0,
335
- "model_type": "qwen3_omni_moe_audio_encoder",
336
- "n_window": 50,
337
- "n_window_infer": 800,
338
- "no_repeat_ngram_size": 0,
339
- "num_beam_groups": 1,
340
- "num_beams": 1,
341
- "num_hidden_layers": 32,
342
- "num_mel_bins": 128,
343
- "num_return_sequences": 1,
344
- "output_attentions": false,
345
- "output_dim": 2048,
346
- "output_hidden_states": false,
347
- "output_scores": false,
348
- "pad_token_id": null,
349
- "prefix": null,
350
- "problem_type": null,
351
- "pruned_heads": {},
352
- "remove_invalid_values": false,
353
- "repetition_penalty": 1.0,
354
- "return_dict": true,
355
- "return_dict_in_generate": false,
356
- "scale_embedding": false,
357
- "sep_token_id": null,
358
- "suppress_tokens": null,
359
- "task_specific_params": null,
360
- "temperature": 1.0,
361
- "tf_legacy_loss": false,
362
- "tie_encoder_decoder": false,
363
- "tie_word_embeddings": true,
364
- "tokenizer_class": null,
365
- "top_k": 50,
366
- "top_p": 1.0,
367
- "torchscript": false,
368
- "typical_p": 1.0,
369
- "use_bfloat16": false
370
- },
371
- "audio_end_token_id": 151670,
372
- "audio_start_token_id": 151669,
373
- "audio_token_id": 151675,
374
- "dtype": "bfloat16",
375
- "image_token_id": 151655,
376
- "initializer_range": 0.02,
377
- "model_type": "qwen3_omni_moe_thinker",
378
- "position_id_per_seconds": 13,
379
- "seconds_per_chunk": 2,
380
- "text_config": {
381
- "_name_or_path": "",
382
- "add_cross_attention": false,
383
- "architectures": null,
384
- "attention_bias": false,
385
- "attention_dropout": 0.0,
386
- "bad_words_ids": null,
387
- "begin_suppress_tokens": null,
388
- "bos_token_id": null,
389
- "chunk_size_feed_forward": 0,
390
- "cross_attention_hidden_size": null,
391
- "decoder_sparse_step": 1,
392
- "decoder_start_token_id": null,
393
- "diversity_penalty": 0.0,
394
- "do_sample": false,
395
- "dtype": "bfloat16",
396
- "early_stopping": false,
397
- "encoder_no_repeat_ngram_size": 0,
398
- "eos_token_id": null,
399
- "exponential_decay_length_penalty": null,
400
- "finetuning_task": null,
401
- "forced_bos_token_id": null,
402
- "forced_eos_token_id": null,
403
- "head_dim": 128,
404
- "hidden_act": "silu",
405
- "hidden_size": 2048,
406
- "id2label": {
407
- "0": "LABEL_0",
408
- "1": "LABEL_1"
409
- },
410
- "initializer_range": 0.02,
411
- "intermediate_size": 768,
412
- "is_decoder": false,
413
- "is_encoder_decoder": false,
414
- "label2id": {
415
- "LABEL_0": 0,
416
- "LABEL_1": 1
417
- },
418
- "length_penalty": 1.0,
419
- "max_length": 20,
420
- "max_position_embeddings": 65536,
421
- "min_length": 0,
422
- "mlp_only_layers": [],
423
- "model_type": "qwen3_omni_moe_text",
424
- "moe_intermediate_size": 768,
425
- "no_repeat_ngram_size": 0,
426
- "norm_topk_prob": true,
427
- "num_attention_heads": 32,
428
- "num_beam_groups": 1,
429
- "num_beams": 1,
430
- "num_experts": 128,
431
- "num_experts_per_tok": 8,
432
- "num_hidden_layers": 48,
433
- "num_key_value_heads": 4,
434
- "num_return_sequences": 1,
435
- "output_attentions": false,
436
- "output_hidden_states": false,
437
- "output_router_logits": false,
438
- "output_scores": false,
439
- "pad_token_id": null,
440
- "prefix": null,
441
- "problem_type": null,
442
- "pruned_heads": {},
443
- "remove_invalid_values": false,
444
- "repetition_penalty": 1.0,
445
- "return_dict": true,
446
- "return_dict_in_generate": false,
447
- "rms_norm_eps": 1e-06,
448
- "rope_parameters": {
449
- "interleaved": true,
450
- "mrope_interleaved": true,
451
- "mrope_section": [
452
- 24,
453
- 20,
454
- 20
455
- ],
456
- "rope_theta": 1000000,
457
- "rope_type": "default",
458
- "type": "default"
459
- },
460
- "rope_scaling": {
461
- "interleaved": true,
462
- "mrope_interleaved": true,
463
- "mrope_section": [
464
- 24,
465
- 20,
466
- 20
467
- ],
468
- "rope_type": "default",
469
- "type": "default"
470
- },
471
- "rope_theta": 1000000.0,
472
- "router_aux_loss_coef": 0.001,
473
- "sep_token_id": null,
474
- "shared_expert_intermediate_size": 0,
475
- "sliding_window": null,
476
- "suppress_tokens": null,
477
- "task_specific_params": null,
478
- "temperature": 1.0,
479
- "tf_legacy_loss": false,
480
- "tie_encoder_decoder": false,
481
- "tie_word_embeddings": false,
482
- "tokenizer_class": null,
483
- "top_k": 50,
484
- "top_p": 1.0,
485
- "torchscript": false,
486
- "typical_p": 1.0,
487
- "use_bfloat16": false,
488
- "use_cache": true,
489
- "use_qk_norm": true,
490
- "use_sliding_window": false,
491
- "vocab_size": 152064
492
- },
493
- "tie_word_embeddings": false,
494
- "user_token_id": 872,
495
- "video_token_id": 151656,
496
- "vision_config": {
497
- "_name_or_path": "",
498
- "add_cross_attention": false,
499
- "apply_vit_abs_pos_embed": true,
500
- "architectures": null,
501
- "bad_words_ids": null,
502
- "begin_suppress_tokens": null,
503
- "bos_token_id": null,
504
- "chunk_size_feed_forward": 0,
505
- "cross_attention_hidden_size": null,
506
- "decoder_start_token_id": null,
507
- "deepstack_visual_indexes": [
508
- 8,
509
- 16,
510
- 24
511
- ],
512
- "depth": 27,
513
- "diversity_penalty": 0.0,
514
- "do_sample": false,
515
- "dtype": "bfloat16",
516
- "early_stopping": false,
517
- "encoder_no_repeat_ngram_size": 0,
518
- "eos_token_id": null,
519
- "exponential_decay_length_penalty": null,
520
- "finetuning_task": null,
521
- "forced_bos_token_id": null,
522
- "forced_eos_token_id": null,
523
- "hidden_act": "gelu_pytorch_tanh",
524
- "hidden_size": 1152,
525
- "id2label": {
526
- "0": "LABEL_0",
527
- "1": "LABEL_1"
528
- },
529
- "image_size": 768,
530
- "in_channels": 3,
531
- "in_chans": 3,
532
- "initializer_range": 0.02,
533
- "intermediate_size": 4304,
534
- "is_decoder": false,
535
- "is_encoder_decoder": false,
536
- "label2id": {
537
- "LABEL_0": 0,
538
- "LABEL_1": 1
539
- },
540
- "length_penalty": 1.0,
541
- "max_length": 20,
542
- "min_length": 0,
543
- "model_type": "qwen3_omni_moe_vision_encoder",
544
- "no_repeat_ngram_size": 0,
545
- "num_beam_groups": 1,
546
- "num_beams": 1,
547
- "num_heads": 16,
548
- "num_position_embeddings": 2304,
549
- "num_return_sequences": 1,
550
- "out_hidden_size": 2048,
551
- "output_attentions": false,
552
- "output_hidden_states": false,
553
- "output_scores": false,
554
- "pad_token_id": null,
555
- "patch_size": 16,
556
- "prefix": null,
557
- "problem_type": null,
558
- "pruned_heads": {},
559
- "remove_invalid_values": false,
560
- "repetition_penalty": 1.0,
561
- "return_dict": true,
562
- "return_dict_in_generate": false,
563
- "sep_token_id": null,
564
- "spatial_merge_size": 2,
565
- "spatial_patch_size": 16,
566
- "suppress_tokens": null,
567
- "task_specific_params": null,
568
- "temperature": 1.0,
569
- "temporal_patch_size": 2,
570
- "tf_legacy_loss": false,
571
- "tie_encoder_decoder": false,
572
- "tie_word_embeddings": true,
573
- "tokenizer_class": null,
574
- "tokens_per_second": 2,
575
- "top_k": 50,
576
- "top_p": 1.0,
577
- "torchscript": false,
578
- "typical_p": 1.0,
579
- "use_bfloat16": false
580
- },
581
- "vision_end_token_id": 151653,
582
- "vision_start_token_id": 151652
583
- },
584
- "transformers_version": "4.57.1",
585
- "tts_bos_token_id": 151672,
586
- "tts_eos_token_id": 151673,
587
- "tts_pad_token_id": 151671,
588
- "user_token_id": 872
589
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/generation/generation_config.json DELETED
@@ -1,8 +0,0 @@
1
- {
2
- "talker_max_new_tokens": 4096,
3
- "talker_repetition_penalty": 1.05,
4
- "talker_temperature": 0.9,
5
- "talker_top_k": 50,
6
- "talker_top_p": 1.0,
7
- "transformers_version": "4.57.1"
8
- }
 
 
 
 
 
 
 
 
 
processors/generation/processor_config.json DELETED
@@ -1,109 +0,0 @@
1
- {
2
- "feature_extractor": {
3
- "chunk_length": 30,
4
- "dither": 0.0,
5
- "feature_extractor_type": "WhisperFeatureExtractor",
6
- "feature_size": 128,
7
- "hop_length": 160,
8
- "image_mean": [
9
- 0.5,
10
- 0.5,
11
- 0.5
12
- ],
13
- "image_processor_type": "Qwen2VLImageProcessor",
14
- "image_std": [
15
- 0.5,
16
- 0.5,
17
- 0.5
18
- ],
19
- "max_pixels": 12845056,
20
- "merge_size": 2,
21
- "min_pixels": 3136,
22
- "n_fft": 400,
23
- "n_samples": 480000,
24
- "nb_max_frames": 3000,
25
- "padding_side": "right",
26
- "padding_value": 0.0,
27
- "patch_size": 16,
28
- "return_attention_mask": true,
29
- "sampling_rate": 16000,
30
- "temporal_patch_size": 2
31
- },
32
- "image_processor": {
33
- "dither": 0.0,
34
- "do_convert_rgb": true,
35
- "do_normalize": true,
36
- "do_rescale": true,
37
- "do_resize": true,
38
- "feature_size": 128,
39
- "hop_length": 160,
40
- "image_mean": [
41
- 0.5,
42
- 0.5,
43
- 0.5
44
- ],
45
- "image_processor_type": "Qwen2VLImageProcessor",
46
- "image_std": [
47
- 0.5,
48
- 0.5,
49
- 0.5
50
- ],
51
- "merge_size": 2,
52
- "n_fft": 400,
53
- "n_samples": 4800000,
54
- "nb_max_frames": 30000,
55
- "padding_side": "right",
56
- "padding_value": 0.0,
57
- "patch_size": 16,
58
- "resample": 3,
59
- "rescale_factor": 0.00392156862745098,
60
- "return_attention_mask": true,
61
- "sampling_rate": 16000,
62
- "size": {
63
- "longest_edge": 12845056,
64
- "shortest_edge": 3136
65
- },
66
- "temporal_patch_size": 2
67
- },
68
- "processor_class": "Qwen3OmniMoeProcessor",
69
- "video_processor": {
70
- "dither": 0.0,
71
- "do_convert_rgb": true,
72
- "do_normalize": true,
73
- "do_rescale": true,
74
- "do_resize": true,
75
- "do_sample_frames": false,
76
- "feature_size": 128,
77
- "hop_length": 160,
78
- "image_mean": [
79
- 0.5,
80
- 0.5,
81
- 0.5
82
- ],
83
- "image_std": [
84
- 0.5,
85
- 0.5,
86
- 0.5
87
- ],
88
- "max_frames": 768,
89
- "merge_size": 2,
90
- "min_frames": 4,
91
- "n_fft": 400,
92
- "n_samples": 4800000,
93
- "nb_max_frames": 30000,
94
- "padding_side": "right",
95
- "padding_value": 0.0,
96
- "patch_size": 16,
97
- "resample": 3,
98
- "rescale_factor": 0.00392156862745098,
99
- "return_attention_mask": true,
100
- "return_metadata": false,
101
- "sampling_rate": 16000,
102
- "size": {
103
- "longest_edge": 12845056,
104
- "shortest_edge": 3136
105
- },
106
- "temporal_patch_size": 2,
107
- "video_processor_type": "Qwen2VLVideoProcessor"
108
- }
109
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/reasoning/added_tokens.json DELETED
@@ -1,24 +0,0 @@
1
- {
2
- "</tool_call>": 151658,
3
- "<tool_call>": 151657,
4
- "<|AUDIO|>": 151646,
5
- "<|IMAGE|>": 151655,
6
- "<|VIDEO|>": 151656,
7
- "<|audio_bos|>": 151647,
8
- "<|audio_eos|>": 151648,
9
- "<|box_end|>": 151649,
10
- "<|endoftext|>": 151643,
11
- "<|file_sep|>": 151664,
12
- "<|fim_middle|>": 151660,
13
- "<|fim_pad|>": 151662,
14
- "<|fim_prefix|>": 151659,
15
- "<|fim_suffix|>": 151661,
16
- "<|im_end|>": 151645,
17
- "<|im_start|>": 151644,
18
- "<|quad_end|>": 151651,
19
- "<|quad_start|>": 151650,
20
- "<|repo_name|>": 151663,
21
- "<|vision_bos|>": 151652,
22
- "<|vision_eos|>": 151653,
23
- "<|vision_pad|>": 151654
24
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/reasoning/chat_template.jinja DELETED
@@ -1,7 +0,0 @@
1
- {% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system
2
- You are a helpful assistant.<|im_end|>
3
- {% endif %}<|im_start|>{{ message['role'] }}
4
- {% if message['content'] is string %}{{ message['content'] }}<|im_end|>
5
- {% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>
6
- {% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant
7
- {% endif %}
 
 
 
 
 
 
 
 
processors/reasoning/chat_template.json DELETED
@@ -1,3 +0,0 @@
1
- {
2
- "chat_template": "{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}"
3
- }
 
 
 
 
processors/reasoning/config.json DELETED
@@ -1,650 +0,0 @@
1
- {
2
- "architectures": [
3
- "Qwen2_5OmniForConditionalGeneration"
4
- ],
5
- "dtype": "bfloat16",
6
- "enable_audio_output": false,
7
- "enable_talker": true,
8
- "model_type": "qwen2_5_omni",
9
- "talker_config": {
10
- "_attn_implementation_autoset": true,
11
- "_name_or_path": "Qwen2.5-Omni-3B/talker",
12
- "architectures": [
13
- "Qwen2OmniTalkerForConditionalGeneration"
14
- ],
15
- "attention_dropout": 0.0,
16
- "audio_end_token_id": 151648,
17
- "audio_start_token_id": 151647,
18
- "audio_token_index": 151646,
19
- "dtype": "bfloat16",
20
- "embedding_size": 2048,
21
- "head_dim": 64,
22
- "hidden_act": "silu",
23
- "hidden_size": 896,
24
- "image_token_index": 151655,
25
- "init_std": 0.02,
26
- "initializer_range": 0.02,
27
- "intermediate_size": 4864,
28
- "layer_types": [
29
- "full_attention",
30
- "full_attention",
31
- "full_attention",
32
- "full_attention",
33
- "full_attention",
34
- "full_attention",
35
- "full_attention",
36
- "full_attention",
37
- "full_attention",
38
- "full_attention",
39
- "full_attention",
40
- "full_attention",
41
- "full_attention",
42
- "full_attention",
43
- "full_attention",
44
- "full_attention",
45
- "full_attention",
46
- "full_attention",
47
- "full_attention",
48
- "full_attention",
49
- "full_attention",
50
- "full_attention",
51
- "full_attention",
52
- "full_attention"
53
- ],
54
- "max_position_embeddings": 32768,
55
- "max_window_layers": 28,
56
- "model_type": "qwen2_5_omni_talker",
57
- "num_attention_heads": 14,
58
- "num_hidden_layers": 24,
59
- "num_key_value_heads": 2,
60
- "position_id_per_seconds": 25,
61
- "rms_norm_eps": 1e-06,
62
- "rope_scaling": {
63
- "mrope_section": [
64
- 16,
65
- 16,
66
- 0
67
- ],
68
- "rope_type": "default",
69
- "type": "default"
70
- },
71
- "rope_theta": 1000000.0,
72
- "seconds_per_chunk": 2,
73
- "sliding_window": null,
74
- "spatial_merge_size": 2,
75
- "tts_codec_end_token_id": 8294,
76
- "tts_codec_mask_token_id": 8296,
77
- "tts_codec_pad_token_id": 8292,
78
- "tts_codec_start_token_id": 8293,
79
- "tts_text_end_token_id": 151861,
80
- "tts_text_pad_token_id": 151859,
81
- "tts_text_start_token_id": 151860,
82
- "use_cache": true,
83
- "use_sliding_window": false,
84
- "video_token_index": 151656,
85
- "vision_end_token_id": 151653,
86
- "vision_start_token_id": 151652,
87
- "vocab_size": 8448
88
- },
89
- "thinker_config": {
90
- "_attn_implementation_autoset": true,
91
- "_name_or_path": "Qwen2.5-Omni-3B/thinker",
92
- "architectures": [
93
- "Qwen2OmniNaViTThinkerForConditionalGeneration"
94
- ],
95
- "audio_config": {
96
- "_attn_implementation_autoset": true,
97
- "_name_or_path": "",
98
- "activation_dropout": 0.0,
99
- "activation_function": "gelu",
100
- "add_cross_attention": false,
101
- "architectures": null,
102
- "attention_dropout": 0.0,
103
- "bad_words_ids": null,
104
- "begin_suppress_tokens": null,
105
- "bos_token_id": null,
106
- "chunk_size_feed_forward": 0,
107
- "cross_attention_hidden_size": null,
108
- "d_model": 1280,
109
- "decoder_start_token_id": null,
110
- "diversity_penalty": 0.0,
111
- "do_sample": false,
112
- "dropout": 0.0,
113
- "dtype": null,
114
- "early_stopping": false,
115
- "encoder_attention_heads": 20,
116
- "encoder_ffn_dim": 5120,
117
- "encoder_layerdrop": 0.0,
118
- "encoder_layers": 32,
119
- "encoder_no_repeat_ngram_size": 0,
120
- "eos_token_id": null,
121
- "exponential_decay_length_penalty": null,
122
- "finetuning_task": null,
123
- "forced_bos_token_id": null,
124
- "forced_eos_token_id": null,
125
- "id2label": {
126
- "0": "LABEL_0",
127
- "1": "LABEL_1"
128
- },
129
- "init_std": 0.02,
130
- "initializer_range": 0.02,
131
- "is_decoder": false,
132
- "is_encoder_decoder": false,
133
- "label2id": {
134
- "LABEL_0": 0,
135
- "LABEL_1": 1
136
- },
137
- "length_penalty": 1.0,
138
- "max_length": 20,
139
- "max_source_positions": 1500,
140
- "min_length": 0,
141
- "model_type": "qwen2_5_omni_audio_encoder",
142
- "n_window": 100,
143
- "no_repeat_ngram_size": 0,
144
- "num_beam_groups": 1,
145
- "num_beams": 1,
146
- "num_hidden_layers": 32,
147
- "num_mel_bins": 128,
148
- "num_return_sequences": 1,
149
- "output_attentions": false,
150
- "output_dim": 2048,
151
- "output_hidden_states": false,
152
- "output_scores": false,
153
- "pad_token_id": null,
154
- "prefix": null,
155
- "problem_type": null,
156
- "pruned_heads": {},
157
- "remove_invalid_values": false,
158
- "repetition_penalty": 1.0,
159
- "return_dict": true,
160
- "return_dict_in_generate": false,
161
- "scale_embedding": false,
162
- "sep_token_id": null,
163
- "suppress_tokens": null,
164
- "task_specific_params": null,
165
- "temperature": 1.0,
166
- "tf_legacy_loss": false,
167
- "tie_encoder_decoder": false,
168
- "tie_word_embeddings": true,
169
- "tokenizer_class": null,
170
- "top_k": 50,
171
- "top_p": 1.0,
172
- "torchscript": false,
173
- "typical_p": 1.0,
174
- "use_bfloat16": false
175
- },
176
- "audio_end_token_id": 151648,
177
- "audio_start_token_id": 151647,
178
- "audio_token_index": 151646,
179
- "bos_token_id": 151644,
180
- "dtype": "bfloat16",
181
- "eos_token_id": 151645,
182
- "ignore_index": -100,
183
- "image_token_index": 151655,
184
- "init_std": 0.02,
185
- "initializer_range": 0.02,
186
- "model_type": "qwen2_5_omni_thinker",
187
- "pad_token_id": 151643,
188
- "position_id_per_seconds": 25,
189
- "seconds_per_chunk": 2,
190
- "text_config": {
191
- "_name_or_path": "",
192
- "add_cross_attention": false,
193
- "architectures": null,
194
- "attention_dropout": 0.0,
195
- "bad_words_ids": null,
196
- "begin_suppress_tokens": null,
197
- "bos_token_id": null,
198
- "chunk_size_feed_forward": 0,
199
- "cross_attention_hidden_size": null,
200
- "decoder_start_token_id": null,
201
- "diversity_penalty": 0.0,
202
- "do_sample": false,
203
- "dtype": null,
204
- "early_stopping": false,
205
- "encoder_no_repeat_ngram_size": 0,
206
- "eos_token_id": null,
207
- "exponential_decay_length_penalty": null,
208
- "finetuning_task": null,
209
- "forced_bos_token_id": null,
210
- "forced_eos_token_id": null,
211
- "hidden_act": "silu",
212
- "hidden_size": 2048,
213
- "id2label": {
214
- "0": "LABEL_0",
215
- "1": "LABEL_1"
216
- },
217
- "init_std": 0.02,
218
- "initializer_range": 0.02,
219
- "intermediate_size": 11008,
220
- "is_decoder": false,
221
- "is_encoder_decoder": false,
222
- "label2id": {
223
- "LABEL_0": 0,
224
- "LABEL_1": 1
225
- },
226
- "layer_types": [
227
- "full_attention",
228
- "full_attention",
229
- "full_attention",
230
- "full_attention",
231
- "full_attention",
232
- "full_attention",
233
- "full_attention",
234
- "full_attention",
235
- "full_attention",
236
- "full_attention",
237
- "full_attention",
238
- "full_attention",
239
- "full_attention",
240
- "full_attention",
241
- "full_attention",
242
- "full_attention",
243
- "full_attention",
244
- "full_attention",
245
- "full_attention",
246
- "full_attention",
247
- "full_attention",
248
- "full_attention",
249
- "full_attention",
250
- "full_attention",
251
- "full_attention",
252
- "full_attention",
253
- "full_attention",
254
- "full_attention",
255
- "full_attention",
256
- "full_attention",
257
- "full_attention",
258
- "full_attention",
259
- "full_attention",
260
- "full_attention",
261
- "full_attention",
262
- "full_attention"
263
- ],
264
- "length_penalty": 1.0,
265
- "max_length": 20,
266
- "max_position_embeddings": 32768,
267
- "max_window_layers": 70,
268
- "min_length": 0,
269
- "model_type": "qwen2_5_omni_text",
270
- "no_repeat_ngram_size": 0,
271
- "num_attention_heads": 16,
272
- "num_beam_groups": 1,
273
- "num_beams": 1,
274
- "num_hidden_layers": 36,
275
- "num_key_value_heads": 2,
276
- "num_return_sequences": 1,
277
- "output_attentions": false,
278
- "output_hidden_states": false,
279
- "output_scores": false,
280
- "pad_token_id": null,
281
- "prefix": null,
282
- "problem_type": null,
283
- "pruned_heads": {},
284
- "remove_invalid_values": false,
285
- "repetition_penalty": 1.0,
286
- "return_dict": true,
287
- "return_dict_in_generate": false,
288
- "rms_norm_eps": 1e-06,
289
- "rope_scaling": {
290
- "mrope_section": [
291
- 16,
292
- 24,
293
- 24
294
- ],
295
- "rope_type": "default",
296
- "type": "default"
297
- },
298
- "rope_theta": 1000000.0,
299
- "sep_token_id": null,
300
- "sliding_window": null,
301
- "suppress_tokens": null,
302
- "task_specific_params": null,
303
- "temperature": 1.0,
304
- "tf_legacy_loss": false,
305
- "tie_encoder_decoder": false,
306
- "tie_word_embeddings": false,
307
- "tokenizer_class": null,
308
- "top_k": 50,
309
- "top_p": 1.0,
310
- "torchscript": false,
311
- "typical_p": 1.0,
312
- "use_bfloat16": false,
313
- "use_cache": true,
314
- "use_sliding_window": false,
315
- "vocab_size": 151936,
316
- "rope_parameters": {
317
- "interleaved": true,
318
- "mrope_interleaved": true,
319
- "mrope_section": [
320
- 24,
321
- 20,
322
- 20
323
- ],
324
- "rope_theta": 1000000,
325
- "rope_type": "default",
326
- "type": "default"
327
- }
328
- },
329
- "tie_word_embeddings": false,
330
- "use_cache": true,
331
- "user_token_id": 872,
332
- "video_token_index": 151656,
333
- "vision_config": {
334
- "_attn_implementation_autoset": true,
335
- "_name_or_path": "",
336
- "add_cross_attention": false,
337
- "architectures": null,
338
- "bad_words_ids": null,
339
- "begin_suppress_tokens": null,
340
- "bos_token_id": null,
341
- "chunk_size_feed_forward": 0,
342
- "cross_attention_hidden_size": null,
343
- "decoder_start_token_id": null,
344
- "depth": 32,
345
- "diversity_penalty": 0.0,
346
- "do_sample": false,
347
- "dtype": null,
348
- "early_stopping": false,
349
- "embed_dim": 1280,
350
- "encoder_no_repeat_ngram_size": 0,
351
- "eos_token_id": null,
352
- "exponential_decay_length_penalty": null,
353
- "finetuning_task": null,
354
- "forced_bos_token_id": null,
355
- "forced_eos_token_id": null,
356
- "fullatt_block_indexes": [
357
- 7,
358
- 15,
359
- 23,
360
- 31
361
- ],
362
- "hidden_act": "silu",
363
- "hidden_size": 1280,
364
- "id2label": {
365
- "0": "LABEL_0",
366
- "1": "LABEL_1"
367
- },
368
- "in_channels": 3,
369
- "in_chans": 3,
370
- "init_std": 0.02,
371
- "initializer_range": 0.02,
372
- "intermediate_size": 3420,
373
- "is_decoder": false,
374
- "is_encoder_decoder": false,
375
- "label2id": {
376
- "LABEL_0": 0,
377
- "LABEL_1": 1
378
- },
379
- "length_penalty": 1.0,
380
- "max_length": 20,
381
- "min_length": 0,
382
- "model_type": "qwen2_5_omni_vision_encoder",
383
- "no_repeat_ngram_size": 0,
384
- "num_beam_groups": 1,
385
- "num_beams": 1,
386
- "num_heads": 16,
387
- "num_return_sequences": 1,
388
- "out_hidden_size": 2048,
389
- "output_attentions": false,
390
- "output_hidden_states": false,
391
- "output_scores": false,
392
- "pad_token_id": null,
393
- "patch_size": 14,
394
- "prefix": null,
395
- "problem_type": null,
396
- "pruned_heads": {},
397
- "remove_invalid_values": false,
398
- "repetition_penalty": 1.0,
399
- "return_dict": true,
400
- "return_dict_in_generate": false,
401
- "sep_token_id": null,
402
- "spatial_merge_size": 2,
403
- "spatial_patch_size": 14,
404
- "suppress_tokens": null,
405
- "task_specific_params": null,
406
- "temperature": 1.0,
407
- "temporal_patch_size": 2,
408
- "tf_legacy_loss": false,
409
- "tie_encoder_decoder": false,
410
- "tie_word_embeddings": true,
411
- "tokenizer_class": null,
412
- "tokens_per_second": 25,
413
- "top_k": 50,
414
- "top_p": 1.0,
415
- "torchscript": false,
416
- "typical_p": 1.0,
417
- "use_bfloat16": false,
418
- "window_size": 112
419
- },
420
- "vision_end_token_id": 151653,
421
- "vision_start_token_id": 151652,
422
- "vision_token_id": 151654
423
- },
424
- "token2wav_config": {
425
- "_attn_implementation_autoset": true,
426
- "bigvgan_config": {
427
- "_attn_implementation_autoset": true,
428
- "_name_or_path": "",
429
- "add_cross_attention": false,
430
- "architectures": null,
431
- "bad_words_ids": null,
432
- "begin_suppress_tokens": null,
433
- "bos_token_id": null,
434
- "chunk_size_feed_forward": 0,
435
- "cross_attention_hidden_size": null,
436
- "decoder_start_token_id": null,
437
- "diversity_penalty": 0.0,
438
- "do_sample": false,
439
- "dtype": null,
440
- "early_stopping": false,
441
- "encoder_no_repeat_ngram_size": 0,
442
- "eos_token_id": null,
443
- "exponential_decay_length_penalty": null,
444
- "finetuning_task": null,
445
- "forced_bos_token_id": null,
446
- "forced_eos_token_id": null,
447
- "id2label": {
448
- "0": "LABEL_0",
449
- "1": "LABEL_1"
450
- },
451
- "is_decoder": false,
452
- "is_encoder_decoder": false,
453
- "label2id": {
454
- "LABEL_0": 0,
455
- "LABEL_1": 1
456
- },
457
- "length_penalty": 1.0,
458
- "max_length": 20,
459
- "mel_dim": 80,
460
- "min_length": 0,
461
- "model_type": "qwen2_5_omni_bigvgan",
462
- "no_repeat_ngram_size": 0,
463
- "num_beam_groups": 1,
464
- "num_beams": 1,
465
- "num_return_sequences": 1,
466
- "output_attentions": false,
467
- "output_hidden_states": false,
468
- "output_scores": false,
469
- "pad_token_id": null,
470
- "prefix": null,
471
- "problem_type": null,
472
- "pruned_heads": {},
473
- "remove_invalid_values": false,
474
- "repetition_penalty": 1.0,
475
- "resblock_dilation_sizes": [
476
- [
477
- 1,
478
- 3,
479
- 5
480
- ],
481
- [
482
- 1,
483
- 3,
484
- 5
485
- ],
486
- [
487
- 1,
488
- 3,
489
- 5
490
- ]
491
- ],
492
- "resblock_kernel_sizes": [
493
- 3,
494
- 7,
495
- 11
496
- ],
497
- "return_dict": true,
498
- "return_dict_in_generate": false,
499
- "sep_token_id": null,
500
- "suppress_tokens": null,
501
- "task_specific_params": null,
502
- "temperature": 1.0,
503
- "tf_legacy_loss": false,
504
- "tie_encoder_decoder": false,
505
- "tie_word_embeddings": true,
506
- "tokenizer_class": null,
507
- "top_k": 50,
508
- "top_p": 1.0,
509
- "torchscript": false,
510
- "typical_p": 1.0,
511
- "upsample_initial_channel": 1536,
512
- "upsample_kernel_sizes": [
513
- 11,
514
- 7,
515
- 4,
516
- 4,
517
- 4,
518
- 4
519
- ],
520
- "upsample_rates": [
521
- 5,
522
- 3,
523
- 2,
524
- 2,
525
- 2,
526
- 2
527
- ],
528
- "use_bfloat16": false,
529
- "use_bias_at_final": false
530
- },
531
- "dit_config": {
532
- "_attn_implementation_autoset": true,
533
- "_name_or_path": "",
534
- "add_cross_attention": false,
535
- "architectures": null,
536
- "bad_words_ids": null,
537
- "begin_suppress_tokens": null,
538
- "block_size": 24,
539
- "bos_token_id": null,
540
- "chunk_size_feed_forward": 0,
541
- "cross_attention_hidden_size": null,
542
- "decoder_start_token_id": null,
543
- "depth": 22,
544
- "dim": 1024,
545
- "diversity_penalty": 0.0,
546
- "do_sample": false,
547
- "dropout": 0.1,
548
- "dtype": "float32",
549
- "early_stopping": false,
550
- "emb_dim": 512,
551
- "enc_attention_channels": 64,
552
- "enc_channels": [
553
- 256,
554
- 256,
555
- 256,
556
- 256,
557
- 768
558
- ],
559
- "enc_dilations": [
560
- 1,
561
- 2,
562
- 3,
563
- 4,
564
- 1
565
- ],
566
- "enc_dim": 128,
567
- "enc_emb_dim": 192,
568
- "enc_global_context": true,
569
- "enc_kernel_sizes": [
570
- 5,
571
- 3,
572
- 3,
573
- 3,
574
- 1
575
- ],
576
- "enc_lin_neurons": 192,
577
- "enc_res2net_scale": 2,
578
- "enc_se_channels": 64,
579
- "encoder_no_repeat_ngram_size": 0,
580
- "eos_token_id": null,
581
- "exponential_decay_length_penalty": null,
582
- "ff_mult": 2,
583
- "finetuning_task": null,
584
- "forced_bos_token_id": null,
585
- "forced_eos_token_id": null,
586
- "head_dim": 64,
587
- "heads": 16,
588
- "hidden_size": 1024,
589
- "id2label": {
590
- "0": "LABEL_0",
591
- "1": "LABEL_1"
592
- },
593
- "is_decoder": false,
594
- "is_encoder_decoder": false,
595
- "label2id": {
596
- "LABEL_0": 0,
597
- "LABEL_1": 1
598
- },
599
- "length_penalty": 1.0,
600
- "look_ahead_layers": [
601
- 10
602
- ],
603
- "look_backward_layers": [
604
- 0,
605
- 20
606
- ],
607
- "max_length": 20,
608
- "max_position_embeddings": 32768,
609
- "mel_dim": 80,
610
- "min_length": 0,
611
- "model_type": "qwen2_5_omni_dit",
612
- "no_repeat_ngram_size": 0,
613
- "num_attention_heads": 16,
614
- "num_beam_groups": 1,
615
- "num_beams": 1,
616
- "num_embeds": 8193,
617
- "num_hidden_layers": 22,
618
- "num_return_sequences": 1,
619
- "output_attentions": false,
620
- "output_hidden_states": false,
621
- "output_scores": false,
622
- "pad_token_id": null,
623
- "prefix": null,
624
- "problem_type": null,
625
- "pruned_heads": {},
626
- "remove_invalid_values": false,
627
- "repeats": 2,
628
- "repetition_penalty": 1.0,
629
- "return_dict": true,
630
- "return_dict_in_generate": false,
631
- "rope_theta": 10000.0,
632
- "sep_token_id": null,
633
- "suppress_tokens": null,
634
- "task_specific_params": null,
635
- "temperature": 1.0,
636
- "tf_legacy_loss": false,
637
- "tie_encoder_decoder": false,
638
- "tie_word_embeddings": true,
639
- "tokenizer_class": null,
640
- "top_k": 50,
641
- "top_p": 1.0,
642
- "torchscript": false,
643
- "typical_p": 1.0,
644
- "use_bfloat16": false
645
- },
646
- "dtype": "bfloat16",
647
- "model_type": "qwen2_5_omni_token2wav"
648
- },
649
- "transformers_version": "4.57.1"
650
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/reasoning/generation_config.json DELETED
@@ -1,4 +0,0 @@
1
- {
2
- "_from_model_config": true,
3
- "transformers_version": "4.57.1"
4
- }
 
 
 
 
 
processors/reasoning/merges.txt DELETED
The diff for this file is too large to render. See raw diff
 
processors/reasoning/preprocessor_config.json DELETED
@@ -1,31 +0,0 @@
1
- {
2
- "chunk_length": 300,
3
- "dither": 0.0,
4
- "feature_extractor_type": "WhisperFeatureExtractor",
5
- "feature_size": 128,
6
- "hop_length": 160,
7
- "image_mean": [
8
- 0.48145466,
9
- 0.4578275,
10
- 0.40821073
11
- ],
12
- "image_processor_type": "Qwen2VLImageProcessor",
13
- "image_std": [
14
- 0.26862954,
15
- 0.26130258,
16
- 0.27577711
17
- ],
18
- "max_pixels": 12845056,
19
- "merge_size": 2,
20
- "min_pixels": 3136,
21
- "n_fft": 400,
22
- "n_samples": 4800000,
23
- "nb_max_frames": 30000,
24
- "padding_side": "right",
25
- "padding_value": 0.0,
26
- "patch_size": 14,
27
- "processor_class": "Qwen2_5OmniProcessor",
28
- "return_attention_mask": true,
29
- "sampling_rate": 16000,
30
- "temporal_patch_size": 2
31
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/reasoning/special_tokens_map.json DELETED
@@ -1,38 +0,0 @@
1
- {
2
- "additional_special_tokens": [
3
- "<|im_start|>",
4
- "<|im_end|>",
5
- "<|AUDIO|>",
6
- "<|audio_bos|>",
7
- "<|audio_eos|>",
8
- "<|box_end|>",
9
- "<|quad_start|>",
10
- "<|quad_end|>",
11
- "<|vision_bos|>",
12
- "<|vision_eos|>",
13
- "<|vision_pad|>",
14
- "<|IMAGE|>",
15
- "<|VIDEO|>"
16
- ],
17
- "audio_bos_token": "<|audio_bos|>",
18
- "audio_eos_token": "<|audio_eos|>",
19
- "audio_token": "<|AUDIO|>",
20
- "eos_token": {
21
- "content": "<|im_end|>",
22
- "lstrip": false,
23
- "normalized": false,
24
- "rstrip": false,
25
- "single_word": false
26
- },
27
- "image_token": "<|IMAGE|>",
28
- "pad_token": {
29
- "content": "<|endoftext|>",
30
- "lstrip": false,
31
- "normalized": false,
32
- "rstrip": false,
33
- "single_word": false
34
- },
35
- "video_token": "<|VIDEO|>",
36
- "vision_bos_token": "<|vision_bos|>",
37
- "vision_eos_token": "<|vision_eos|>"
38
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/reasoning/tokenizer.json DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:8441917e39ae0244e06d704b95b3124795cec478e297f9afac39ba670d7e9d99
3
- size 11421870
 
 
 
 
processors/reasoning/tokenizer_config.json DELETED
@@ -1,222 +0,0 @@
1
- {
2
- "add_prefix_space": false,
3
- "added_tokens_decoder": {
4
- "151643": {
5
- "content": "<|endoftext|>",
6
- "lstrip": false,
7
- "normalized": false,
8
- "rstrip": false,
9
- "single_word": false,
10
- "special": true
11
- },
12
- "151644": {
13
- "content": "<|im_start|>",
14
- "lstrip": false,
15
- "normalized": false,
16
- "rstrip": false,
17
- "single_word": false,
18
- "special": true
19
- },
20
- "151645": {
21
- "content": "<|im_end|>",
22
- "lstrip": false,
23
- "normalized": false,
24
- "rstrip": false,
25
- "single_word": false,
26
- "special": true
27
- },
28
- "151646": {
29
- "content": "<|AUDIO|>",
30
- "lstrip": false,
31
- "normalized": false,
32
- "rstrip": false,
33
- "single_word": false,
34
- "special": true
35
- },
36
- "151647": {
37
- "content": "<|audio_bos|>",
38
- "lstrip": false,
39
- "normalized": false,
40
- "rstrip": false,
41
- "single_word": false,
42
- "special": true
43
- },
44
- "151648": {
45
- "content": "<|audio_eos|>",
46
- "lstrip": false,
47
- "normalized": false,
48
- "rstrip": false,
49
- "single_word": false,
50
- "special": true
51
- },
52
- "151649": {
53
- "content": "<|box_end|>",
54
- "lstrip": false,
55
- "normalized": false,
56
- "rstrip": false,
57
- "single_word": false,
58
- "special": true
59
- },
60
- "151650": {
61
- "content": "<|quad_start|>",
62
- "lstrip": false,
63
- "normalized": false,
64
- "rstrip": false,
65
- "single_word": false,
66
- "special": true
67
- },
68
- "151651": {
69
- "content": "<|quad_end|>",
70
- "lstrip": false,
71
- "normalized": false,
72
- "rstrip": false,
73
- "single_word": false,
74
- "special": true
75
- },
76
- "151652": {
77
- "content": "<|vision_bos|>",
78
- "lstrip": false,
79
- "normalized": false,
80
- "rstrip": false,
81
- "single_word": false,
82
- "special": true
83
- },
84
- "151653": {
85
- "content": "<|vision_eos|>",
86
- "lstrip": false,
87
- "normalized": false,
88
- "rstrip": false,
89
- "single_word": false,
90
- "special": true
91
- },
92
- "151654": {
93
- "content": "<|vision_pad|>",
94
- "lstrip": false,
95
- "normalized": false,
96
- "rstrip": false,
97
- "single_word": false,
98
- "special": true
99
- },
100
- "151655": {
101
- "content": "<|IMAGE|>",
102
- "lstrip": false,
103
- "normalized": false,
104
- "rstrip": false,
105
- "single_word": false,
106
- "special": true
107
- },
108
- "151656": {
109
- "content": "<|VIDEO|>",
110
- "lstrip": false,
111
- "normalized": false,
112
- "rstrip": false,
113
- "single_word": false,
114
- "special": true
115
- },
116
- "151657": {
117
- "content": "<tool_call>",
118
- "lstrip": false,
119
- "normalized": false,
120
- "rstrip": false,
121
- "single_word": false,
122
- "special": false
123
- },
124
- "151658": {
125
- "content": "</tool_call>",
126
- "lstrip": false,
127
- "normalized": false,
128
- "rstrip": false,
129
- "single_word": false,
130
- "special": false
131
- },
132
- "151659": {
133
- "content": "<|fim_prefix|>",
134
- "lstrip": false,
135
- "normalized": false,
136
- "rstrip": false,
137
- "single_word": false,
138
- "special": false
139
- },
140
- "151660": {
141
- "content": "<|fim_middle|>",
142
- "lstrip": false,
143
- "normalized": false,
144
- "rstrip": false,
145
- "single_word": false,
146
- "special": false
147
- },
148
- "151661": {
149
- "content": "<|fim_suffix|>",
150
- "lstrip": false,
151
- "normalized": false,
152
- "rstrip": false,
153
- "single_word": false,
154
- "special": false
155
- },
156
- "151662": {
157
- "content": "<|fim_pad|>",
158
- "lstrip": false,
159
- "normalized": false,
160
- "rstrip": false,
161
- "single_word": false,
162
- "special": false
163
- },
164
- "151663": {
165
- "content": "<|repo_name|>",
166
- "lstrip": false,
167
- "normalized": false,
168
- "rstrip": false,
169
- "single_word": false,
170
- "special": false
171
- },
172
- "151664": {
173
- "content": "<|file_sep|>",
174
- "lstrip": false,
175
- "normalized": false,
176
- "rstrip": false,
177
- "single_word": false,
178
- "special": false
179
- }
180
- },
181
- "additional_special_tokens": [
182
- "<|im_start|>",
183
- "<|im_end|>",
184
- "<|AUDIO|>",
185
- "<|audio_bos|>",
186
- "<|audio_eos|>",
187
- "<|box_end|>",
188
- "<|quad_start|>",
189
- "<|quad_end|>",
190
- "<|vision_bos|>",
191
- "<|vision_eos|>",
192
- "<|vision_pad|>",
193
- "<|IMAGE|>",
194
- "<|VIDEO|>"
195
- ],
196
- "audio_bos_token": "<|audio_bos|>",
197
- "audio_eos_token": "<|audio_eos|>",
198
- "audio_token": "<|AUDIO|>",
199
- "bos_token": null,
200
- "clean_up_tokenization_spaces": false,
201
- "eos_token": "<|im_end|>",
202
- "errors": "replace",
203
- "extra_special_tokens": {
204
- "audio_bos_token": "<|audio_bos|>",
205
- "audio_eos_token": "<|audio_eos|>",
206
- "audio_token": "<|AUDIO|>",
207
- "image_token": "<|IMAGE|>",
208
- "video_token": "<|VIDEO|>",
209
- "vision_bos_token": "<|vision_bos|>",
210
- "vision_eos_token": "<|vision_eos|>"
211
- },
212
- "image_token": "<|IMAGE|>",
213
- "model_max_length": 32768,
214
- "pad_token": "<|endoftext|>",
215
- "processor_class": "Qwen2_5OmniProcessor",
216
- "split_special_tokens": false,
217
- "tokenizer_class": "Qwen2Tokenizer",
218
- "unk_token": null,
219
- "video_token": "<|VIDEO|>",
220
- "vision_bos_token": "<|vision_bos|>",
221
- "vision_eos_token": "<|vision_eos|>"
222
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/reasoning/video_preprocessor_config.json DELETED
@@ -1,55 +0,0 @@
1
- {
2
- "chunk_length": 300,
3
- "crop_size": null,
4
- "data_format": "channels_first",
5
- "default_to_square": true,
6
- "device": null,
7
- "dither": 0.0,
8
- "do_center_crop": null,
9
- "do_convert_rgb": true,
10
- "do_normalize": true,
11
- "do_rescale": true,
12
- "do_resize": true,
13
- "do_sample_frames": false,
14
- "feature_extractor_type": "WhisperFeatureExtractor",
15
- "feature_size": 128,
16
- "fps": null,
17
- "hop_length": 160,
18
- "image_mean": [
19
- 0.48145466,
20
- 0.4578275,
21
- 0.40821073
22
- ],
23
- "image_std": [
24
- 0.26862954,
25
- 0.26130258,
26
- 0.27577711
27
- ],
28
- "input_data_format": null,
29
- "max_frames": 768,
30
- "max_pixels": 12845056,
31
- "merge_size": 2,
32
- "min_frames": 4,
33
- "min_pixels": 3136,
34
- "n_fft": 400,
35
- "n_samples": 4800000,
36
- "nb_max_frames": 30000,
37
- "num_frames": null,
38
- "pad_size": null,
39
- "padding_side": "right",
40
- "padding_value": 0.0,
41
- "patch_size": 14,
42
- "processor_class": "Qwen2_5OmniProcessor",
43
- "resample": 3,
44
- "rescale_factor": 0.00392156862745098,
45
- "return_attention_mask": true,
46
- "return_metadata": false,
47
- "sampling_rate": 16000,
48
- "size": {
49
- "longest_edge": 12845056,
50
- "shortest_edge": 3136
51
- },
52
- "temporal_patch_size": 2,
53
- "video_metadata": null,
54
- "video_processor_type": "Qwen2VLVideoProcessor"
55
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
processors/reasoning/vocab.json DELETED
The diff for this file is too large to render. See raw diff
 
pyproject.toml CHANGED
@@ -4,33 +4,23 @@ build-backend = "setuptools.build_meta"
4
 
5
  [project]
6
  name = "myna-hokkien"
7
- version = "0.1.0"
8
- description = "iNLP-Lab Myna-Hokkien inference runtime"
9
  requires-python = ">=3.10"
10
  dependencies = [
11
  "torch>=2.6",
12
- "torchvision",
13
  "transformers==4.57.1",
14
- "accelerate==1.14.0",
15
- "peft==0.17.1",
16
- "huggingface_hub==0.36.2",
17
- "safetensors==0.8.0",
18
- "numpy==2.2.6",
19
- "librosa==0.11.0",
20
- "soundfile==0.14.0",
21
- "pillow>=10.0",
22
  ]
23
 
24
  [project.urls]
25
  Homepage = "https://huggingface.co/iNLP-Lab/MynaHokkien"
26
  Repository = "https://huggingface.co/iNLP-Lab/MynaHokkien"
27
 
28
- [tool.setuptools]
29
- py-modules = [
30
- "configuration_mynahokkien",
31
- "modeling_mynahokkien",
32
- "processing_mynahokkien",
33
- ]
34
-
35
  [tool.setuptools.packages.find]
36
  include = ["mynahokkien*"]
 
4
 
5
  [project]
6
  name = "myna-hokkien"
7
+ version = "0.2.0"
8
+ description = "iNLP-Lab Myna-Hokkien native Qwen3-Omni inference runtime"
9
  requires-python = ">=3.10"
10
  dependencies = [
11
  "torch>=2.6",
 
12
  "transformers==4.57.1",
13
+ "accelerate>=1.10,<2",
14
+ "huggingface_hub>=0.36,<2",
15
+ "safetensors>=0.4",
16
+ "numpy>=1.26,<3",
17
+ "librosa>=0.11,<1",
18
+ "soundfile>=0.13,<1",
 
 
19
  ]
20
 
21
  [project.urls]
22
  Homepage = "https://huggingface.co/iNLP-Lab/MynaHokkien"
23
  Repository = "https://huggingface.co/iNLP-Lab/MynaHokkien"
24
 
 
 
 
 
 
 
 
25
  [tool.setuptools.packages.find]
26
  include = ["mynahokkien*"]
release_config.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "release": "Myna-Hokkien release_v2",
3
+ "architecture": "native Qwen3-Omni-30B-A3B-Instruct",
4
+ "thinker": "release_v2 Thinker Stage 2",
5
+ "talker": "release_v2 Talker Stage 4 epoch 1",
6
+ "speaker": "Ethan",
7
+ "language": "nan",
8
+ "input_sampling_rate": 16000,
9
+ "output_sampling_rate": 24000,
10
+ "system_prompt": "You are Qwen, a virtual human developed by the Qwen Team, Alibaba Group, capable of perceiving auditory inputs and generating text and speech.",
11
+ "default_audio_prompt": "Listen to the spoken Hokkien and reply naturally in concise Singapore Hokkien. Always answer in colloquial Singapore Hokkien written in Hanji. Never answer in Mandarin or English. Do not repeat or transcribe the input; respond to it directly.",
12
+ "default_text_prompt": "Read the text and reply naturally in concise Singapore Hokkien. Always answer in colloquial Singapore Hokkien written in Hanji. Never answer in Mandarin or English. Do not repeat the input; respond to it directly.",
13
+ "prompt_position": "user turn",
14
+ "thinker_do_sample": false,
15
+ "talker_do_sample": false,
16
+ "thinker_max_new_tokens": 256,
17
+ "talker_max_new_tokens": 4096
18
+ }
requirements.txt CHANGED
@@ -1,11 +1,8 @@
1
  torch>=2.6
2
- torchvision
3
  transformers==4.57.1
4
- accelerate==1.14.0
5
- peft==0.17.1
6
- huggingface_hub==0.36.2
7
- safetensors==0.8.0
8
- numpy==2.2.6
9
- librosa==0.11.0
10
- soundfile==0.14.0
11
- pillow>=10.0
 
1
  torch>=2.6
 
2
  transformers==4.57.1
3
+ accelerate>=1.10,<2
4
+ huggingface_hub>=0.36,<2
5
+ safetensors>=0.4
6
+ numpy>=1.26,<3
7
+ librosa>=0.11,<1
8
+ soundfile>=0.13,<1
 
 
processors/generation/special_tokens_map.json → special_tokens_map.json RENAMED
File without changes
processors/generation/tokenizer.json → tokenizer.json RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b34e316e4a3886855d9af88d0713da9a41dc8e00359c890ac022df6c6a1101a3
3
- size 11424429
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:09267689b8362020b9763b65dd5be7e086b31e28d72e02837a9e781de9a91bc7
3
+ size 11423986
processors/generation/tokenizer_config.json → tokenizer_config.json RENAMED
@@ -305,12 +305,8 @@
305
  "vision_eos_token": "<|vision_end|>"
306
  },
307
  "image_token": "<|image_pad|>",
308
- "max_length": null,
309
  "model_max_length": 131072,
310
- "pad_to_multiple_of": null,
311
  "pad_token": "<|endoftext|>",
312
- "pad_token_type_id": 0,
313
- "padding_side": "left",
314
  "processor_class": "Qwen3OmniMoeProcessor",
315
  "split_special_tokens": false,
316
  "tokenizer_class": "Qwen2Tokenizer",
 
305
  "vision_eos_token": "<|vision_end|>"
306
  },
307
  "image_token": "<|image_pad|>",
 
308
  "model_max_length": 131072,
 
309
  "pad_token": "<|endoftext|>",
 
 
310
  "processor_class": "Qwen3OmniMoeProcessor",
311
  "split_special_tokens": false,
312
  "tokenizer_class": "Qwen2Tokenizer",
processors/generation/video_preprocessor_config.json → video_preprocessor_config.json RENAMED
File without changes