ACE-Step/acestep-captioner

zetlyn/models-hf model hf ACE-Step/acestep-captioner known 2026-01-23

https://huggingface.co/ACE-Step/acestep-captioner

Properties

authorACE-Step
receipt
Source
Hugging Face models
Its words
ACE-Step
Read by
field:author
Said since
2026-10-02 12:02 UTC
Last answered
2026-10-04 18:21 UTC
Original
open at the source
What the source handed over
{
  "_asked": "ACE-Step/acestep-captioner",
  "_id": "69734cc640779d7b34895660",
  "author": "ACE-Step",
  "cardData": {
    "library_name": "transformers",
    "license": "mit",
    "tags": [
      "music",
      "audio"
    ]
  },
  "config": {
    "architectures": [
      "Qwen2_5OmniForConditionalGeneration"
    ],
    "chat_template_jinja": "{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}",
    "model_type": "qwen2_5_omni",
    "tokenizer_config": {
      "bos_token": null,
      "eos_token": "<|im_end|>",
      "pad_token": "<|endoftext|>",
      "unk_token": null
    }
  },
  "createdAt": "2026-01-23T10:26:14.000Z",
  "disabled": false,
  "downloads": 1792,
  "gated": false,
  "id": "ACE-Step/acestep-captioner",
  "lastModified": "2026-02-03T06:33:55.000Z",
  "library_name": "transformers",
  "likes": 74,
  "model-index": null,
  "modelId": "ACE-Step/acestep-captioner",
  "pipeline_tag": "text-to-audio",
  "private": false,
  "safetensors": {
    "parameters": {
      "BF16": 10283174144,
      "F32": 449051296
    },
    "total": 10732225408
  },
  "sha": "7109e9af1c6cfe3371730310b3a5cb773145b11e",
  "siblings": [
    {
      "rfilename": ".gitattributes"
    },
    {
      "rfilename": "README.md"
    },
    {
      "rfilename": "added_tokens.json"
    },
    {
      "rfilename": "args.json"
    },
    {
      "rfilename": "chat_template.jinja"
    },
    {
      "rfilename": "config.json"
    },
    {
      "rfilename": "file_list.txt"
    },
    {
      "rfilename": "generation_config.json"
    },
    {
      "rfilename": "merges.txt"
    },
    {
      "rfilename": "model-00001-of-00005.safetensors"
    },
    {
      "rfilename": "model-00002-of-00005.safetensors"
    },
    {
      "rfilename": "model-00003-of-00005.safetensors"
    },
    {
      "rfilename": "model-00004-of-00005.safetensors"
    },
    {
      "rfilename": "model-00005-of-00005.safetensors"
    },
    {
      "rfilename": "model.safetensors.index.json"
    },
    {
      "rfilename": "preprocessor_config.json"
    },
    {
      "rfilename": "special_tokens_map.json"
    },
    {
      "rfilename": "spk_dict.pt"
    },
    {
      "rfilename": "tokenizer.json"
    },
    {
      "rfilename": "tokenizer_config.json"
    },
    {
      "rfilename": "video_preprocessor_config.json"
    },
    {
      "rfilename": "vocab.json"
    }
  ],
  "spaces": [
    "DandyDonUnhinged/spock-body-lora-training"
  ],
  "tags": [
    "transformers",
    "safetensors",
    "qwen2_5_omni",
    "text-to-audio",
    "music",
    "audio",
    "arxiv:2602.00744",
    "license:mit",
    "endpoints_compatible",
    "region:us"
  ],
  "transformersInfo": {
    "auto_model": "AutoModelForMultimodalLM",
    "pipeline_tag": "text-to-audio",
    "processor": "AutoProcessor"
  },
  "usedStorage": 22374542766
}
downloads1792
receipt
Source
Hugging Face models
Its words
1792
Read by
field:downloads
Said since
2026-10-04 12:16 UTC
Last answered
2026-10-04 18:21 UTC
Original
open at the source
2026-10-04 12:16 UTC1792
2026-10-03 12:09 UTC1841
2026-10-02 12:02 UTC1874
What the source handed over
{
  "_asked": "ACE-Step/acestep-captioner",
  "_id": "69734cc640779d7b34895660",
  "author": "ACE-Step",
  "cardData": {
    "library_name": "transformers",
    "license": "mit",
    "tags": [
      "music",
      "audio"
    ]
  },
  "config": {
    "architectures": [
      "Qwen2_5OmniForConditionalGeneration"
    ],
    "chat_template_jinja": "{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}",
    "model_type": "qwen2_5_omni",
    "tokenizer_config": {
      "bos_token": null,
      "eos_token": "<|im_end|>",
      "pad_token": "<|endoftext|>",
      "unk_token": null
    }
  },
  "createdAt": "2026-01-23T10:26:14.000Z",
  "disabled": false,
  "downloads": 1792,
  "gated": false,
  "id": "ACE-Step/acestep-captioner",
  "lastModified": "2026-02-03T06:33:55.000Z",
  "library_name": "transformers",
  "likes": 74,
  "model-index": null,
  "modelId": "ACE-Step/acestep-captioner",
  "pipeline_tag": "text-to-audio",
  "private": false,
  "safetensors": {
    "parameters": {
      "BF16": 10283174144,
      "F32": 449051296
    },
    "total": 10732225408
  },
  "sha": "7109e9af1c6cfe3371730310b3a5cb773145b11e",
  "siblings": [
    {
      "rfilename": ".gitattributes"
    },
    {
      "rfilename": "README.md"
    },
    {
      "rfilename": "added_tokens.json"
    },
    {
      "rfilename": "args.json"
    },
    {
      "rfilename": "chat_template.jinja"
    },
    {
      "rfilename": "config.json"
    },
    {
      "rfilename": "file_list.txt"
    },
    {
      "rfilename": "generation_config.json"
    },
    {
      "rfilename": "merges.txt"
    },
    {
      "rfilename": "model-00001-of-00005.safetensors"
    },
    {
      "rfilename": "model-00002-of-00005.safetensors"
    },
    {
      "rfilename": "model-00003-of-00005.safetensors"
    },
    {
      "rfilename": "model-00004-of-00005.safetensors"
    },
    {
      "rfilename": "model-00005-of-00005.safetensors"
    },
    {
      "rfilename": "model.safetensors.index.json"
    },
    {
      "rfilename": "preprocessor_config.json"
    },
    {
      "rfilename": "special_tokens_map.json"
    },
    {
      "rfilename": "spk_dict.pt"
    },
    {
      "rfilename": "tokenizer.json"
    },
    {
      "rfilename": "tokenizer_config.json"
    },
    {
      "rfilename": "video_preprocessor_config.json"
    },
    {
      "rfilename": "vocab.json"
    }
  ],
  "spaces": [
    "DandyDonUnhinged/spock-body-lora-training"
  ],
  "tags": [
    "transformers",
    "safetensors",
    "qwen2_5_omni",
    "text-to-audio",
    "music",
    "audio",
    "arxiv:2602.00744",
    "license:mit",
    "endpoints_compatible",
    "region:us"
  ],
  "transformersInfo": {
    "auto_model": "AutoModelForMultimodalLM",
    "pipeline_tag": "text-to-audio",
    "processor": "AutoProcessor"
  },
  "usedStorage": 22374542766
}
gatedfalse
receipt
Source
Hugging Face models
Its words
false
Read by
field:gated
Said since
2026-10-02 12:02 UTC
Last answered
2026-10-04 18:21 UTC
Original
open at the source
What the source handed over
{
  "_asked": "ACE-Step/acestep-captioner",
  "_id": "69734cc640779d7b34895660",
  "author": "ACE-Step",
  "cardData": {
    "library_name": "transformers",
    "license": "mit",
    "tags": [
      "music",
      "audio"
    ]
  },
  "config": {
    "architectures": [
      "Qwen2_5OmniForConditionalGeneration"
    ],
    "chat_template_jinja": "{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}",
    "model_type": "qwen2_5_omni",
    "tokenizer_config": {
      "bos_token": null,
      "eos_token": "<|im_end|>",
      "pad_token": "<|endoftext|>",
      "unk_token": null
    }
  },
  "createdAt": "2026-01-23T10:26:14.000Z",
  "disabled": false,
  "downloads": 1792,
  "gated": false,
  "id": "ACE-Step/acestep-captioner",
  "lastModified": "2026-02-03T06:33:55.000Z",
  "library_name": "transformers",
  "likes": 74,
  "model-index": null,
  "modelId": "ACE-Step/acestep-captioner",
  "pipeline_tag": "text-to-audio",
  "private": false,
  "safetensors": {
    "parameters": {
      "BF16": 10283174144,
      "F32": 449051296
    },
    "total": 10732225408
  },
  "sha": "7109e9af1c6cfe3371730310b3a5cb773145b11e",
  "siblings": [
    {
      "rfilename": ".gitattributes"
    },
    {
      "rfilename": "README.md"
    },
    {
      "rfilename": "added_tokens.json"
    },
    {
      "rfilename": "args.json"
    },
    {
      "rfilename": "chat_template.jinja"
    },
    {
      "rfilename": "config.json"
    },
    {
      "rfilename": "file_list.txt"
    },
    {
      "rfilename": "generation_config.json"
    },
    {
      "rfilename": "merges.txt"
    },
    {
      "rfilename": "model-00001-of-00005.safetensors"
    },
    {
      "rfilename": "model-00002-of-00005.safetensors"
    },
    {
      "rfilename": "model-00003-of-00005.safetensors"
    },
    {
      "rfilename": "model-00004-of-00005.safetensors"
    },
    {
      "rfilename": "model-00005-of-00005.safetensors"
    },
    {
      "rfilename": "model.safetensors.index.json"
    },
    {
      "rfilename": "preprocessor_config.json"
    },
    {
      "rfilename": "special_tokens_map.json"
    },
    {
      "rfilename": "spk_dict.pt"
    },
    {
      "rfilename": "tokenizer.json"
    },
    {
      "rfilename": "tokenizer_config.json"
    },
    {
      "rfilename": "video_preprocessor_config.json"
    },
    {
      "rfilename": "vocab.json"
    }
  ],
  "spaces": [
    "DandyDonUnhinged/spock-body-lora-training"
  ],
  "tags": [
    "transformers",
    "safetensors",
    "qwen2_5_omni",
    "text-to-audio",
    "music",
    "audio",
    "arxiv:2602.00744",
    "license:mit",
    "endpoints_compatible",
    "region:us"
  ],
  "transformersInfo": {
    "auto_model": "AutoModelForMultimodalLM",
    "pipeline_tag": "text-to-audio",
    "processor": "AutoProcessor"
  },
  "usedStorage": 22374542766
}
licencemit
receipt
Source
Hugging Face models
Its words
mit
Read by
field:cardData.license
Said since
2026-10-02 12:02 UTC
Last answered
2026-10-04 18:21 UTC
Original
open at the source
What the source handed over
{
  "_asked": "ACE-Step/acestep-captioner",
  "_id": "69734cc640779d7b34895660",
  "author": "ACE-Step",
  "cardData": {
    "library_name": "transformers",
    "license": "mit",
    "tags": [
      "music",
      "audio"
    ]
  },
  "config": {
    "architectures": [
      "Qwen2_5OmniForConditionalGeneration"
    ],
    "chat_template_jinja": "{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}",
    "model_type": "qwen2_5_omni",
    "tokenizer_config": {
      "bos_token": null,
      "eos_token": "<|im_end|>",
      "pad_token": "<|endoftext|>",
      "unk_token": null
    }
  },
  "createdAt": "2026-01-23T10:26:14.000Z",
  "disabled": false,
  "downloads": 1792,
  "gated": false,
  "id": "ACE-Step/acestep-captioner",
  "lastModified": "2026-02-03T06:33:55.000Z",
  "library_name": "transformers",
  "likes": 74,
  "model-index": null,
  "modelId": "ACE-Step/acestep-captioner",
  "pipeline_tag": "text-to-audio",
  "private": false,
  "safetensors": {
    "parameters": {
      "BF16": 10283174144,
      "F32": 449051296
    },
    "total": 10732225408
  },
  "sha": "7109e9af1c6cfe3371730310b3a5cb773145b11e",
  "siblings": [
    {
      "rfilename": ".gitattributes"
    },
    {
      "rfilename": "README.md"
    },
    {
      "rfilename": "added_tokens.json"
    },
    {
      "rfilename": "args.json"
    },
    {
      "rfilename": "chat_template.jinja"
    },
    {
      "rfilename": "config.json"
    },
    {
      "rfilename": "file_list.txt"
    },
    {
      "rfilename": "generation_config.json"
    },
    {
      "rfilename": "merges.txt"
    },
    {
      "rfilename": "model-00001-of-00005.safetensors"
    },
    {
      "rfilename": "model-00002-of-00005.safetensors"
    },
    {
      "rfilename": "model-00003-of-00005.safetensors"
    },
    {
      "rfilename": "model-00004-of-00005.safetensors"
    },
    {
      "rfilename": "model-00005-of-00005.safetensors"
    },
    {
      "rfilename": "model.safetensors.index.json"
    },
    {
      "rfilename": "preprocessor_config.json"
    },
    {
      "rfilename": "special_tokens_map.json"
    },
    {
      "rfilename": "spk_dict.pt"
    },
    {
      "rfilename": "tokenizer.json"
    },
    {
      "rfilename": "tokenizer_config.json"
    },
    {
      "rfilename": "video_preprocessor_config.json"
    },
    {
      "rfilename": "vocab.json"
    }
  ],
  "spaces": [
    "DandyDonUnhinged/spock-body-lora-training"
  ],
  "tags": [
    "transformers",
    "safetensors",
    "qwen2_5_omni",
    "text-to-audio",
    "music",
    "audio",
    "arxiv:2602.00744",
    "license:mit",
    "endpoints_compatible",
    "region:us"
  ],
  "transformersInfo": {
    "auto_model": "AutoModelForMultimodalLM",
    "pipeline_tag": "text-to-audio",
    "processor": "AutoProcessor"
  },
  "usedStorage": 22374542766
}
likes74
receipt
Source
Hugging Face models
Its words
74
Read by
field:likes
Said since
2026-10-04 00:12 UTC
Last answered
2026-10-04 18:21 UTC
Original
open at the source
2026-10-04 00:12 UTC74
2026-10-02 12:02 UTC73
What the source handed over
{
  "_asked": "ACE-Step/acestep-captioner",
  "_id": "69734cc640779d7b34895660",
  "author": "ACE-Step",
  "cardData": {
    "library_name": "transformers",
    "license": "mit",
    "tags": [
      "music",
      "audio"
    ]
  },
  "config": {
    "architectures": [
      "Qwen2_5OmniForConditionalGeneration"
    ],
    "chat_template_jinja": "{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}",
    "model_type": "qwen2_5_omni",
    "tokenizer_config": {
      "bos_token": null,
      "eos_token": "<|im_end|>",
      "pad_token": "<|endoftext|>",
      "unk_token": null
    }
  },
  "createdAt": "2026-01-23T10:26:14.000Z",
  "disabled": false,
  "downloads": 1792,
  "gated": false,
  "id": "ACE-Step/acestep-captioner",
  "lastModified": "2026-02-03T06:33:55.000Z",
  "library_name": "transformers",
  "likes": 74,
  "model-index": null,
  "modelId": "ACE-Step/acestep-captioner",
  "pipeline_tag": "text-to-audio",
  "private": false,
  "safetensors": {
    "parameters": {
      "BF16": 10283174144,
      "F32": 449051296
    },
    "total": 10732225408
  },
  "sha": "7109e9af1c6cfe3371730310b3a5cb773145b11e",
  "siblings": [
    {
      "rfilename": ".gitattributes"
    },
    {
      "rfilename": "README.md"
    },
    {
      "rfilename": "added_tokens.json"
    },
    {
      "rfilename": "args.json"
    },
    {
      "rfilename": "chat_template.jinja"
    },
    {
      "rfilename": "config.json"
    },
    {
      "rfilename": "file_list.txt"
    },
    {
      "rfilename": "generation_config.json"
    },
    {
      "rfilename": "merges.txt"
    },
    {
      "rfilename": "model-00001-of-00005.safetensors"
    },
    {
      "rfilename": "model-00002-of-00005.safetensors"
    },
    {
      "rfilename": "model-00003-of-00005.safetensors"
    },
    {
      "rfilename": "model-00004-of-00005.safetensors"
    },
    {
      "rfilename": "model-00005-of-00005.safetensors"
    },
    {
      "rfilename": "model.safetensors.index.json"
    },
    {
      "rfilename": "preprocessor_config.json"
    },
    {
      "rfilename": "special_tokens_map.json"
    },
    {
      "rfilename": "spk_dict.pt"
    },
    {
      "rfilename": "tokenizer.json"
    },
    {
      "rfilename": "tokenizer_config.json"
    },
    {
      "rfilename": "video_preprocessor_config.json"
    },
    {
      "rfilename": "vocab.json"
    }
  ],
  "spaces": [
    "DandyDonUnhinged/spock-body-lora-training"
  ],
  "tags": [
    "transformers",
    "safetensors",
    "qwen2_5_omni",
    "text-to-audio",
    "music",
    "audio",
    "arxiv:2602.00744",
    "license:mit",
    "endpoints_compatible",
    "region:us"
  ],
  "transformersInfo": {
    "auto_model": "AutoModelForMultimodalLM",
    "pipeline_tag": "text-to-audio",
    "processor": "AutoProcessor"
  },
  "usedStorage": 22374542766
}
tasktext-to-audio
receipt
Source
Hugging Face models
Its words
text-to-audio
Read by
field:pipeline_tag
Said since
2026-10-02 12:02 UTC
Last answered
2026-10-04 18:21 UTC
Original
open at the source
What the source handed over
{
  "_asked": "ACE-Step/acestep-captioner",
  "_id": "69734cc640779d7b34895660",
  "author": "ACE-Step",
  "cardData": {
    "library_name": "transformers",
    "license": "mit",
    "tags": [
      "music",
      "audio"
    ]
  },
  "config": {
    "architectures": [
      "Qwen2_5OmniForConditionalGeneration"
    ],
    "chat_template_jinja": "{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}",
    "model_type": "qwen2_5_omni",
    "tokenizer_config": {
      "bos_token": null,
      "eos_token": "<|im_end|>",
      "pad_token": "<|endoftext|>",
      "unk_token": null
    }
  },
  "createdAt": "2026-01-23T10:26:14.000Z",
  "disabled": false,
  "downloads": 1792,
  "gated": false,
  "id": "ACE-Step/acestep-captioner",
  "lastModified": "2026-02-03T06:33:55.000Z",
  "library_name": "transformers",
  "likes": 74,
  "model-index": null,
  "modelId": "ACE-Step/acestep-captioner",
  "pipeline_tag": "text-to-audio",
  "private": false,
  "safetensors": {
    "parameters": {
      "BF16": 10283174144,
      "F32": 449051296
    },
    "total": 10732225408
  },
  "sha": "7109e9af1c6cfe3371730310b3a5cb773145b11e",
  "siblings": [
    {
      "rfilename": ".gitattributes"
    },
    {
      "rfilename": "README.md"
    },
    {
      "rfilename": "added_tokens.json"
    },
    {
      "rfilename": "args.json"
    },
    {
      "rfilename": "chat_template.jinja"
    },
    {
      "rfilename": "config.json"
    },
    {
      "rfilename": "file_list.txt"
    },
    {
      "rfilename": "generation_config.json"
    },
    {
      "rfilename": "merges.txt"
    },
    {
      "rfilename": "model-00001-of-00005.safetensors"
    },
    {
      "rfilename": "model-00002-of-00005.safetensors"
    },
    {
      "rfilename": "model-00003-of-00005.safetensors"
    },
    {
      "rfilename": "model-00004-of-00005.safetensors"
    },
    {
      "rfilename": "model-00005-of-00005.safetensors"
    },
    {
      "rfilename": "model.safetensors.index.json"
    },
    {
      "rfilename": "preprocessor_config.json"
    },
    {
      "rfilename": "special_tokens_map.json"
    },
    {
      "rfilename": "spk_dict.pt"
    },
    {
      "rfilename": "tokenizer.json"
    },
    {
      "rfilename": "tokenizer_config.json"
    },
    {
      "rfilename": "video_preprocessor_config.json"
    },
    {
      "rfilename": "vocab.json"
    }
  ],
  "spaces": [
    "DandyDonUnhinged/spock-body-lora-training"
  ],
  "tags": [
    "transformers",
    "safetensors",
    "qwen2_5_omni",
    "text-to-audio",
    "music",
    "audio",
    "arxiv:2602.00744",
    "license:mit",
    "endpoints_compatible",
    "region:us"
  ],
  "transformersInfo": {
    "auto_model": "AutoModelForMultimodalLM",
    "pipeline_tag": "text-to-audio",
    "processor": "AutoProcessor"
  },
  "usedStorage": 22374542766
}

Text

This source's terms allow its title, its values and a link here, not its text. It is at https://huggingface.co/ACE-Step/acestep-captioner.