diff --git a/doc/reflections/das2rst.das b/doc/reflections/das2rst.das index 352b06bcca..2282b2f283 100644 --- a/doc/reflections/das2rst.das +++ b/doc/reflections/das2rst.das @@ -325,7 +325,7 @@ def document_module_openai(_root : string) { group_by_regex("Audio (TTS and STT)", mod, %regex~(speech|speak|transcribe|translate)$%%), group_by_regex("Moderations", mod, %regex~(moderations)$%%), group_by_regex("Image generation", mod, %regex~(generate_image)$%%), - group_by_regex("Vision", mod, %regex~(chat_vision|vision_request_body|image_file_data_uri)$%%) + group_by_regex("Vision", mod, %regex~(chat_vision|chat_audio|vision_request_body|audio_request_body|image_file_data_uri|file_base64)$%%) ) documents("OpenAI-compatible API client (chat, embeddings, audio, vision, ...)", mod, "openai.rst", groups) } @@ -343,13 +343,13 @@ def document_module_dasllama(_root : string) { var mod = find_module("dasllama") var groups <- array( group_by_regex("Model loading and sessions", mod, %regex~(load_model|create_session|create_kv_pool|release_kv_pages|create_batch_workspace|setup_dasllama_jobque|with_dasllama_jobque|caps|mtp_drafter_sidecar|attach_mtp_drafter|mtp_capable)$%%), - group_by_regex("Prefix cache", mod, %regex~(create_prefix_cache|prefix_attach|prefix_insert|prefix_checkpoint_at|prefix_release|prefix_held_groups|prefix_chain_list)$%%), + group_by_regex("Prefix cache", mod, %regex~(create_prefix_cache|prefix_attach|prefix_insert|prefix_checkpoint_at|prefix_match_len|prefix_release|prefix_held_groups|prefix_chain_list)$%%), group_by_regex("Tokenizer", mod, %regex~(encode|decode|piece)$%%), - group_by_regex("Evaluation and sampling", mod, %regex~(eval|eval_embd|eval_embd_span|eval_embd_span_mrope|eval_batch|sample|set_seed|stats)$%%), + group_by_regex("Evaluation and sampling", mod, %regex~(eval|eval_embd|eval_embd_span|eval_embd_span_mrope|eval_embd_body|media_body_rows|eval_batch|sample|set_seed|stats)$%%), group_by_regex("Generation", mod, %regex~(generate|generate_embd)$%%), group_by_regex("Embeddings", mod, %regex~(embed)$%%), group_by_regex("Vision and audio encoders", mod, %regex~(encode_image|encode_audio)$%%), - group_by_regex("Chat", mod, %regex~(create_chat|create_chat_renderer|add_user|add_user_audio|add_user_image|add_user_image_rows|add_user_audio_rows|add_assistant|render_turn|render_turn_marked|render_turn_image|render_turn_audio|render_assistant|render_close|respond|set_thinking)$%%), + group_by_regex("Chat", mod, %regex~(create_chat|create_chat_renderer|add_user|add_user_audio|add_user_image|add_user_image_rows|add_user_audio_rows|add_user_span|set_chat_date|add_assistant|render_turn|render_turn_marked|render_turn_image|render_turn_audio|render_assistant|render_close|respond|set_thinking)$%%), group_by_regex("Tool calling", mod, %regex~(set_tools|add_tool_results|render_assistant_calls|parse_calls)$%%), group_by_regex("Reasoning (thinking models)", mod, %regex~(split_reasoning|make_think_stream|think_feed|think_finish|think_drain|effective_stop_ids|turn_stop_ids|make_nothink_guard|nothink_stop_here)$%%), group_by_regex("Operations: prepared images and dispatch", mod, %regex~(dlim_inventory|dlim_clean|set_dispatch_worker_limit|get_dispatch_worker_limit|set_jobque_spin_us|get_jobque_spin_us|set_jobque_spin_gpu_us|get_jobque_spin_gpu_us|get_jobque_spin_in_force|set_single_thread|get_single_thread|select_matmul_backend_for_load|kernel_backend_available)$%%), @@ -402,7 +402,7 @@ def document_module_dasllama(_root : string) { dasllama_type_stanza(f, "struct-dasllama_vision_embedder-VisionState", "VisionState", "Caller-owned scratch for the embedder forward of whichever family is carried: the buffers ``encode_image`` reuses across calls. One per embedder user; holds no image state between calls.") dasllama_type_stanza(f, "struct-dasllama_audio_embedder-AudioEmbedder", "AudioEmbedder", - "A loaded audio encoder of whatever family the mmproj GGUF turned out to be (gemma4a — the gemma-4 E-series Conformer), as produced by ``load_audio_embedder``, which probes the file; ``audio_probe_proj_dim`` answers 0 where absence is an answer. ``audio_proj_dim`` must match the decoder's embedding width. The server's media worker owns one per armed slot.") + "A loaded audio encoder of whatever family the mmproj GGUF turned out to be (gemma4a — the gemma-4 E-series Conformer; or a whisper-class tower that ``load_audio_tower`` also serves — qwen2-audio, qwen2.5-omni, ultravox, voxtral), as produced by ``load_audio_embedder``, which probes the file; ``audio_probe_proj_dim`` answers 0 where absence is an answer. ``audio_proj_dim`` must match the decoder's embedding width. The server's media worker owns one per armed slot.") dasllama_type_stanza(f, "struct-dasllama_audio_embedder-AudioState", "AudioState", "Caller-owned scratch for the audio encoder forward of whichever family is carried: the buffers ``encode_audio`` reuses across calls. One per embedder user; holds no clip state between calls.") dasllama_type_stanza(f, "struct-dasllama_image-DlimImageInfo", "DlimImageInfo", diff --git a/doc/source/reference/tutorials/dasLLAMA_02_chat.rst b/doc/source/reference/tutorials/dasLLAMA_02_chat.rst index 80bc15f8d4..83ee3022f5 100644 --- a/doc/source/reference/tutorials/dasLLAMA_02_chat.rst +++ b/doc/source/reference/tutorials/dasLLAMA_02_chat.rst @@ -116,6 +116,26 @@ memory spent only when the stream actually runs: var toks : array render_assistant(m, rchat, "Paris.", toks) // the exchange, as tokens +A dated system turn +=================== + +The Llama-3.1+ template writes two lines into every system turn: the model's +knowledge cutoff and today's date. Today's date changes the prompt every +midnight. A test that compares token streams, or a benchmark with a fixed +prompt, pins the date with ``set_chat_date``; ``""`` gives the clock back: + +.. code-block:: das + + set_chat_date("26 Jul 2024") + var dated = create_chat_renderer(m, SYSTEM) + add_user(dated, "What day is it?") + print(decode(m, render_turn(m, dated))) + set_chat_date("") + +On Llama-3.2-1B-Instruct the system turn then reads +``Cutting Knowledge Date: December 2023`` and ``Today Date: 26 Jul 2024``. +A template that states no date - ChatML on SmolLM2, gemma - ignores the pin. + .. seealso:: Full source: :download:`tutorials/dasLLAMA/02_chat.das <../../../../tutorials/dasLLAMA/02_chat.das>` diff --git a/doc/source/reference/tutorials/dasLLAMA_08_audio_chat.rst b/doc/source/reference/tutorials/dasLLAMA_08_audio_chat.rst index 563fdf67cf..fed7382042 100644 --- a/doc/source/reference/tutorials/dasLLAMA_08_audio_chat.rst +++ b/doc/source/reference/tutorials/dasLLAMA_08_audio_chat.rst @@ -15,9 +15,10 @@ decoder reads inline with text. Supported pairs (decoder + mmproj GGUF): Qwen2-Audio, Qwen2.5-Omni (audio side), Ultravox v0.5 (over *stock* Llama-3 decoders), and Voxtral-Mini. The chat template picks the audio framing automatically — the code below is identical for every pair. Qwen3-Omni and -Gemma-4 E-series audio are served too, but through the ASR surface -(:ref:`tutorial 07 `'s two-path -``load_asr_model``), not ``load_audio_tower``. +Gemma-4 E-series audio are served too, but not by ``load_audio_tower``: +:ref:`tutorial 07 `'s two-path +``load_asr_model`` transcribes them, and a Gemma-4 E-series pair chats on this +page through the ``AudioEmbedder`` carrier rail at the end. Run:: @@ -105,34 +106,84 @@ encoder's rows between them. Unlike an image span, audio rows stay *causal*: sound has a left-to-right order. The rows themselves come from the ``AudioEmbedder`` carrier — the -family-neutral encoder a scheduler owns. Probe the mmproj with -``audio_probe_proj_dim`` (0 means no carrier-served audio tower), load it with -``load_audio_embedder``, and ``encode_audio`` turns 16 kHz PCM into the -soft-token rows that splice between the two spans — ``encode_image``'s audio -twin, and exactly what the server's media worker does per clip. The tutorial -probes the mmproj first: a carrier-served file (the gemma-4 E-series) takes the -carrier rail — ``section_render_spans`` plus ``section_carrier_encode`` — while -every ``load_audio_tower`` pair takes the chat rail above. +family-neutral encoder a scheduler owns. It serves every audio mmproj on this +page: the ``load_audio_tower`` families and the gemma-4 E-series alike. Probe +the mmproj with ``audio_probe_proj_dim`` (0 means no audio family serves the +file), load it with ``load_audio_embedder``, and ``encode_audio`` turns 16 kHz +PCM into the soft-token rows that splice between the two spans — +``encode_image``'s audio twin, and exactly what the server's media worker does +per clip. The tutorial asks a second question first: +``audio_tower_probe_proj_dim`` answers non-zero for a file ``load_audio_tower`` +serves. Such a pair runs every section, the carrier ones last. A gemma-4 +E-series file answers 0 there, so it runs only the carrier rail — +``section_render_spans`` plus ``section_carrier_encode``. The carrier rail closes the loop at the chat layer with the pre-encoded-rows seam, ``add_user_image_rows``'s audio twin: ``add_user_audio_rows`` moves the encoder's rows onto a *plain* chat — no tower attached — and ``respond`` runs the spliced turn, the audio span rendered around the rows. That is how a -carrier-served family hears in a conversation at all, and the path for a +gemma-4 E-series pair hears in a conversation at all, and the path for a scheduler that owns its own encoder; the rows are ``dim``-wide on every family -and the call length-checks them: +and the call length-checks them. ``audio_span_bare(e)`` says whether the span +takes markers: an ultravox span sits bare in a stock Llama template, which +knows no audio marker, so pass it as ``bare``: -.. das-doc: given var rows : array; let n = 0l +.. das-doc: given var rows : array; let n = 0l; var e = AudioEmbedder() .. code-block:: das var chat <- create_chat(m, "", 96l) - add_user_audio_rows(m, chat, rows, n) // moves the rows in + add_user_audio_rows(m, chat, rows, n, audio_span_bare(e)) // moves the rows in add_user(chat, "What did you hear?") respond(m, chat, SamplingParams()) $(piece) { print("{piece}") return true } +The audio in its place: one body by hand +======================================== + +So far the audio led the turn. A user message often carries its media in the +middle: "Here is a recording.