diff --git a/clams/app/__init__.py b/clams/app/__init__.py
index 4445427..454f89b 100644
--- a/clams/app/__init__.py
+++ b/clams/app/__init__.py
@@ -711,6 +711,26 @@ class ClamsPromptableApp(ClamsApp):
'values grow the KV cache linearly and increase GPU memory '
'usage; reduce if VRAM is constrained.',
},
+ {
+ 'name': 'useReasoning', 'type': 'boolean', 'default': False,
+ 'description':
+ 'Request the model\'s reasoning ("thinking") mode. Off by '
+ 'default. Honored only by apps whose backing model has a '
+ 'distinct reasoning mode; apps without one ignore it. When '
+ 'honored and enabled, the reasoning trace is split from the '
+ 'answer and stored in the ``modelReasoningTrace`` property of '
+ 'the output ``TextDocument`` (kept out of the document text). '
+ 'Reasoning is markedly slower and far more token-hungry: the '
+ 'whole trace is generated before the answer and drawn from the '
+ 'same budget capped by ``maxNewTokens``, so raise '
+ '``maxNewTokens`` substantially (thousands of tokens, not '
+ 'hundreds) when enabling this, or the trace may consume the '
+ 'entire budget and the answer be truncated or empty. Small '
+ 'reasoning models (as a rule of thumb, roughly 4B parameters '
+ 'and under) are especially prone to non-terminating "thinking '
+ 'loops" that exhaust the budget without producing an answer; '
+ 'validate termination per model before relying on it.',
+ },
{
'name': 'temperature', 'type': 'number', 'default': 0.0,
'description':
@@ -1109,37 +1129,56 @@ def split_tagged_reasoning_trace(
text: str,
open_tag: str = '',
close_tag: str = '',
+ assume_open: bool = False,
) -> Tuple[str, Optional[str]]:
"""
Split a reasoning model's output into ``(answer, trace)`` for the
- inline XML-tag style: a single reasoning block delimited by
- ``open_tag`` / ``close_tag`` (default ```` / ````)
- embedded in the decoded text, e.g. as emitted by DeepSeek-R1 or
- Qwen3.
-
- It does NOT handle models that isolate reasoning in a separate
- token-delimited channel rather than an inline tag block (e.g.
- gpt-oss, Gemma 4); those need their own parsing in the app. Trace
- handling is the app's responsibility either way -- call this when
- the model fits the inline-tag style, or parse it yourself.
-
- ``answer`` is everything after the final ``close_tag`` (stripped);
- ``trace`` is the text between the tags, or ``None`` when no closed
- block is present -- so it is safe to call on non-reasoning output,
- which returns ``(text.strip(), None)``.
+ inline XML-tag style: a reasoning block delimited by ``open_tag`` /
+ ``close_tag`` (default ```` / ````), e.g. as emitted
+ by DeepSeek-R1 or Qwen3.5.
+
+ Prefilled opening tags: some chat templates inject the OPENING tag
+ into the prompt when reasoning is enabled (Qwen3.5 does this), so the
+ decoded output carries only the CLOSING tag -- the model never emits
+ the opener. Pass ``assume_open=True`` in that situation (typically
+ wired to the app's reasoning toggle) so the trace is still recovered.
+ The cases:
+
+ * close tag present: ``answer`` is the text after it; ``trace`` is the
+ text between the tags, or everything before the close tag when the
+ opening tag is absent (prefilled).
+ * close tag absent, ``assume_open=True``: the output is unterminated
+ reasoning (e.g. the trace overran ``maxNewTokens`` before closing) --
+ there is no answer, so returns ``('', text.strip())``.
+ * close tag absent, ``assume_open=False`` (default): treated as a
+ plain, non-reasoning answer -- returns ``(text.strip(), None)``.
+
+ Only the inline-tag style is handled; models that isolate reasoning in
+ a separate token-delimited channel (e.g. gpt-oss, Gemma) need their own
+ parsing in the app.
:param text: raw decoded model output.
:param open_tag: opening marker of the inline reasoning block.
:param close_tag: closing marker of the inline reasoning block.
+ :param assume_open: treat the text as the body of a reasoning block
+ whose opening tag was prefilled into the prompt, so a missing close
+ tag means unterminated reasoning (empty answer) rather than a plain
+ answer.
:return: ``(answer, trace_or_None)``.
"""
ci = text.rfind(close_tag)
if ci == -1:
- return text.strip(), None
+ return ('', text.strip()) if assume_open else (text.strip(), None)
answer = text[ci + len(close_tag):].strip()
oi = text.find(open_tag)
- trace = (text[oi + len(open_tag):ci].strip()
- if oi != -1 and oi < ci else None)
+ if oi != -1 and oi < ci:
+ trace = text[oi + len(open_tag):ci].strip()
+ elif assume_open:
+ # opening tag was prefilled into the prompt (not generated), so the
+ # trace is everything up to the close tag
+ trace = text[:ci].strip()
+ else:
+ trace = None
return answer, trace
class ClamsHFPromptableApp(ClamsPromptableApp):
"""
@@ -1511,10 +1550,13 @@ def build_template_kwargs(self, **generation_params) -> dict:
implementation returns ``{}``.
Override to inject model-specific chat-template controls without
- having to reimplement :py:meth:`generate`. Common cases:
+ having to reimplement :py:meth:`generate`. This is where an app
+ honors the SDK ``useReasoning`` parameter, mapping it onto the
+ backend's reasoning switch. Common cases:
- * ``{'enable_thinking': False}`` to run a reasoning ("thinking")
- model in non-thinking mode;
+ * ``{'enable_thinking': bool(generation_params.get('use_reasoning'))}``
+ to bind ``useReasoning`` to a reasoning ("thinking") model's
+ template switch;
* ``{'tools': [...]}`` or ``{'documents': [...]}`` for tool-use /
RAG chat templates.
diff --git a/clams/develop/templates/app/cli.py.template b/clams/develop/templates/app/cli.py.template
index e29d3be..06f6fc3 100644
--- a/clams/develop/templates/app/cli.py.template
+++ b/clams/develop/templates/app/cli.py.template
@@ -6,13 +6,17 @@ DO NOT CHANGE the name of the file
"""
import argparse
+import json
import sys
from contextlib import redirect_stdout
+import jsonschema
+
import app
import clams.app
from clams import AppMetadata
+from clams.envelop import EnvelopeError
def metadata_to_argparser(app_metadata: AppMetadata) -> argparse.ArgumentParser:
@@ -81,11 +85,24 @@ if __name__ == "__main__":
params[pname] = pvalue
else:
params[pname] = [pvalue]
- if args.OUT_MMIF_FILE.name == '':
- with redirect_stdout(sys.stderr):
+ # Mirror the HTTP server's error handling (see clams.restify): an invalid
+ # input is reported and exits non-zero, while an app-level failure is
+ # recorded as an error view instead of crashing with a raw traceback.
+ try:
+ if args.OUT_MMIF_FILE.name == '':
+ with redirect_stdout(sys.stderr):
+ out_mmif = clamsapp.annotate(in_data, **params)
+ else:
out_mmif = clamsapp.annotate(in_data, **params)
- else:
- out_mmif = clamsapp.annotate(in_data, **params)
+ except (jsonschema.exceptions.ValidationError, json.JSONDecodeError, EnvelopeError) as e:
+ detail = e.message if isinstance(e, jsonschema.exceptions.ValidationError) else str(e)
+ print(f"Invalid input data. See below for validation error.\n\n{detail}", file=sys.stderr)
+ sys.exit(1)
+ except Exception:
+ clamsapp.logger.exception("Error in annotation")
+ out_mmif = clamsapp.record_error(in_data, **params).serialize(pretty=True)
+ args.OUT_MMIF_FILE.write(out_mmif)
+ sys.exit(1)
args.OUT_MMIF_FILE.write(out_mmif)
else:
arg_parser.print_help()
diff --git a/documentation/app-baseclasses.rst b/documentation/app-baseclasses.rst
index 6ac5cfc..e140d40 100644
--- a/documentation/app-baseclasses.rst
+++ b/documentation/app-baseclasses.rst
@@ -174,6 +174,17 @@ from :class:`~clams.app.ClamsApp`. These names are reserved; see
- Maximum number of new tokens generated per inference call. Larger values
grow the KV cache linearly and add to GPU memory usage; reduce if VRAM
is constrained.
+ * - ``useReasoning``
+ - boolean
+ - ``false``
+ - no
+ - Request the model's reasoning ("thinking") mode. Off by default;
+ honored only by apps whose model has a distinct reasoning mode
+ (others ignore it). Much slower and more token-hungry -- the trace is
+ generated before the answer and drawn from the ``maxNewTokens`` budget,
+ and small models can loop without terminating. When honored, the trace
+ is stored in the ``modelReasoningTrace`` property of the output
+ ``TextDocument``. See :ref:`promptable-reasoning`.
* - ``temperature``
- number
- ``0.0``
@@ -320,6 +331,52 @@ inferences, final reply returned.
``turn-taking`` is the default because it costs a single inference call
and is the more common multi-element pattern.
+.. _promptable-reasoning:
+
+Reasoning traces (``useReasoning``)
+^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+
+Some models expose a distinct reasoning ("thinking") mode that emits an
+intermediate reasoning trace before the answer. ``useReasoning`` is the
+standard toggle for that mode. It is inert unless the app honors it, and
+honoring is a per-app choice with three wiring points:
+
+- map ``useReasoning`` onto the backend's reasoning switch (for HF chat
+ templates, typically ``enable_thinking``) by overriding
+ :meth:`~clams.app.ClamsHFPromptableApp.build_template_kwargs`;
+- split the trace from the answer with
+ :meth:`~clams.app.ClamsPromptableApp.split_tagged_reasoning_trace`
+ (inline ``...``-style traces only; channel-delimited
+ formats need app-specific parsing). Many chat templates prefill the
+ opening ```` into the prompt, so only the closing tag appears in
+ the output; pass ``assume_open=True`` (wired to the reasoning toggle) so
+ the trace is still recovered and a missing close tag is read as
+ unterminated reasoning (empty answer) rather than a plain answer;
+- pass the trace to
+ :meth:`~clams.app.ClamsPromptableApp.response_to_grounded_textdocument`
+ through its ``reasoning_trace`` argument, which stores it in the
+ ``modelReasoningTrace`` property of the produced ``TextDocument``,
+ separate from the document text.
+
+An app whose model has no reasoning mode leaves ``useReasoning`` unwired;
+callers may still set it and it has no effect, like ``topK`` under greedy
+decoding.
+
+Cost and failure modes
+""""""""""""""""""""""
+
+Reasoning is much slower and far more token-hungry than a direct answer:
+the entire trace is generated first, from the same budget capped by
+``maxNewTokens``. Enabling ``useReasoning`` without raising ``maxNewTokens``
+substantially (thousands of tokens, not hundreds) risks the trace consuming
+the whole budget, leaving the answer truncated or empty. Small reasoning
+models (as a rule of thumb, roughly 4B parameters and under) are especially
+prone to non-terminating "thinking loops" that never reach an answer; budget
+generously and validate termination per model. When the trace overruns the
+budget before closing, ``assume_open=True`` yields an empty answer plus the
+unterminated trace -- an app should surface that (e.g. a warning and an empty
+``TextDocument``) rather than emitting the raw reasoning as the answer.
+
Helpers
^^^^^^^
@@ -512,7 +569,10 @@ supplies:
* a default
:py:meth:`~clams.app.ClamsHFPromptableApp.build_gen_kwargs` that
maps the SDK promptable parameters to HF ``model.generate()``
- kwargs.
+ kwargs;
+* a :py:meth:`~clams.app.ClamsHFPromptableApp.build_template_kwargs`
+ hook for chat-template controls, the override point for honoring
+ ``useReasoning`` (see :ref:`promptable-reasoning`).
See each method's docstring for full details.
diff --git a/documentation/conf.py b/documentation/conf.py
index 437c394..23836ad 100644
--- a/documentation/conf.py
+++ b/documentation/conf.py
@@ -26,7 +26,9 @@
# -- Project information -----------------------------------------------------
-project = proj_root_dir.name
+# The canonical package name — NOT derived from the checkout directory,
+# which is `clone-repo` when the shared docs-publish workflow builds here.
+project = 'clams-python'
blob_base_url = f'https://github.com/clamsproject/{project}/blob'
copyright = f'{datetime.date.today().year}, Brandeis LLC'
author = 'Brandeis LLC'
diff --git a/documentation/target-versions.csv b/documentation/target-versions.csv
index a93d300..65b3e4f 100644
--- a/documentation/target-versions.csv
+++ b/documentation/target-versions.csv
@@ -1,4 +1,5 @@
"``clams-python`` version","``mmif-python`` version","Target MMIF Specification"
+`1.7.3 `__,`1.5.1 `__,`1.2.0 `__
`1.7.2 `__,`1.5.1 `__,`1.2.0 `__
`1.7.1 `__,`1.5.1 `__,`1.2.0 `__
`1.7.0 `__,`1.5.0 `__,`1.2.0 `__
diff --git a/pyproject.toml b/pyproject.toml
index b7f9fc1..2abe89a 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -19,7 +19,7 @@ classifiers = [
"Programming Language :: Python :: 3 :: Only",
]
dependencies = [
- "mmif-python==1.5.1",
+ "mmif-python==1.5.3",
"Flask>=2",
"Flask-RESTful>=0.3.9",
"gunicorn>=20",
diff --git a/tests/test_promptable.py b/tests/test_promptable.py
index c2f5ae0..f394fa7 100644
--- a/tests/test_promptable.py
+++ b/tests/test_promptable.py
@@ -446,6 +446,28 @@ def test_non_thinking_output_safe_to_call(self):
self.assertEqual(answer, 'plain caption')
self.assertIsNone(trace)
+ def test_assume_open_close_present_open_absent(self):
+ # Prefilled opening tag (e.g. Qwen3.5): only the close tag is emitted.
+ answer, trace = self.split(
+ 'reasoning bodyThe answer.', assume_open=True)
+ self.assertEqual(answer, 'The answer.')
+ self.assertEqual(trace, 'reasoning body')
+
+ def test_assume_open_no_close_is_unterminated_reasoning(self):
+ # Trace overran the budget before closing: empty answer, whole text is
+ # the unterminated trace.
+ raw = 'still thinking, never closed'
+ answer, trace = self.split(raw, assume_open=True)
+ self.assertEqual(answer, '')
+ self.assertEqual(trace, raw)
+
+ def test_open_absent_default_keeps_backward_compatible(self):
+ # Without assume_open, a lone close tag is not treated as a prefilled
+ # block: answer after the close, no trace.
+ answer, trace = self.split('reasoning bodyThe answer.')
+ self.assertEqual(answer, 'The answer.')
+ self.assertIsNone(trace)
+
# ---------------------------------------------------------------------------
# Transport-neutral parameter casting